From e02c1cc1192a3ec9cf7e4c55e6da970c9014e513 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Fri, 2 Oct 2026 18:50:14 +0000 Subject: [PATCH 1/2] refactor: call the operator IR a DAG, not a graph or tree QueryExpr and SummaryNode children are Rc-shared, so the IR is a DAG. Rename types, functions, locals, tests, comments and docs accordingly: - ExportDAG/NamedDAG/WorkloadDAG/SummaryDAG (was DagGraph/NamedGraph/ WorkloadGraph/SummaryDagGraph), ReferenceDAG, ResolveDAGError, SharedSubDAG(Strategy), share_common_sub_dags, *_sub_dag test names. - Export JSON keys: graph -> dag, post_graph -> post_dag, pre_asap_subgraph -> pre_asap_sub_dag; dag-viewer and fixtures follow. Exports written before this change no longer load in the viewer. Unchanged: BTreeMap/BTreeSet, DataFusion TreeNode, the vendored MetricsQL parser, git "working tree", `cargo tree`, URLs, and Mermaid keywords. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- .../src/accuracy/allocation.rs | 2 +- .../src/accuracy/composition.rs | 2 +- .../src/accuracy/reconciliation.rs | 46 +-- crates/asap-aware-mapping/src/cost_model.rs | 68 ++-- .../src/exact_composition.rs | 2 +- crates/asap-aware-mapping/src/explanation.rs | 42 +-- crates/asap-aware-mapping/src/grouping.rs | 2 +- crates/asap-aware-mapping/src/lib.rs | 12 +- .../src/maintained_population.rs | 4 +- crates/asap-aware-mapping/src/pass/major.rs | 4 +- crates/asap-aware-mapping/src/recurrence.rs | 44 +-- crates/asap-aware-mapping/src/replacement.rs | 299 ++++++++-------- crates/asap-aware-mapping/src/rewrite.rs | 16 +- crates/asap-aware-mapping/src/rollup.rs | 14 +- .../src/summary_maintenance_cost/evidence.rs | 2 +- .../src/summary_maintenance_dag_export.rs | 24 +- .../src/summary_maintenance_lifecycle.rs | 4 +- crates/asap-physical-operators/README.md | 4 +- .../src/physical_planner/candidates.rs | 2 +- .../src/physical_planner/compiled.rs | 14 +- .../src/physical_planner/mod.rs | 52 +-- .../src/physical_planner/precompute.rs | 13 +- .../src/physical_planner/promql_fallback.rs | 2 +- .../asap-physical-operators/src/plan/mod.rs | 2 +- .../src/runtime/batch_execution.rs | 28 +- .../src/runtime/tests.rs | 4 +- .../tests/current_series_heap.rs | 8 +- .../tests/deployment_computation.rs | 11 +- .../tests/physical_dag.rs | 17 +- .../tests/precompute_population.rs | 27 +- .../tests/promql_binary.rs | 32 +- .../tests/promql_fallback.rs | 6 +- .../tests/promql_values.rs | 56 +-- .../asap-physical-operators/tests/raw_scan.rs | 16 +- .../tests/summary_projection.rs | 4 +- .../tests/weighted_topk_binding.rs | 23 +- crates/devtools/src/bin/analyze_corpora.rs | 2 +- crates/devtools/src/bin/dag_export.rs | 232 ++++++------- crates/devtools/src/bin/sketch_coverage.rs | 2 +- crates/devtools/src/bin/variant_coverage.rs | 2 +- crates/frontend-metricsql/tests/lowering.rs | 8 +- crates/frontend-promql/src/error.rs | 14 +- crates/frontend-promql/src/lib.rs | 2 +- crates/frontend-promql/src/promql.rs | 38 +- .../tests/histogram_metadata.rs | 2 +- .../awesome_prometheus_alerts.rs | 4 +- .../tests/observability/o11y_bench_promql.rs | 2 +- .../tests/observability/promql_corpus.rs | 6 +- .../tests/promql_binding_regressions.rs | 4 +- .../tests/promql_conformance.rs | 22 +- .../tests/promql_equivalence.rs | 20 +- .../frontend-promql/tests/promql_lowering.rs | 36 +- .../tests/univmon_candidates.rs | 4 +- crates/frontend-sql/src/error.rs | 14 +- crates/frontend-sql/src/lib.rs | 4 +- .../src/sql/collection_planning.rs | 2 +- crates/frontend-sql/src/sql/expr.rs | 2 +- crates/frontend-sql/src/sql/mod.rs | 8 +- crates/frontend-sql/src/sql/types.rs | 2 +- .../synthetic_packet_trace.rs | 6 +- .../tests/maintained_population.rs | 8 +- crates/frontend-sql/tests/netflow/netflow.rs | 2 +- crates/frontend-sql/tests/sql_lowering.rs | 30 +- crates/integration-tests/src/lib.rs | 2 +- crates/integration-tests/tests/cse.rs | 16 +- .../tests/exact_composition.rs | 4 +- crates/integration-tests/tests/nested.rs | 6 +- .../tests/precompute_raw_samples.rs | 9 +- .../tests/promql_to_post_asap.rs | 8 +- crates/integration-tests/tests/schema.rs | 2 +- .../tests/sql_to_post_asap.rs | 8 +- .../summary_maintenance_lifecycle_e2e.rs | 6 +- crates/planner/tests/summary_sharing.rs | 4 +- crates/types/Cargo.toml | 2 +- crates/types/src/cost.rs | 4 +- crates/types/src/dag_export.rs | 324 +++++++++--------- crates/types/src/post_asap/cse.rs | 22 +- .../src/post_asap/execution_data_state.rs | 8 +- crates/types/src/post_asap/expr.rs | 6 +- crates/types/src/post_asap/guarantee.rs | 4 +- crates/types/src/post_asap/mod.rs | 2 +- crates/types/src/post_asap/post_asap_dag.rs | 2 +- crates/types/src/post_asap/sketch.rs | 2 +- crates/types/src/pre_asap/agg_intent.rs | 4 +- crates/types/src/pre_asap/canonicalize.rs | 12 +- .../types/src/pre_asap/column_resolution.rs | 2 +- crates/types/src/pre_asap/cse.rs | 118 +++---- crates/types/src/pre_asap/expr_ir.rs | 6 +- crates/types/src/pre_asap/mod.rs | 12 +- crates/types/src/pre_asap/query_expr.rs | 46 +-- crates/types/src/pre_asap/resolve.rs | 40 +-- crates/types/src/pre_asap/schema_resolver.rs | 54 +-- crates/types/tests/planner_vocabulary.rs | 4 +- .../architecture/parse-and-canonicalize.md | 2 +- .../architecture/physical-plan-integration.md | 2 +- docs/design_docs/concepts/post-asap-ir.md | 10 +- docs/design_docs/concepts/pre-asap-ir.md | 2 +- .../decisions/concat-unique-keys.md | 20 +- docs/design_docs/decisions/cse-cost-model.md | 30 +- .../physical-planning-and-deployment.md | 4 +- .../analytical-resource-cost.md | 2 +- .../end-to-end-accuracy-guarantees.md | 2 +- .../maintained-populations.md | 2 +- .../asap-aware-mapping/optimizations.md | 2 +- .../proposals/asapquery-rule-coverage.md | 4 +- .../proposals/decoupling_op_and_expr.md | 10 +- .../design_docs/proposals/operator-sharing.md | 36 +- .../proposals/univmon-frequency-summary.md | 4 +- .../asap-aware-mapping-architecture.md | 16 +- .../asap-aware-mapping-contracts.md | 12 +- .../develop_docs/extend-asap-aware-mapping.md | 18 +- docs/develop_docs/library-api.md | 22 +- .../metrics-observability-corpora.md | 2 +- docs/develop_docs/native-promql-inputs.md | 2 +- .../develop_docs/physical-compile-coverage.md | 4 +- .../planner-vocabulary-migration.md | 4 +- docs/develop_docs/pre-asap-ir.md | 10 +- docs/user_guide_docs/run-a-query.md | 6 +- old_docs/README.md | 14 +- old_docs/docs/design.md | 12 +- old_docs/docs/l1-query-language.md | 34 +- old_docs/docs/l2-intent-algebra.md | 30 +- old_docs/docs/l3-summary-bound-ir.md | 22 +- old_docs/docs/l4-physical-plan.md | 4 +- tools/dag-viewer/README.md | 16 +- tools/dag-viewer/dag.example.json | 8 +- tools/dag-viewer/index.html | 4 +- tools/dag-viewer/post_asap_fixture.json | 24 +- tools/dag-viewer/render.py | 48 +-- tools/dag-viewer/test_render.py | 76 ++-- tools/dag-viewer/viewer.js | 104 +++--- 132 files changed, 1379 insertions(+), 1372 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index fdfedd82f..3bad13863 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -2,7 +2,7 @@ - Work in a dedicated Git worktree unless the user explicitly says otherwise. - Create new worktrees from `origin/main` unless another base is specified. - Fetch the base first, and never disturb an existing dirty working tree. + Fetch the base first, and never disturb an existing dirty working DAG. - Keep communication and generated prose concise. - Prefer the minimally complex implementation that satisfies the requirement. - Do not introduce a conceptual layer, abstraction, or public interface without diff --git a/crates/asap-aware-mapping/src/accuracy/allocation.rs b/crates/asap-aware-mapping/src/accuracy/allocation.rs index e06761ec6..77849d9c8 100644 --- a/crates/asap-aware-mapping/src/accuracy/allocation.rs +++ b/crates/asap-aware-mapping/src/accuracy/allocation.rs @@ -23,7 +23,7 @@ pub struct AccuracyAllocation { impl AccuracyAllocation { /// The end-to-end budget left for everything below `layers[0]` — what - /// the inner subtree must satisfy as a whole (it re-splits internally). + /// the inner sub-DAG must satisfy as a whole (it re-splits internally). /// `None` for a single-layer allocation. pub fn inner_target(&self, shape: &CompositionShape) -> Option { let inner = &self.layers[1..]; diff --git a/crates/asap-aware-mapping/src/accuracy/composition.rs b/crates/asap-aware-mapping/src/accuracy/composition.rs index f3355b281..f42e8f158 100644 --- a/crates/asap-aware-mapping/src/accuracy/composition.rs +++ b/crates/asap-aware-mapping/src/accuracy/composition.rs @@ -25,7 +25,7 @@ impl DefaultAccuracyModel { /// `(1 + ε_total) = Π (1 + ε_i)` ⇒ for two factors /// `ε_in + ε_out + ε_in·ε_out`; written out as the sum of all - /// cross-products so the expression tree is exact for any input count. + /// cross-products so the expression DAG is exact for any input count. fn multiplicative( op: &CompositionOperator, inputs: &[ResultGuarantee], diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 46cbc06f5..9215d2182 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -3,20 +3,20 @@ //! //! ## The gap this closes //! -//! `asap_types::pre_asap::cse::share_common_subtrees` (pre-ASAP CSE) only -//! ever merges two subtrees that are *exactly* [`PartialEq`]-equal, +//! `asap_types::pre_asap::cse::share_common_sub_dags` (pre-ASAP CSE) only +//! ever merges two sub-DAGs that are *exactly* [`PartialEq`]-equal, //! including their [`AggIntent`]'s `accuracy: AccuracyTarget` field. Two //! otherwise-identical aggregates that differ *only* in how tight an //! accuracy bound they ask for — `quantile(0.99, x)` at `epsilon=0.01` for //! one consumer, the same `quantile(0.99, x)` at `epsilon=0.05` for //! another — are therefore never the same `Rc`, never collapse into one -//! [`crate::replacement::TargetSubDAGCandidates`], and [`crate::replacement::SharedSubtreeStrategy`] +//! [`crate::replacement::TargetSubDAGCandidates`], and [`crate::replacement::SharedSubDagStrategy`] //! never even gets a `TargetSubDAG` with `consumer_count >= 2` to propose //! sharing for. This crate would build two entirely independent sketches //! for what is conceptually one computation, even though a single sketch //! built to the tighter of the two bounds would answer both. //! -//! This module is **additive**, not a relaxation of `share_common_subtrees` +//! This module is **additive**, not a relaxation of `share_common_sub_dags` //! itself: `accuracy` still participates in exact structural equality //! everywhere else in this crate (correctness elsewhere — e.g. a downstream //! consumer that pattern-matches on a specific `AccuracyTarget` — depends on @@ -24,7 +24,7 @@ //! enough to share" that sits entirely inside the [`ReplacementStrategy`] //! extension point: one more candidate a [`crate::cost_model::CostModel`] //! may or may not prefer, never a forced rewrite and never a change to what -//! `share_common_subtrees` itself merges. +//! `share_common_sub_dags` itself merges. //! //! ## What counts as a "near-duplicate", and why //! @@ -39,7 +39,7 @@ //! intent has no `AccuracyTarget` to reconcile in the first place. //! 2. Same `reduction` (grouping), same `output_names`, and the same shared //! `child` (`Rc::ptr_eq`, or value-equal for two independently-built but -//! identical subtrees CSE conservatively declined to alias) — the same +//! identical sub-DAGs CSE conservatively declined to alias) — the same //! "identical everything else" bar [`crate::rollup::RollupStrategy`] and //! [`crate::topk_reuse::TopKLimitReuseStrategy`] already hold their own //! sibling-reuse candidates to. @@ -50,7 +50,7 @@ //! trivially "always tightest"). //! 5. The tighter candidate's own **output** schema carries a provable //! unique key (`Schema::has_unique_key`) — the exact legality gate -//! `share_common_subtrees` itself applies (see `cse.rs`'s "Legality" +//! `share_common_sub_dags` itself applies (see `cse.rs`'s "Legality" //! section) and [`crate::rollup::RollupStrategy::is_legal_rollup_source`] //! already reuses verbatim for the identical reason: a producer's output //! is only safely reusable across a second, independent consumer when @@ -120,12 +120,12 @@ //! tag), because it needs its own cost treatment in //! [`crate::cost_model::DefaultCostModel::estimate_cost`], not just its own //! label. Every other `Replacement::Rewrite` shape that reaches -//! `estimate_cost` (`SharedSubtreeStrategy`'s `CseRecompute`, `Rollup`'s and +//! `estimate_cost` (`SharedSubDagStrategy`'s `CseRecompute`, `Rollup`'s and //! `TopKLimitReuse`'s `LogicalRewrite`) really does rebuild `target` from a //! different source, so pricing it as "one `cse_recompute_cost` of `target` //! itself, per consumer" is the right shape of cost. This strategy's //! candidate never rebuilds `target` at all — it reads `rc` (the tighter -//! sibling), which — per this module's own safety argument — is a subtree +//! sibling), which — per this module's own safety argument — is a sub-DAG //! this crate is already going to build regardless of whether `target` //! reads from it too. Pricing it with the same "rebuild `target`, once per //! consumer" formula would charge it for work it never does, and — because @@ -137,7 +137,7 @@ //! pin against. `estimate_cost` instead prices this shape as a //! [`crate::cost_model::CostModel::cse_shared_maintenance_cost`] read //! against `rc`'s **own** bound summary — the same order-of-magnitude, -//! per-family cost `SharedSubtreeStrategy`'s own `CseShare` candidate is +//! per-family cost `SharedSubDagStrategy`'s own `CseShare` candidate is //! priced with, reflecting "one more reference into a structure that's //! already being maintained" rather than "build a whole new one." //! @@ -146,7 +146,7 @@ //! that sibling group propagate the uses through its selected implementation. //! Accuracy edges are directed strictly from looser to tighter budgets, so //! they cannot cycle among themselves; both near-duplicates also have the -//! same structural child, so adding the edge preserves the reference graph's +//! same structural child, so adding the edge preserves the reference DAG's //! parent-before-child topological ordering. use std::cmp::Ordering; @@ -301,7 +301,7 @@ impl AccuracyReconciliationStrategy { /// /// Also requires the candidate's own *output* schema to carry a provable /// unique key ([`Schema::has_unique_key`]) — the exact legality gate - /// `pre_asap::cse::share_common_subtrees` already applies to its own + /// `pre_asap::cse::share_common_sub_dags` already applies to its own /// sharing decisions, and [`crate::rollup::RollupStrategy`] already /// reuses verbatim for the identical reason (see that module's /// `is_legal_rollup_source` doc, point 4): a producer's output is only @@ -392,12 +392,12 @@ mod tests { use super::*; use crate::cost_model::{CostModel, DefaultCostModel}; use asap_types::post_asap::SketchAlgorithm; - use asap_types::pre_asap::cse::share_common_subtrees; + use asap_types::pre_asap::cse::share_common_sub_dags; use asap_types::pre_asap::query_expr::{GroupKeys, Source}; use asap_types::pre_asap::schema::{Column, ColumnId, DataType, Schema}; /// `[ts(0), value(1), job(2)]`. - /// A unique-keyed scan (`[ts]`) so `share_common_subtrees` is actually + /// A unique-keyed scan (`[ts]`) so `share_common_sub_dags` is actually /// willing to hoist it — see `Schema::has_unique_key`/`cse.rs`'s own /// "Legality" section: a producer with no provable unique key is always /// inserted fresh, never hoisted, regardless of structural equality. @@ -658,24 +658,24 @@ mod tests { assert!(!strategy.matches(&TargetSubDAG::new(&exact))); } - // ── exact structural equality / share_common_subtrees is unchanged ──── + // ── exact structural equality / share_common_sub_dags is unchanged ──── #[test] - fn share_common_subtrees_still_never_merges_differing_accuracy() { + fn share_common_sub_dags_still_never_merges_differing_accuracy() { // The additive guarantee this issue explicitly must not violate: // pre-ASAP CSE's own exact-equality merge stays exact. Two // aggregates differing only in `accuracy` must come back as two // distinct `Rc`s, not one shared `Rc` — AccuracyReconciliationStrategy // is the *only* place cross-accuracy sharing gets proposed, never - // `share_common_subtrees` itself. + // `share_common_sub_dags` itself. let scan = metric_scan(); let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); - let roots = share_common_subtrees(vec![("a", a), ("b", b)]); + let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!( !Rc::ptr_eq(&roots[0].1, &roots[1].1), - "share_common_subtrees must not merge aggregates with different AccuracyTarget" + "share_common_sub_dags must not merge aggregates with different AccuracyTarget" ); assert_ne!( roots[0].1, roots[1].1, @@ -696,14 +696,14 @@ mod tests { #[test] fn identical_accuracy_still_merges_via_ordinary_cse() { // Sanity check the fixture itself: truly identical aggregates - // (same accuracy too) still merge via share_common_subtrees's own + // (same accuracy too) still merge via share_common_sub_dags's own // exact equality — unrelated to this module, but pins the contrast // with the test above. let scan = metric_scan(); let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let roots = share_common_subtrees(vec![("a", a), ("b", b)]); + let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); } @@ -714,7 +714,7 @@ mod tests { // `by(vec![])` (global aggregation) reports no unique key // (`aggregate_output_schema`'s own `unique_keys = if by.is_empty() .. // { vec![] } ..`) — the same legality gate - // `share_common_subtrees`/`RollupStrategy` apply, which this + // `share_common_sub_dags`/`RollupStrategy` apply, which this // strategy must not bypass (module docs, point 5). let scan = metric_scan(); let tight = global_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); @@ -848,7 +848,7 @@ mod tests { // independently-built but structurally identical loose queries // merge onto one Rc via ordinary CSE), *and* a separate, // single-consumer tight sibling exists over the same input — the - // scenario the issue itself targets: `SharedSubtreeStrategy`'s own + // scenario the issue itself targets: `SharedSubDagStrategy`'s own // CseShare/CseRecompute pair is on the table for the loose target's // own 2 consumers at the same time as this strategy's "read the // tight sibling instead" candidate. diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index 0bf7c0d80..27b33fba1 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -35,8 +35,8 @@ //! ## CSE sharing (issue #237, #223 stage 4) //! //! [`CseCandidate`]/[`ShareDecision`]/[`CostModel::cse_share_decision`] below -//! decide whether a CSE-detected shared subtree -//! ([`asap_types::pre_asap::cse::share_common_subtrees`], issue #223 stages +//! decide whether a CSE-detected shared sub-DAG +//! ([`asap_types::pre_asap::cse::share_common_sub_dags`], issue #223 stages //! 1-2, PR #235) is actually worth sharing, via a real Volcano/Cascades-style //! cost comparison rather than a fixed rule. See //! `docs/design_docs/cse-cost-model-decision.md` for the full design discussion (why @@ -266,24 +266,24 @@ fn finite_rate(units_per_second: f64) -> Option { .then_some(CostRate(units_per_second)) } -/// A CSE-detected, legality-gated shared subtree with two or more consumers +/// A CSE-detected, legality-gated shared sub-DAG with two or more consumers /// — the unit [`CostModel::cse_share_decision`] decides over. Built by /// [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted) /// (via [`crate::replacement`]'s own `cse_preference`) the first time it -/// needs a representative bound node for a subtree that -/// [`asap_types::pre_asap::cse::share_common_subtrees`] already collapsed +/// needs a representative bound node for a sub-DAG that +/// [`asap_types::pre_asap::cse::share_common_sub_dags`] already collapsed /// onto one `Rc` for two or more workload roots. See /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { - /// The shared pre-ASAP subtree itself. - pub subtree: &'a QueryExpr, - /// The `SummaryNode` this subtree bound to — gives the cost model the + /// The shared pre-ASAP sub-DAG itself. + pub sub_dag: &'a QueryExpr, + /// The `SummaryNode` this sub-DAG bound to — gives the cost model the /// concrete `SummaryFamilyType`/`(kind, params)` actually at stake, not /// just the pre-ASAP shape. pub bound_summary: &'a SummaryNode, - /// How many workload roots reference this exact shared subtree, counted + /// How many workload roots reference this exact shared sub-DAG, counted /// once up front over the whole workload (always >= 2 — a candidate is - /// only ever constructed for an actually-shared subtree). + /// only ever constructed for an actually-shared sub-DAG). pub consumer_count: usize, } @@ -367,23 +367,23 @@ pub enum ShareDecision { } /// Default [`CostModel::cse_recompute_cost`]: a structural-size proxy — the -/// number of *unique* nodes in `subtree`'s DAG +/// number of *unique* nodes in `sub_dag`'s DAG /// ([`asap_types::pre_asap::cse::dag_node_count`], the same module this /// candidate's sharing was detected in). Deliberately **not** a raw -/// `serde_json` serialization length: after CSE, `subtree` is generally a -/// DAG, not a tree (a `CseCandidate` only exists because something got +/// `serde_json` serialization length: after CSE, `sub_dag` generally has +/// internal sharing (a `CseCandidate` only exists because something got /// shared), and a naive full serialization re-serializes — over-counts — -/// any descendant `subtree` already shares internally, once per parent +/// any descendant `sub_dag` already shares internally, once per parent /// that references it, instead of once for the whole DAG. `dag_node_count` /// dedupes by `Rc` pointer identity, so it charges each unique node's /// contribution exactly once regardless of how many places within -/// `subtree` reference it. Cheap to compute (one pass, no serialization), +/// `sub_dag` reference it. Cheap to compute (one pass, no serialization), /// and still scales with real structural complexity — a genuinely tiny -/// leaf costs little to recompute, a deep multi-join subtree costs a lot. +/// leaf costs little to recompute, a deep multi-join sub-DAG costs a lot. /// A deployment with real per-row/per-update cost knowledge should /// override [`CostModel::cse_recompute_cost`] instead of relying on this. -pub fn default_cse_recompute_cost(subtree: &QueryExpr) -> Cost { - Cost(asap_types::pre_asap::cse::dag_node_count(subtree) as f64) +pub fn default_cse_recompute_cost(sub_dag: &QueryExpr) -> Cost { + Cost(asap_types::pre_asap::cse::dag_node_count(sub_dag) as f64) } /// Default [`CostModel::cse_shared_maintenance_cost`]: a small @@ -567,12 +567,12 @@ pub trait CostModel { ) } - /// Estimate the one-time cost of recomputing `candidate.subtree` + /// Estimate the one-time cost of recomputing `candidate.sub_dag` /// independently at a single use site. Default: /// [`default_cse_recompute_cost`] (a structural-size proxy). See /// `docs/design_docs/cse-cost-model-decision.md`. fn cse_recompute_cost(&self, candidate: &CseCandidate) -> Cost { - default_cse_recompute_cost(candidate.subtree) + default_cse_recompute_cost(candidate.sub_dag) } /// Estimate the cost of maintaining `candidate.bound_summary` as one @@ -666,7 +666,7 @@ pub trait CostModel { Cost(1.0) } - /// Cost of recomputing `candidate.subtree` once, from the pre-ASAP/raw + /// Cost of recomputing `candidate.sub_dag` once, from the pre-ASAP/raw /// path. Units: cost units per recomputation — the `raw_recompute_cost` /// term of `recompute_cost_rate`. Default: delegates to /// [`cse_recompute_cost`](Self::cse_recompute_cost) (the same @@ -740,14 +740,14 @@ pub trait CostModel { /// already belongs to [`rank_candidates`](Self::rank_candidates) (for a /// [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) /// group) and [`cse_share_decision`](Self::cse_share_decision) (for a - /// [`SharedSubtreeStrategy`](crate::replacement::SharedSubtreeStrategy) + /// [`SharedSubDagStrategy`](crate::replacement::SharedSubDagStrategy) /// group). /// /// One method covers both candidate shapes this crate ships: /// `candidate.replacement`'s [`Replacement::Summary`] arm (a /// `SketchAlgorithmStrategy` candidate — the bound `SummaryNode` is right /// there, nothing to reconstruct) and its [`Replacement::Rewrite`] arm - /// (a `SharedSubtreeStrategy` share-vs-recompute candidate — no bound + /// (a `SharedSubDagStrategy` share-vs-recompute candidate — no bound /// `SummaryNode` of its own, since sharing is a decision about a target /// already bound some other way; a representative binding is recovered /// from `target` itself). `target` is threaded through explicitly @@ -1044,7 +1044,7 @@ impl CostModel for DefaultCostModel { /// `cse_share_decision` already compares against each other. `NaN` /// only if `target` itself can't be bound at all (schema derivation /// failed) — never expected for a target that's already part of a - /// legitimate workload tree. + /// legitimate workload DAG. /// /// **Exception**: a [`ReplacementProvenance::AccuracyReconciliation`] /// candidate (issue #273) never rebuilds `target` — it reads a @@ -1062,7 +1062,7 @@ impl CostModel for DefaultCostModel { match &candidate.replacement { Replacement::Summary(node) => { let cse = CseCandidate { - subtree: target.root, + sub_dag: target.root, bound_summary: node, consumer_count, }; @@ -1075,7 +1075,7 @@ impl CostModel for DefaultCostModel { return f64::NAN; }; let cse = CseCandidate { - subtree: rc, + sub_dag: rc, bound_summary: &sibling_bound, // One additional reference into `rc`'s own (already // necessary) build, from this one consumer's @@ -1092,7 +1092,7 @@ impl CostModel for DefaultCostModel { return f64::NAN; }; let cse = CseCandidate { - subtree: target.root, + sub_dag: target.root, bound_summary: &bound, consumer_count, }; @@ -1410,11 +1410,11 @@ mod tests { assert!(default_cse_recompute_cost(&nested) > default_cse_recompute_cost(&leaf)); } - /// The DAG-awareness this proxy exists for: a subtree that internally + /// The DAG-awareness this proxy exists for: a sub-DAG that internally /// re-references one shared descendant (e.g. after single-query CSE, /// `x op x` collapsing both branches onto one `Rc`) must cost the same /// as if that descendant only appeared once — not double, the way a - /// naive tree-shaped size measure (a full serialization, or an + /// naive per-path size measure (a full serialization, or an /// identity-blind recursive walk) would count it. #[test] fn default_recompute_cost_does_not_double_count_an_internally_shared_descendant() { @@ -1448,7 +1448,7 @@ mod tests { default_cse_recompute_cost(&with_sharing), Cost(2.0), "internal sharing: Join + 1 shared Scan (referenced twice) = \ - 2 unique nodes, not 3 — a tree-shaped size measure would \ + 2 unique nodes, not 3 — a per-path size measure would \ wrongly charge for the shared Scan twice" ); } @@ -1473,7 +1473,7 @@ mod tests { #[test] fn cse_share_decision_shares_when_recompute_dominates_maintenance() { let candidate = CseCandidate { - subtree: &scan(), + sub_dag: &scan(), bound_summary: &summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, @@ -1491,7 +1491,7 @@ mod tests { #[test] fn cse_share_decision_recomputes_when_maintenance_dominates_recompute() { let candidate = CseCandidate { - subtree: &scan(), + sub_dag: &scan(), bound_summary: &summary_node(SummaryFamilyType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { @@ -1531,7 +1531,7 @@ mod tests { // actually calls through to the overridable hooks rather than // hardcoding a comparison against its own defaults. let candidate = CseCandidate { - subtree: &scan(), + sub_dag: &scan(), bound_summary: &summary_node(SummaryFamilyType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { @@ -1626,7 +1626,7 @@ mod tests { } /// `DefaultCostModel::estimate_cost` for a [`Replacement::Rewrite`] pair - /// (the `SharedSubtreeStrategy` share-vs-recompute shape) agrees with + /// (the `SharedSubDagStrategy` share-vs-recompute shape) agrees with /// what `cse_share_decision` would already pick for the same target: with /// many consumers of a cheap-to-recompute leaf, the "share" candidate /// (the target's own `Rc`) must cost less than the "recompute diff --git a/crates/asap-aware-mapping/src/exact_composition.rs b/crates/asap-aware-mapping/src/exact_composition.rs index a4508dc3b..0a683550c 100644 --- a/crates/asap-aware-mapping/src/exact_composition.rs +++ b/crates/asap-aware-mapping/src/exact_composition.rs @@ -412,7 +412,7 @@ impl<'a> ExactCompositionStrategy<'a> { rationale: format!( "{} is an exact fold whose input is the readout of {} — a maintained \ accumulator cannot consume query-time values, so instead of collapsing \ - the whole tree into KeepPreAsap this applies the fold as an \ + the whole DAG into KeepPreAsap this applies the fold as an \ ExactRead over whichever summary readout global_selection \ commits for the child target (asap_aware_mapping::exact_composition)", describe_intent(&intent), diff --git a/crates/asap-aware-mapping/src/explanation.rs b/crates/asap-aware-mapping/src/explanation.rs index 5f930dbc4..f5de3e145 100644 --- a/crates/asap-aware-mapping/src/explanation.rs +++ b/crates/asap-aware-mapping/src/explanation.rs @@ -17,7 +17,7 @@ //! //! Earlier (PR #247, superseded by this module — see "What this replaces" //! below), "is optimization X applicable here?" was a yes/no fact each rule -//! re-derived by walking the tree itself. That made sense before there was +//! re-derived by walking the DAG itself. That made sense before there was //! any other structure to consult. But [`crate::replacement::search_workload`] //! (issue #252) now *already* computes, for every //! [`TargetSubDAG`](crate::replacement::TargetSubDAG) in the workload, every @@ -36,7 +36,7 @@ //! candidate isn't an *opportunity*, it's just the target's existing shape //! reflected back. A `TargetSubDAG` with more than one candidate (several //! sketch families to choose between), or one candidate that is itself a -//! genuine alternative to the status quo (share this already-shared subtree +//! genuine alternative to the status quo (share this already-shared sub-DAG //! instead of recomputing it at every consumer), *is* an applicability //! finding — [`explain_replacements`] and //! [`explain_replacements_with`] just translate [`CandidateLogicalASAPDAGs`]'s @@ -50,8 +50,8 @@ //! `realizations_for_intent` would have committed to on its own. //! - [`ExplanationKind::CommonSubexpressionReuse`] — the `TargetSubDAG` //! has two or more consumers *and* its candidate list contains the -//! [`SharedSubtreeStrategy`] "build once and share" candidate (the one -//! whose `Rc` is the group's own `target`) — i.e. sharing this subtree +//! [`SharedSubDagStrategy`] "build once and share" candidate (the one +//! whose `Rc` is the group's own `target`) — i.e. sharing this sub-DAG //! instead of recomputing it independently is a real, reported choice, not //! just an accident of how the workload happened to be built. //! @@ -76,7 +76,7 @@ //! [`crate::replacement::search_workload`]/[`crate::replacement::search_workload_with`]'s //! own signature shape. Only the *data source* changed: this module now //! calls those two functions and translates the result, rather than running -//! its own rules and their supporting traversal over the tree a second time. +//! its own rules and their supporting traversal over the DAG a second time. //! All of that old traversal is deleted, not kept alongside the new //! implementation — see "Two guarantees the old traversal made, re-verified" //! below for the two properties it's important that deletion didn't quietly @@ -178,7 +178,7 @@ //! [`Replacement::Summary`]: crate::replacement::Replacement::Summary //! [`Replacement::Rewrite`]: crate::replacement::Replacement::Rewrite //! [`SketchAlgorithmStrategy`]: crate::replacement::SketchAlgorithmStrategy -//! [`SharedSubtreeStrategy`]: crate::replacement::SharedSubtreeStrategy +//! [`SharedSubDagStrategy`]: crate::replacement::SharedSubDagStrategy //! [`CandidateLogicalASAPDAGs`]: crate::replacement::CandidateLogicalASAPDAGs //! [`TargetSubDAGCandidates`]: crate::replacement::TargetSubDAGCandidates @@ -213,7 +213,7 @@ pub enum ExplanationKind { /// have committed to on its own. SketchApproximation, /// A `TargetSubDAG` has two or more consumers *and* its candidate list - /// contains [`crate::replacement::SharedSubtreeStrategy`]'s "build once + /// contains [`crate::replacement::SharedSubDagStrategy`]'s "build once /// and share" candidate — the catalog's cross-statistic / cross-metrics / /// cross-subpopulation reuse entries, all the same underlying structural /// fact. @@ -222,7 +222,7 @@ pub enum ExplanationKind { /// [`Replacement::ExactComposition`] — /// [`crate::exact_composition::ExactCompositionStrategy`] found an exact /// operator that can be composed with a summary plan across an explicit - /// update/readout boundary instead of collapsing the whole tree into + /// update/readout boundary instead of collapsing the whole DAG into /// `KeepPreAsap` (issue #171). ExactComposition, } @@ -234,7 +234,7 @@ pub enum ExplanationKind { /// [`crate::replacement::ReplacementSubDAG::rationale`]). /// /// `node_hash` is [`structural_hash`](asap_types::pre_asap::cse::structural_hash) -/// of the `TargetSubDAG`'s own `target` subtree — the same function, on the +/// of the `TargetSubDAG`'s own `target` sub-DAG — the same function, on the /// same `Rc` shape, that [`asap_types::dag_export::DagNode::hash`] /// is computed with. A downstream consumer that independently exported the /// same `QueryExpr` (e.g. via `asap_types::dag_export::export`) can match @@ -455,7 +455,7 @@ fn visit( /// `node`'s own **relational-skeleton** operator children — the same scope /// `crate::replacement`'s own target-discovery `walk_children` (and -/// `asap_types::pre_asap::cse::share_common_subtrees`'s `rebuild_children`) +/// `asap_types::pre_asap::cse::share_common_sub_dags`'s `rebuild_children`) /// use. Exhaustive over every `QueryExpr` variant: a new variant fails to /// compile here until this match is extended too. fn visit_children( @@ -563,15 +563,15 @@ mod tests { } /// `node_hash` must be the literal `structural_hash` a downstream - /// consumer would compute over the *same* `QueryExpr` subtree via + /// consumer would compute over the *same* `QueryExpr` sub-DAG via /// `asap_types::dag_export::export` — the whole point of carrying it is - /// that two independent exports of the same tree agree, with no + /// that two independent exports of the same DAG agree, with no /// string-matching against `location` required. #[test] - fn node_hash_matches_dag_export_hash_for_the_same_subtree() { + fn node_hash_matches_dag_export_hash_for_the_same_sub_dag() { let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let graph = asap_types::dag_export::export(&q); - let expected_hash = graph.nodes[graph.root as usize].hash; + let dag = asap_types::dag_export::export(&q); + let expected_hash = dag.nodes[dag.root as usize].hash; let findings = explain_replacements(vec![("dashboard_p99", q)]); let sketch = findings @@ -582,7 +582,7 @@ mod tests { Some(sketch.node_hash), expected_hash, "ReplacementExplanation::node_hash must match dag_export's DagNode::hash \ - for the same QueryExpr subtree" + for the same QueryExpr sub-DAG" ); } @@ -637,7 +637,7 @@ mod tests { /// A sketch-applicable `Aggregate` reachable via two paths that CSE /// collapses onto one `Rc` — the same `median(x) == median(x)` shape - /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_subtree` + /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_sub_dag` /// test uses — must be reported once, not once per path: it is exactly /// one [`crate::replacement::TargetSubDAGCandidates`], keyed by `Rc` pointer identity, /// not one per path that reaches it. @@ -670,7 +670,7 @@ mod tests { #[test] fn two_roots_with_the_same_grouped_aggregate_share_a_reuse_finding() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema - // carries a provable unique key — share_common_subtrees's legality + // carries a provable unique key — share_common_sub_dags's legality // gate — and identical across both roots, so it is shareable. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); @@ -722,7 +722,7 @@ mod tests { #[test] fn ungrouped_identical_aggregates_are_not_shareable_so_no_finding() { - // Empty `by`: no provable unique key — share_common_subtrees never + // Empty `by`: no provable unique key — share_common_sub_dags never // hoists these, so consumer_count stays 1 for each and this module // must not report a finding either. let a = agg(vec![], AggIntent::Sum { col: None }, metric_scan(&["job"])); @@ -762,13 +762,13 @@ mod tests { /// A shared node nested three levels under two *different*, unshared /// `Filter` parents (mirrors `crate::replacement::tests:: - /// nested_shared_subtree_below_an_unshared_parent_is_still_discovered`) + /// nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered`) /// must still be exactly one finding — the maximal-`TargetSubDAG` /// guarantee the module docs describe, now provided by /// `crate::replacement`'s own target discovery rather than this module's /// (deleted) traversal. #[test] - fn a_deeply_shared_subtree_under_different_parents_is_reported_once() { + fn a_deeply_shared_sub_dag_under_different_parents_is_reported_once() { use asap_types::pre_asap::expr_ir::ScalarValue; use asap_types::pre_asap::query_expr::Predicate; diff --git a/crates/asap-aware-mapping/src/grouping.rs b/crates/asap-aware-mapping/src/grouping.rs index b2899f0d4..f74205205 100644 --- a/crates/asap-aware-mapping/src/grouping.rs +++ b/crates/asap-aware-mapping/src/grouping.rs @@ -43,7 +43,7 @@ //! whole-recursive-bind decision procedure toward a specific `SketchKind`, //! the same pattern [`crate::replacement::SketchAlgorithmStrategy`]'s own module //! docs explain was deliberately deleted from this crate as an anti-pattern: -//! forcing a choice via a whole-tree `CostModel` adapter had a real bug where +//! forcing a choice via a whole-DAG `CostModel` adapter had a real bug where //! the forced choice could leak into a target's own nested aggregates. This //! module never needs that: [`crate::replacement::realizations_for_intent`] //! already returns every ranked candidate `Realization` directly, so diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 3af72bf5f..b52fa1e6d 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -2,13 +2,13 @@ //! //! This crate sits between the language-agnostic IR ([`asap_ir`]) and //! any runtime: it consumes pre-ASAP [`QueryExpr`](asap_types::pre_asap::QueryExpr) -//! trees and makes the cost-aware decisions the pre-ASAP IR deliberately +//! DAGs and makes the cost-aware decisions the pre-ASAP IR deliberately //! leaves open — which sketch (if any) realises each approximate intent. //! //! **Common sub-expression elimination (CSE) is not this crate's job.** //! Detection is a primary pass over the pre-ASAP `QueryExpr` IR itself //! (`asap_types::pre_asap`, design tracked in issue #223), run before a -//! tree ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 +//! DAG ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 //! for why (batch query optimization needs to see shared work across a //! `QueryWorkload` before summary binding, not after). This crate may //! eventually run a second, narrower CSE pass of its own over an @@ -74,7 +74,7 @@ //! prose), meant for the same downstream consumer (e.g. a //! DAG-visualization view) the crate doc's planning workflows section above //! already names for [`replacement::CandidateLogicalASAPDAGs`] itself. Superseded PR -//! #247's own rule-based traversal, which re-walked the tree once per +//! #247's own rule-based traversal, which re-walked the DAG once per //! optimization before [`replacement::search_workload`] existed to read //! from instead — see that module's docs for the full reframing. //! - [`rollup`] — [`rollup::RollupStrategy`] wraps group-by-lattice roll-up @@ -83,7 +83,7 @@ //! re-deriving it from an already-computed, strictly finer sibling //! `Aggregate` over identical child IR instead of an independent pass //! over the raw source — the cross-aggregate sibling of -//! `pre_asap::cse::share_common_subtrees`'s identical-subtree sharing. +//! `pre_asap::cse::share_common_sub_dags`'s identical-sub-DAG sharing. //! [`rollup::is_legal_rollup_source`] is the standalone legality predicate //! other axes (e.g. issue #256's `GroupingStrategy`) are expected to //! consult directly, so it and this module's `RollupStrategy` can never @@ -101,7 +101,7 @@ //! is a [`replacement::ReplacementStrategy`] that reshapes a bare `avg` //! node — which [`replacement::realizations_for_intent`] can only //! dispatch to `Realization::PassThrough`, so it can never be a -//! [`replacement::SharedSubtreeStrategy`] target — into a `sum`/`count` +//! [`replacement::SharedSubDagStrategy`] target — into a `sum`/`count` //! pair under the same grouping, re-divided back by a wrapping `Project`, //! so those *are* ordinary mergeable accumulators sharing/sketching can //! reach. It only reshapes; [`replacement::search_workload`]'s cost-based @@ -211,7 +211,7 @@ pub use replacement::{ search_workload_with_targets, summary_candidates, CandidateLogicalASAPDAGs, CompositionDecision, GlobalSelection, Matcher, Proposals, RankedTargetSubDAGCandidates, Realization, RealizationError, RecurrenceProfileMap, RejectedCandidate, Replacement, - ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubtreeStrategy, + ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubDAGStrategy, SketchAlgorithmStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, MAX_SEARCH_ITERATIONS, }; diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 8d9460c14..61c88b0ad 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -312,7 +312,7 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { mod tests { use super::*; use crate::test_support::lower_promql; - use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_subtrees}; + use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_sub_dags}; fn lower(q: &str) -> Rc { Rc::new(lower_promql(q, asap_types::types::AccuracyTarget::Exact)) @@ -364,7 +364,7 @@ mod tests { .target_subdag_candidates() .flat_map(|g| &g.candidates) .any(|c| c.strategy == "MaintainedPopulationStrategy")); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() diff --git a/crates/asap-aware-mapping/src/pass/major.rs b/crates/asap-aware-mapping/src/pass/major.rs index d74091af7..3671ece58 100644 --- a/crates/asap-aware-mapping/src/pass/major.rs +++ b/crates/asap-aware-mapping/src/pass/major.rs @@ -8,7 +8,7 @@ use std::rc::Rc; -use asap_types::post_asap::{share_common_summary_subtrees, SummaryNode}; +use asap_types::post_asap::{share_common_summary_sub_dags, SummaryNode}; use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::types::AccuracyTarget; @@ -95,7 +95,7 @@ impl OptimizationPass for MajorPass { assembled.push(dag); } let interned = - share_common_summary_subtrees(assembled.iter().cloned().enumerate().collect()); + share_common_summary_sub_dags(assembled.iter().cloned().enumerate().collect()); let states: Vec<_> = interned .iter() .map(|(_, dag)| summary_states(dag)) diff --git a/crates/asap-aware-mapping/src/recurrence.rs b/crates/asap-aware-mapping/src/recurrence.rs index 48e2ba033..56cc67b44 100644 --- a/crates/asap-aware-mapping/src/recurrence.rs +++ b/crates/asap-aware-mapping/src/recurrence.rs @@ -6,7 +6,7 @@ //! neither reached [`CostModel`]'s CSE share-vs-recompute decision //! ([`CostModel::cse_share_decision`]): that decision only ever compared a //! *structural* consumer count (how many workload locations reference a -//! shared subtree) against a flat per-family maintenance weight — it had no +//! shared sub-DAG) against a flat per-family maintenance weight — it had no //! notion of how *often* those consumers actually run. //! //! This module adds that notion as a generic cost context, not a scheduler: @@ -836,13 +836,13 @@ mod tests { #[test] fn decide_falls_back_to_structural_decision_when_profile_is_empty() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1000, }; @@ -862,13 +862,13 @@ mod tests { #[test] fn decide_rejects_mixed_one_shot_and_repeating_without_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 2, }; @@ -881,13 +881,13 @@ mod tests { #[test] fn decide_accepts_mixed_one_shot_and_repeating_with_an_explicit_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 2, }; @@ -942,13 +942,13 @@ mod tests { /// it's read. #[test] fn high_frequency_selects_maintained_low_frequency_selects_recompute() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -998,13 +998,13 @@ mod tests { /// cost" acceptance criterion directly against the trait hooks. #[test] fn update_rate_only_affects_maintained_cost_evaluation_rate_affects_both() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1038,13 +1038,13 @@ mod tests { /// `raw_recompute_cost` is expensive. #[test] fn one_shot_only_consumer_decides_without_an_explicit_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1071,13 +1071,13 @@ mod tests { /// single one-shot consumer strictly prefers `RecomputeIndependently`. #[test] fn one_shot_only_single_consumer_does_not_unconditionally_prefer_share() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1098,13 +1098,13 @@ mod tests { /// own `DeterministicUnitCostModel`. #[test] fn batch_only_workload_does_not_unconditionally_prefer_share_under_default_cost_model() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1175,7 +1175,7 @@ mod tests { /// themselves structurally distinct (so they don't collapse into one /// root the way whole-root-identical fixtures do — see /// `shared_aggregate_across_two_roots_gets_both_strategies_candidates`'s - /// own doc) while letting `share_common_subtrees` unify their + /// own doc) while letting `share_common_sub_dags` unify their /// identical `sum_agg()` children onto one shared `Rc`. fn filtered_root(distinguishing_literal: i64) -> QueryExpr { QueryExpr::Filter { @@ -1366,7 +1366,7 @@ mod tests { assert!(matches!(err, RecurrenceError::InvalidUpdateRate(_))); } - /// Issue #287 review bug 2: a site no root's own structural tree + /// Issue #287 review bug 2: a site no root's own structural DAG /// actually reaches must not have the caller-supplied `update_rate` /// stamped onto it. `AvgToSumOverCountStrategy` (part of /// `default_strategies`, so included by `search_workload`) is a real, @@ -1417,7 +1417,7 @@ mod tests { assert_eq!( count_profile, RecurrenceProfile::EMPTY, - "a site unreachable from any root's own structural tree must fall back to \ + "a site unreachable from any root's own structural DAG must fall back to \ RecurrenceProfile::EMPTY (no update_rate, no evaluation_rate, no one-shot \ consumers), not just an evaluation-rate-free profile that still carries the \ caller's update_rate" @@ -1489,13 +1489,13 @@ mod tests { #[test] fn decide_rejects_a_zero_or_negative_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 2, }; diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 809955996..c3b2c1f6a 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -34,7 +34,7 @@ //! - [`TargetSubDAG`] — a reference to a pre-ASAP [`QueryExpr`] node that is a //! candidate for replacement, plus how many places in the workload already //! reference it (its `consumer_count`) — the one piece of cross-node -//! context [`SharedSubtreeStrategy`] needs that a bare node reference alone +//! context [`SharedSubDagStrategy`] needs that a bare node reference alone //! doesn't carry. //! - [`ReplacementSubDAG`] — one candidate replacement for a `TargetSubDAG`: //! either a fully bound [`SummaryNode`] or a pre-ASAP [`QueryExpr`] rewrite @@ -76,16 +76,16 @@ //! ranked list directly: for the same bindable-`Aggregate` shape this crate //! binds (single intent, no `HAVING`), every entry becomes its own bound //! candidate. -//! - [`SharedSubtreeStrategy`] wraps -//! `asap_types::pre_asap::cse::share_common_subtrees`'s sharing decision. +//! - [`SharedSubDagStrategy`] wraps +//! `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing decision. //! Wherever a [`TargetSubDAG`] already has two or more consumers (i.e. -//! `share_common_subtrees` already collapsed two or more workload +//! `share_common_sub_dags` already collapsed two or more workload //! locations onto the same `Rc` — [`discover_targets`] below //! does the identical workload-wide discovery for [`search_workload_with`]; //! this module's own tests reuse the same dedup logic to build realistic //! fixtures), it reports the two-way candidate CSE's own detection pass //! deliberately declines to pick between on its own: build once and share -//! the already-interned subtree, or build it independently at each +//! the already-interned sub-DAG, or build it independently at each //! consumer. [`crate::cost_model::CostModel::cse_share_decision`] is where //! that choice actually gets made *today* (a fixed comparison, not a //! search) — this strategy exposes the same two-way choice as an explicit, @@ -136,15 +136,15 @@ //! ``` //! //! Read literally, this enumerates whole *plans* — full copies of the -//! workload's tree, one per combination of per-target choices. A workload +//! workload's DAG, one per combination of per-target choices. A workload //! with `N` independently-choosable targets would produce up to `2^N` flat -//! plans, each one duplicating every untouched sibling subtree. This module +//! plans, each one duplicating every untouched sibling sub-DAG. This module //! does not do that: //! //! 1. **Per-target candidates, not flat plans.** [`TargetSubDAGCandidates`] //! stores the alternatives for one distinct [`TargetSubDAG`] (identified by //! its own `Rc` pointer identity — the same currency -//! [`asap_types::pre_asap::cse::share_common_subtrees`] already +//! [`asap_types::pre_asap::cse::share_common_sub_dags`] already //! established across the workload) holding every //! [`ReplacementSubDAG`] alternative discovered for it. [`CandidateLogicalASAPDAGs`] is //! a collection of these groups, keyed by `TargetSubDAG` — a candidate @@ -175,12 +175,12 @@ //! line above stands for: every `TargetSubDAG` this pass discovers is one //! iteration of that loop. It walks every workload root's whole DAG (the //! same **relational-skeleton** operator-child scope -//! `asap_types::pre_asap::cse::share_common_subtrees` itself uses — see +//! `asap_types::pre_asap::cse::share_common_sub_dags` itself uses — see //! that module's "Algorithm" section), discovering one `TargetSubDAG` per //! distinct `Rc` and a *real* `consumer_count`: how many operator-child //! positions anywhere in the workload reference that exact `Rc`, not just //! how many of the workload's own top-level roots happen to be it — a -//! `SharedSubtreeStrategy` candidate three levels under an unshared +//! `SharedSubDagStrategy` candidate three levels under an unshared //! `Filter` is exactly as real a target as a shared whole root, so this //! module's discovery can't stop at the top level. //! @@ -214,7 +214,7 @@ //! not already known, and any found become next round's frontier. Both shipped //! strategies are idempotent in exactly this sense: [`SketchAlgorithmStrategy`] //! produces terminal [`Replacement::Summary`] candidates (no `QueryExpr` -//! children to scan at all), and [`SharedSubtreeStrategy`]'s two +//! children to scan at all), and [`SharedSubDagStrategy`]'s two //! [`Replacement::Rewrite`] candidates both reuse the target's own //! already-known child `Rc`s verbatim (`Rc::clone`/a shallow top-level //! `.clone()` — see that strategy's own doc). So for both, the frontier is @@ -244,7 +244,7 @@ //! being the whole story; it isn't a contradiction of #237, it's the scope //! change #237 itself named). Concretely, per [`TargetSubDAGCandidates`]: //! -//! - A group whose candidates are the [`SharedSubtreeStrategy`] +//! - A group whose candidates are the [`SharedSubDagStrategy`] //! share-vs-recompute pair is ranked by calling //! [`CostModel::cse_share_decision`] via this module's own //! [`cse_preference`] — rather than re-deriving a competing comparison. @@ -265,9 +265,9 @@ //! interact — which both shipped strategies' one-round convergence (see //! "Termination" above) makes the common case — but it's the wrong answer //! whenever they do. Concretely: [`CostModel::cse_share_decision`] costs a -//! [`SharedSubtreeStrategy`] group by comparing a `consumer_count`-scaled +//! [`SharedSubDagStrategy`] group by comparing a `consumer_count`-scaled //! recompute cost against a fixed maintenance cost — but a **nested** -//! `SharedSubtreeStrategy` group's *true* recompute burden isn't its own +//! `SharedSubDagStrategy` group's *true* recompute burden isn't its own //! raw [`TargetSubDAGCandidates::consumer_count`] (how many operator-child positions //! directly reference it) whenever an ancestor on the path to it is //! *itself* being recomputed independently rather than shared: recomputing @@ -280,12 +280,12 @@ //! [`CandidateLogicalASAPDAGs::global_selection`] is that missing step: a single //! **top-down dynamic-programming pass** over the discovered sites, //! processed in the topological order [`topological_order`] computes over a -//! small [`ReferenceGraph`] built for exactly this purpose (parent before +//! small [`ReferenceDAG`] built for exactly this purpose (parent before //! every child, so a site's `effective_consumer_count` is always computed //! from *already-decided* ancestors). For every site it computes the //! **effective consumer count** — how many times that site actually runs //! once every ancestor's own selected candidate is accounted for — and, for -//! every [`SharedSubtreeStrategy`]-shaped group, re-decides +//! every [`SharedSubDagStrategy`]-shaped group, re-decides //! [`CostModel::cse_share_decision`] against *that* corrected count instead //! of the group's raw structural one. When that group also contains a //! non-CSE alternative such as a semantic rewrite, the chosen CSE candidate @@ -296,7 +296,7 @@ //! multiplicity to exactly `1` for everything beneath it (one shared //! execution backs every use of it); a group that chooses //! `RecomputeIndependently` — or has no Share/Recompute decision of its own -//! at all, i.e. isn't itself a `SharedSubtreeStrategy` shape — passes its +//! at all, i.e. isn't itself a `SharedSubDagStrategy` shape — passes its //! *own* effective count straight through to whatever it references, //! transitively composing contributions from every ancestor on the path, //! not just the immediate parent. @@ -330,10 +330,10 @@ //! groups included. //! - This is not an exhaustive search over combinations of choices for a //! provably-global optimum in every case. [`CostModel::cse_share_decision`] -//! is still a *local*, pairwise comparison at each `SharedSubtreeStrategy` +//! is still a *local*, pairwise comparison at each `SharedSubDagStrategy` //! site (recompute-total vs. one fixed maintenance cost) — this module //! just now feeds it a *correct* input instead of an *incorrect* one. Two -//! sibling `SharedSubtreeStrategy` groups that could trade off against +//! sibling `SharedSubDagStrategy` groups that could trade off against //! each other under some shared resource budget (memory, say) still //! aren't jointly optimized here — this crate has no //! cardinality/statistics estimation to bound a combinatorial search like @@ -359,7 +359,7 @@ use asap_types::post_asap::{ use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; use asap_types::pre_asap::agg_intent::{agg_is_mergeable, AggIntent}; use asap_types::pre_asap::column_resolution::resolve_column_ref; -use asap_types::pre_asap::cse::{share_common_subtrees, structural_hash, HashCache}; +use asap_types::pre_asap::cse::{share_common_sub_dags, structural_hash, HashCache}; use asap_types::pre_asap::expr_ir::{ArithmeticOpKind, ColumnRef}; use asap_types::pre_asap::query_expr::any_measure_filtered; use asap_types::pre_asap::query_expr::{ @@ -426,20 +426,20 @@ pub enum RealizationError { /// A pre-ASAP sub-DAG a [`ReplacementStrategy`] knows how to replace. /// -/// `root` is a reference into the workload's own [`QueryExpr`] tree (an +/// `root` is a reference into the workload's own [`QueryExpr`] DAG (an /// `Rc`, the same currency [`search_workload`] and -/// `asap_types::pre_asap::cse::share_common_subtrees` already thread through +/// `asap_types::pre_asap::cse::share_common_sub_dags` already thread through /// this crate's public API — not a bare `&QueryExpr` — so a strategy that /// needs the node's own `Rc` identity, not just its shape, has it available /// without the caller re-deriving it). /// /// `consumer_count` is how many locations across the workload reference this -/// exact `Rc` — 1 for an ordinary single-use node and 2+ for a shared subtree. +/// exact `Rc` — 1 for an ordinary single-use node and 2+ for a shared sub-DAG. /// [`search_workload_with`] computes the workload-wide value during target /// discovery. [`TargetSubDAG::new`] defaults it to `1` for callers invoking a /// strategy against one node in isolation. A strategy that only cares about /// `root`'s shape (for example, [`SketchAlgorithmStrategy`]) can ignore the -/// count; [`SharedSubtreeStrategy`] consults it directly. +/// count; [`SharedSubDagStrategy`] consults it directly. /// /// `strictest_sibling_accuracy` is the strictest accuracy among workload /// siblings that read the same summary input as `root`, when stricter than @@ -487,7 +487,7 @@ pub enum Replacement { Summary(Rc), /// A pre-ASAP rewrite: still a logical [`QueryExpr`], structurally /// different from the target's own `root` (e.g. sharing vs. not sharing - /// a subtree) but semantically equivalent to it. + /// a sub-DAG) but semantically equivalent to it. Rewrite(Rc), /// An exact operator composed over another target's *own* selected /// decision across an explicit update/readout boundary (issue #171): @@ -614,7 +614,7 @@ pub struct Proposals { /// of this trait or any existing strategy required. /// /// `replacements` is only meaningful when `matches` would return `true` for -/// the same target; both [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] +/// the same target; both [`SketchAlgorithmStrategy`] and [`SharedSubDagStrategy`] /// return an empty `Vec` rather than panicking when called on a target they /// don't match, so a caller that skips the `matches` check first still gets a /// safe (merely uninformative) answer instead of a crash. @@ -1573,7 +1573,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { proposals.record( format!( "{rationale}; sized under {} budget split of {target:?} across \ - {} approximate layers (this layer {outer_target:?}, child subtree \ + {} approximate layers (this layer {outer_target:?}, child sub-DAG \ {inner_target:?})", allocation.allocator, shape.approximate_layer_count ), @@ -1933,7 +1933,7 @@ pub(crate) fn realize_child_with( // produce — never happens, that match is exhaustive), or every // candidate was accuracy-illegal — either way the same conservative // fallback `SketchAlgorithmStrategy::matches` uses: keep the - // pre-ASAP subtree, executed exactly. + // pre-ASAP sub-DAG, executed exactly. None => keep_pre_asap(root), } } @@ -2336,7 +2336,7 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { out } -/// Wrap an unrewritten pre-ASAP subtree, lifting its schema with every column +/// Wrap an unrewritten pre-ASAP sub-DAG, lifting its schema with every column /// `SummaryFamilyType::Plain`. `pub` so a caller can fall back to this /// explicitly — e.g. when `SketchAlgorithmStrategy::replacements()` returns no /// candidate for a target, or a deployment wants to force a node its own @@ -2351,7 +2351,7 @@ fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationE Ok(Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(expr), schema: lift(&schema), - // A kept pre-ASAP subtree is executed exactly by the runtime + // A kept pre-ASAP sub-DAG is executed exactly by the runtime // (`Realization::PassThrough`'s contract) — zero error. guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), })) @@ -2363,7 +2363,7 @@ fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationE /// `HAVING`. A multi-intent node (SQL `SELECT SUM(a), AVG(b)`), or one with a /// `HAVING` predicate (the filter would need the estimate first), stays /// logical. Unsupported logical parents still conservatively become one -/// [`SummaryExpr::KeepPreAsap`] subtree. Composable query-time value +/// [`SummaryExpr::KeepPreAsap`] sub-DAG. Composable query-time value /// operators (`Project`, `Filter`, `Sort`, and `Limit`) are retained during final /// DAG assembly so their independently planned children remain visible. pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { @@ -2400,7 +2400,7 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// Construct a summary with every model explicit (issue #172). `intent` /// is `expr`'s own [`bindable_intent`], or a copy of it with an allocated /// `AccuracyTarget` substituted (see [`realize_child_with`]). -/// `child_target`, when set, is the end-to-end budget the child subtree is +/// `child_target`, when set, is the end-to-end budget the child sub-DAG is /// re-enumerated under; `allocation` is the provenance note recording the /// split that produced both. `Err(RealizationError::Accuracy)` is the /// fail-closed answer for a composition with no sound rule or one that @@ -3756,15 +3756,15 @@ fn lift(schema: &Schema) -> SummarySchema { } } -// ── SharedSubtreeStrategy ──────────────────────────────────────────────── +// ── SharedSubDagStrategy ──────────────────────────────────────────────── -/// Wraps `asap_types::pre_asap::cse::share_common_subtrees`'s sharing +/// Wraps `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing /// decision as an explicit candidate pair, wherever a [`TargetSubDAG`] /// already has two or more consumers. /// /// This strategy does not decide sharing itself, nor does it discover which /// nodes are shared — by the time a caller builds a `TargetSubDAG` with -/// `consumer_count >= 2`, `share_common_subtrees` has already made that +/// `consumer_count >= 2`, `share_common_sub_dags` has already made that /// (legality-gated, `PartialEq`-checked) call; [`discover_targets`] below /// discovers real consumer counts across a workload the same way for /// [`search_workload_with`] (this module's own tests reuse the identical @@ -3774,9 +3774,9 @@ fn lift(schema: &Schema) -> SummarySchema { /// `Rc`" as the two-way choice a downstream cost model (today, /// [`CostModel::cse_share_decision`]) picks between: build once and share, or /// build independently at each consumer. -pub struct SharedSubtreeStrategy; +pub struct SharedSubDAGStrategy; -impl ReplacementStrategy for SharedSubtreeStrategy { +impl ReplacementStrategy for SharedSubDAGStrategy { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { target.consumer_count >= 2 } @@ -3788,27 +3788,27 @@ impl ReplacementStrategy for SharedSubtreeStrategy { let count = target.consumer_count; vec![ ReplacementSubDAG { - strategy: "SharedSubtreeStrategy", + strategy: "SharedSubDAGStrategy", // The already-interned `Rc` itself: reusing it verbatim *is* // "build once and share" — no new node to construct. replacement: Replacement::Rewrite(Rc::clone(target.root)), provenance: ReplacementProvenance::CseShare, rationale: format!( - "build once and share: share_common_subtrees already interned this \ - subtree once and reused it across {count} consumers — one build can \ + "build once and share: share_common_sub_dags already interned this \ + sub-DAG once and reused it across {count} consumers — one build can \ answer all of them instead of computing it {count} times" ), }, ReplacementSubDAG { - strategy: "SharedSubtreeStrategy", + strategy: "SharedSubDAGStrategy", // A structurally-identical but freshly-allocated `Rc`: same // value (`PartialEq`), deliberately *not* the same pointer, // representing "undo the sharing and recompute independently". replacement: Replacement::Rewrite(Rc::new((**target.root).clone())), provenance: ReplacementProvenance::CseRecompute, rationale: format!( - "build independently: undo the sharing share_common_subtrees found and \ - recompute this subtree separately at each of its {count} consumers — \ + "build independently: undo the sharing share_common_sub_dags found and \ + recompute this sub-DAG separately at each of its {count} consumers — \ worth it only when independence outweighs the shared-maintenance cost, \ a CostModel's call (e.g. CostModel::cse_share_decision) and not this \ strategy's" @@ -3827,7 +3827,7 @@ impl ReplacementStrategy for SharedSubtreeStrategy { /// A generous, documented backstop against a hypothetically ill-behaved /// future [`ReplacementStrategy`] (see the module docs' "Termination" /// section) — not a bound either shipped strategy could ever approach. -/// [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] both converge in +/// [`SketchAlgorithmStrategy`] and [`SharedSubDagStrategy`] both converge in /// exactly 2 passes over a fixed target set, regardless of workload size. pub const MAX_SEARCH_ITERATIONS: usize = 1_000; @@ -3910,7 +3910,7 @@ impl TargetSubDAGCandidates { /// /// Structural (`QueryExpr`) value equality alone is *not* enough here: this /// module's one shipped multi-candidate `Replacement::Rewrite` source, -/// [`SharedSubtreeStrategy`], deliberately returns **two** candidates that +/// [`SharedSubDagStrategy`], deliberately returns **two** candidates that /// are value-equal to each other (`build once and share` vs. `build /// independently` — see that strategy's own doc) but represent genuinely /// different physical choices, distinguished *only* by whether the @@ -3932,7 +3932,7 @@ impl TargetSubDAGCandidates { /// (currently hypothetical, since neither shipped strategy causes it) /// case of the exact same alternative being proposed twice. A fresh /// [`HashCache`] per call: this is a pairwise check between two candidates -/// for one group, not a bottom-up pass over a whole tree, so there is no +/// for one group, not a bottom-up pass over a whole DAG, so there is no /// wider traversal to amortize the cache across the way `InternTable`'s own /// use of `structural_hash` does. fn is_duplicate_rewrite( @@ -3980,7 +3980,7 @@ fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc` whose /// group holds its alternatives. pub struct CandidateLogicalASAPDAGs { - /// The workload's roots, after the one `share_common_subtrees` pass + /// The workload's roots, after the one `share_common_sub_dags` pass /// [`search_workload_with`] runs up front — the same post-CSE roots /// every `TargetSubDAG` in `groups` was discovered from. pub roots: Vec<(Id, Rc)>, @@ -4214,7 +4214,7 @@ impl CandidateLogicalASAPDAGs { .collect::, _>>(); match roots { Ok(roots) => { - let roots = asap_types::post_asap::share_common_summary_subtrees(roots); + let roots = asap_types::post_asap::share_common_summary_sub_dags(roots); use std::hash::{Hash, Hasher}; let mut hash = std::collections::hash_map::DefaultHasher::new(); let mut pending = roots @@ -4533,7 +4533,7 @@ impl CandidateLogicalASAPDAGs { /// `self.roots[i]` — the same order [`search_workload`]/ /// [`search_workload_with`] were originally called with (post-CSE /// dedup preserves both root count and order — see - /// `asap_types::pre_asap::cse::share_common_subtrees`'s own + /// `asap_types::pre_asap::cse::share_common_sub_dags`'s own /// `.map(...).collect()` body). This keeps `Id` fully opaque (no `Eq`/ /// `Hash`/`Clone` bound needed on it at all — issue #287's "keep /// caller/query identifiers opaque" requirement) at the cost of the @@ -4568,7 +4568,7 @@ impl CandidateLogicalASAPDAGs { /// effective structural execution rate rather than mere reachability. /// /// **Unreachable sites**: [`CandidateLogicalASAPDAGs`] can contain a site no root's own - /// structural tree actually reaches — e.g. one only ever produced by a + /// structural DAG actually reaches — e.g. one only ever produced by a /// [`Replacement::Rewrite`] candidate a [`ReplacementStrategy`] invented /// (this walk only follows [`TargetSubDAGCandidates::target`]'s own structural /// children, the same scope [`discover_targets`] uses for the original @@ -4873,7 +4873,7 @@ fn rank_group<'a>( return ranked; } - // Shape 1: the exact `SharedSubtreeStrategy` share-vs-recompute pair — + // Shape 1: the exact `SharedSubDagStrategy` share-vs-recompute pair — // rank via `CostModel::cse_share_decision`, the same comparison // the local CSE ranking path already uses. if cse_candidate_pair(group).is_some() { @@ -4974,14 +4974,14 @@ fn rank_group<'a>( } /// For a group whose candidates are all [`Replacement::Rewrite`] (the -/// [`SharedSubtreeStrategy`] shape): does [`CostModel::cse_share_decision`] +/// [`SharedSubDagStrategy`] shape): does [`CostModel::cse_share_decision`] /// prefer the candidate that shares `group.target`'s own `Rc` (`true`), or /// the one that recomputes independently (`false`)? `None` when there's no /// real comparison to make — fewer than 2 consumers (mirrors -/// [`SharedSubtreeStrategy::matches`]'s own gate), or `group.target` can't +/// [`SharedSubDagStrategy::matches`]'s own gate), or `group.target` can't /// actually be bound at all (no candidate and no logical fallback — never /// expected in practice for a target that's already part of a legitimate -/// workload tree, but this degrades to "keep discovery order" rather than +/// workload DAG, but this degrades to "keep discovery order" rather than /// panicking). fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> Option { if group.consumer_count < 2 { @@ -4989,7 +4989,7 @@ fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> } let bound = realize_one(&group.target, cost_model)?; let candidate = CseCandidate { - subtree: &group.target, + sub_dag: &group.target, bound_summary: &bound, consumer_count: group.consumer_count, }; @@ -5047,7 +5047,7 @@ fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { /// One target sub-DAG's selected choice and usage information — the answer /// [`CandidateLogicalASAPDAGs::global_selection`] commits to for one site, after folding in -/// every ancestor [`SharedSubtreeStrategy`] decision on the path from a +/// every ancestor [`SharedSubDagStrategy`] decision on the path from a /// workload root to this site. See the module docs' "Whole-plan /// (cross-group) selection" section for the full recurrence. /// @@ -5071,7 +5071,7 @@ pub struct TargetSubDAGSelection<'a> { /// ancestor's own selected candidate is accounted for — see /// [`multiplier`]'s doc for the exact recurrence. Equal to /// `consumer_count` unless some ancestor on a path from a root to this - /// site has a [`SharedSubtreeStrategy`] alternative that chose + /// site has a [`SharedSubDagStrategy`] alternative that chose /// [`ShareDecision::RecomputeIndependently`]. pub effective_consumer_count: usize, /// The candidate chosen for this target, or `None` when no replacement @@ -5262,7 +5262,7 @@ impl<'a> GlobalSelection<'a> { /// when the operator itself has no summary realization. Its child is /// assembled independently, so a selected summary remains visible /// beneath `Project`/`Filter`/`Sort`/`Limit` instead of being swallowed by - /// one opaque `KeepPreAsap` subtree. + /// one opaque `KeepPreAsap` sub-DAG. fn assemble_residual( &self, target: &Rc, @@ -5655,7 +5655,7 @@ impl CandidateLogicalASAPDAGs { /// "Whole-plan (cross-group) selection" section describes: one /// [`TargetSubDAGSelection`] per discovered site, each ranked against an /// `effective_consumer_count` that accounts for every ancestor - /// [`SharedSubtreeStrategy`] decision on the path to it — unlike + /// [`SharedSubDagStrategy`] decision on the path to it — unlike /// [`Self::cost_sorted`], whose per-group ranking only ever sees a /// group's own raw [`TargetSubDAGCandidates::consumer_count`]. /// Uncertified DDSketch ratios remain in [`CandidateLogicalASAPDAGs`] for downstream @@ -5695,10 +5695,10 @@ impl CandidateLogicalASAPDAGs { horizon: Option, candidate_costs: Option<&CandidateCostOverrides>, ) -> Result, RecurrenceError> { - let graph = reference_graph(self); - let topo = topological_order(&self.order, &graph); + let dag = reference_dag(self); + let topo = topological_order(&self.order, &dag); - let mut effective_uses = graph.external_root_uses.clone(); + let mut effective_uses = dag.external_root_uses.clone(); let mut chosen_share: HashMap<*const QueryExpr, ShareDecision> = HashMap::new(); let mut groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'_>> = HashMap::new(); let mut context = CompositionContext::default(); @@ -5801,7 +5801,7 @@ impl CandidateLogicalASAPDAGs { .then(|| { decide_with_effective_count(group, effective, cost_model).and_then( |decision| { - let candidate = pick_shared_subtree_candidate(group, decision)?; + let candidate = pick_shared_sub_dag_candidate(group, decision)?; chosen_share.insert(*ptr, decision); Some(candidate) }, @@ -5835,7 +5835,7 @@ impl CandidateLogicalASAPDAGs { }; match decision { Some(decision) => { - let cse = pick_shared_subtree_candidate(group, decision); + let cse = pick_shared_sub_dag_candidate(group, decision); let effective_target = TargetSubDAG::with_consumer_count(&group.target, effective); let logical = group @@ -5881,7 +5881,7 @@ impl CandidateLogicalASAPDAGs { } // `realize_child` couldn't produce even a logical fallback — // not expected in practice for a target that's already - // part of a legitimate workload tree (mirrors + // part of a legitimate workload DAG (mirrors // `cse_preference`'s own doc on this same degrade). // Falling back to ordinary local ranking is still a // valid answer, just not a cross-group-aware one; this @@ -6009,7 +6009,7 @@ fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn C /// chose [`ShareDecision::RecomputeIndependently`] (each of its own uses /// gets its own independent execution, so referencing it costs as much as /// its *own* full multiplicity), or it has no Share/Recompute decision at -/// all (not a [`SharedSubtreeStrategy`] shape — nothing here collapses +/// all (not a [`SharedSubDagStrategy`] shape — nothing here collapses /// its multiplicity to one, so whatever multiplicity *its* ancestors /// established simply passes through). /// @@ -6082,7 +6082,7 @@ fn decide_with_effective_count( ) -> Option { let bound = realize_child(&group.target, cost_model).ok()?; let candidate = CseCandidate { - subtree: &group.target, + sub_dag: &group.target, bound_summary: &bound, consumer_count: effective_consumer_count, }; @@ -6100,7 +6100,7 @@ fn decide_group_with_recurrence( return Ok(None); }; let candidate = CseCandidate { - subtree: &group.target, + sub_dag: &group.target, bound_summary: &bound, consumer_count: effective_consumer_count, }; @@ -6111,12 +6111,12 @@ fn decide_group_with_recurrence( )) } -/// The [`SharedSubtreeStrategy`] candidate matching `decision`: the one +/// The [`SharedSubDagStrategy`] candidate matching `decision`: the one /// that shares `group.target`'s own `Rc` for [`ShareDecision::Share`], the /// freshly-allocated one for [`ShareDecision::RecomputeIndependently`] — /// the same `Rc`-identity distinction [`is_duplicate_rewrite`]'s own doc /// explains is the *only* signal this IR carries for that choice. -fn pick_shared_subtree_candidate( +fn pick_shared_sub_dag_candidate( group: &TargetSubDAGCandidates, decision: ShareDecision, ) -> Option<&ReplacementSubDAG> { @@ -6127,7 +6127,7 @@ fn pick_shared_subtree_candidate( }) } -// ── reference graph + topological order ───────────────────────────────── +// ── reference DAG + topological order ───────────────────────────────── /// The parent/child structure [`CandidateLogicalASAPDAGs::global_selection`]'s DP walks — /// built separately from [`discover_targets`]'s own `order`/`nodes`/`counts` @@ -6135,7 +6135,7 @@ fn pick_shared_subtree_candidate( /// breakdown or direction) rather than extending that already-reviewed, /// already-tested pass. Same "small duplicated traversal over reshaping /// proven code" call as [`is_shared_subtree_group`]. -struct ReferenceGraph { +struct ReferenceDAG { /// child ptr -> `(parent ptr, edge count from that one parent)`, for /// every direct operator-child edge in the relational-skeleton scope /// [`walk_children`] itself uses (an edge count above 1 happens when @@ -6147,75 +6147,68 @@ struct ReferenceGraph { /// traversal. children_of: HashMap<*const QueryExpr, Vec<*const QueryExpr>>, /// How many of the workload's own `roots` point directly at each node — - /// a node's "external" use. Nothing inside the tree decides this (it + /// a node's "external" use. Nothing inside the DAG decides this (it /// isn't a reference from another discovered site), so it's never /// subject to any ancestor's Share/Recompute choice — it's the base /// case [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence starts from. external_root_uses: HashMap<*const QueryExpr, usize>, } -/// Build an ordering graph containing every edge that could be selected: +/// Build an ordering DAG containing every edge that could be selected: /// the original target's edges plus every rewrite candidate's edges. An /// accuracy-reconciliation rewrite points at another discovered memo group, /// so it contributes an edge to that group itself; other rewrites contribute /// their relational children as before. The -/// graph is deliberately only used for topological ordering; effective-use +/// DAG is deliberately only used for topological ordering; effective-use /// counts are propagated through the one candidate actually selected. -fn reference_graph(space: &CandidateLogicalASAPDAGs) -> ReferenceGraph { - let mut graph = ReferenceGraph { +fn reference_dag(space: &CandidateLogicalASAPDAGs) -> ReferenceDAG { + let mut dag = ReferenceDAG { parents_of: HashMap::new(), children_of: HashMap::new(), external_root_uses: HashMap::new(), }; for (_, root) in &space.roots { - *graph - .external_root_uses - .entry(Rc::as_ptr(root)) - .or_insert(0) += 1; + *dag.external_root_uses.entry(Rc::as_ptr(root)).or_insert(0) += 1; } for ptr in &space.order { let group = &space.groups[ptr]; - record_possible_edges(*ptr, &group.target, &mut graph); + record_possible_edges(*ptr, &group.target, &mut dag); for candidate in &group.candidates { if let Replacement::Rewrite(rewrite) = &candidate.replacement { if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { - add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut graph); + add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut dag); } else { - record_possible_edges(*ptr, rewrite, &mut graph); + record_possible_edges(*ptr, rewrite, &mut dag); } } } } - graph + dag } /// Record one `parent_ptr -> child` edge (both directions — see -/// [`ReferenceGraph`]'s fields), retaining the greatest multiplicity seen +/// [`ReferenceDAG`]'s fields), retaining the greatest multiplicity seen /// when the target and alternative rewrites expose the same edge. fn add_edge( parent_ptr: *const QueryExpr, child_ptr: *const QueryExpr, edge_count: usize, - graph: &mut ReferenceGraph, + dag: &mut ReferenceDAG, ) { - let siblings = graph.parents_of.entry(child_ptr).or_default(); + let siblings = dag.parents_of.entry(child_ptr).or_default(); match siblings.iter_mut().find(|(p, _)| *p == parent_ptr) { Some((_, count)) => *count = (*count).max(edge_count), None => siblings.push((parent_ptr, edge_count)), } - let kids = graph.children_of.entry(parent_ptr).or_default(); + let kids = dag.children_of.entry(parent_ptr).or_default(); if !kids.contains(&child_ptr) { kids.push(child_ptr); } } -fn record_possible_edges( - parent_ptr: *const QueryExpr, - node: &QueryExpr, - graph: &mut ReferenceGraph, -) { +fn record_possible_edges(parent_ptr: *const QueryExpr, node: &QueryExpr, dag: &mut ReferenceDAG) { for (child_ptr, edge_count) in direct_child_counts(node) { - add_edge(parent_ptr, child_ptr, edge_count, graph); + add_edge(parent_ptr, child_ptr, edge_count, dag); } } @@ -6290,16 +6283,16 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { } /// A topological order over `order` (parent before every child) via Kahn's -/// algorithm on `graph`'s reverse adjacency — needed because +/// algorithm on `dag`'s reverse adjacency — needed because /// [`discover_targets`]'s own `order` is only a valid *discovery* order /// (first-seen-first), not a valid topological one: a node reached via two /// different root paths can have a parent that's discovered *after* it (see /// this function's own test for a worked diamond example), which is exactly /// backwards for [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence. -fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec<*const QueryExpr> { +fn topological_order(order: &[*const QueryExpr], dag: &ReferenceDAG) -> Vec<*const QueryExpr> { let mut in_degree: HashMap<*const QueryExpr, usize> = HashMap::new(); for ptr in order { - let degree = graph.parents_of.get(ptr).map(Vec::len).unwrap_or(0); + let degree = dag.parents_of.get(ptr).map(Vec::len).unwrap_or(0); in_degree.insert(*ptr, degree); } @@ -6312,7 +6305,7 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< let mut topo = Vec::with_capacity(order.len()); while let Some(ptr) = queue.pop_front() { topo.push(ptr); - if let Some(children) = graph.children_of.get(&ptr) { + if let Some(children) = dag.children_of.get(&ptr) { for child in children { if let Some(degree) = in_degree.get_mut(child) { *degree -= 1; @@ -6327,9 +6320,9 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< assert_eq!( topo.len(), order.len(), - "topological_order: the discovered-site reference graph has a cycle — every QueryExpr \ + "topological_order: the discovered-site reference DAG has a cycle — every QueryExpr \ node is built from Rc children, which can't form one, so this indicates a bug in \ - reference_graph rather than a real cyclic workload", + reference_dag rather than a real cyclic workload", ); topo } @@ -6352,10 +6345,10 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< /// included here (issue #253) even though it's a /// [`Replacement::Rewrite`]-only strategy with no [`CostModel`] of its own to /// plug in — it's context-free (`matches`/`replacements` need nothing beyond -/// the target itself) exactly like [`SharedSubtreeStrategy`], so it belongs +/// the target itself) exactly like [`SharedSubDagStrategy`], so it belongs /// in this list rather than being derived per-workload the way /// [`RollupStrategy`] is. Rewriting `avg` into `sum`/`count` upfront is what -/// lets [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] see a +/// lets [`SketchAlgorithmStrategy`] and [`SharedSubDagStrategy`] see a /// mergeable accumulator to sketch or share at all — see that module's own /// doc comment for why a bare `avg` node otherwise never becomes a /// [`ReplacementStrategy`] target for anything. @@ -6363,7 +6356,7 @@ pub fn default_strategies() -> Vec> { vec![ Box::new(SketchAlgorithmStrategy::default_cost_model()), Box::new(HydraGroupingStrategy::default_cost_model()), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::AvgToSumOverCountStrategy), Box::new(ExactCompositionStrategy::default_cost_model()), ] @@ -6378,7 +6371,7 @@ pub fn default_strategies_with<'a>( vec![ Box::new(SketchAlgorithmStrategy::new(cost_model)), Box::new(HydraGroupingStrategy::new(cost_model)), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::SemanticEquivalentRewriteStrategy), Box::new(ExactCompositionStrategy::new(cost_model)), ] @@ -6409,7 +6402,7 @@ pub fn default_strategies_with_evidence<'a>( evidence, ), ), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::AvgToSumOverCountStrategy), Box::new(ExactCompositionStrategy::new(cost_model)), ] @@ -6435,10 +6428,10 @@ pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalA /// [`RollupStrategy`] is derived and added automatically after CSE for both /// entry points, because only this function owns the post-CSE sibling set. /// -/// Runs [`share_common_subtrees`] once over `roots` first — so every +/// Runs [`share_common_sub_dags`] once over `roots` first — so every /// strategy (and, transitively, every /// [`crate::explanation::ReplacementExplanation`] a caller reads off the -/// result) sees the same already-deduplicated tree — then discovers every +/// result) sees the same already-deduplicated DAG — then discovers every /// `TargetSubDAG` (see [`discover_targets`]) and runs the /// fixpoint loop the module docs describe, capped at /// [`MAX_SEARCH_ITERATIONS`] passes (see the module docs' "Termination" @@ -6651,7 +6644,7 @@ fn strictest_sibling_accuracy( } fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { - // `share_common_subtrees` wants owned `QueryExpr`s, not already-`Rc` + // `share_common_sub_dags` wants owned `QueryExpr`s, not already-`Rc` // roots — the same `Rc::try_unwrap`-with-clone-fallback pattern // `asap_types::pre_asap::cse::intern_child` itself uses to recover an // owned node without cloning in the common (uniquely-owned) case. @@ -6662,7 +6655,7 @@ fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> (id, expr) }) .collect(); - share_common_subtrees(owned_roots) + share_common_sub_dags(owned_roots) } fn search_cse_workload_with<'s, Id>( @@ -6714,7 +6707,7 @@ fn search_cse_workload_with<'s, Id>( "search_workload: fixpoint search did not converge within {MAX_SEARCH_ITERATIONS} \ rounds — a registered ReplacementStrategy's Replacement::Rewrite candidates keep \ exposing new, never-before-seen descendant structure every round. \ - SketchAlgorithmStrategy/SharedSubtreeStrategy never do this (see replacement.rs's \ + SketchAlgorithmStrategy/SharedSubDAGStrategy never do this (see replacement.rs's \ module docs' \"Termination\" section); check any custom strategies passed to \ search_workload_with.", ); @@ -6816,7 +6809,7 @@ fn search_cse_workload_with<'s, Id>( /// Materialize share/recompute alternatives for descendants whose raw edge /// count is one but whose effective count can exceed one when a repeated /// ancestor is recomputed. We only do this when an ordinary repeated group -/// proves that `SharedSubtreeStrategy` is part of this search's strategy set. +/// proves that `SharedSubDagStrategy` is part of this search's strategy set. fn add_effective_count_cse_candidates( order: &[*const QueryExpr], groups: &mut HashMap<*const QueryExpr, TargetSubDAGCandidates>, @@ -6867,14 +6860,14 @@ fn add_effective_count_cse_candidates( if potentially_repeated.contains(ptr) && cse_candidate_pair(group).is_none() { let target = Rc::clone(&group.target); let site = TargetSubDAG::with_consumer_count(&target, 2); - for mut candidate in SharedSubtreeStrategy.replacements(&site) { + for mut candidate in SharedSubDAGStrategy.replacements(&site) { candidate.rationale = format!( - "{}: this subtree can become repeated when a repeated ancestor is recomputed; \ + "{}: this sub-DAG can become repeated when a repeated ancestor is recomputed; \ global_selection decides using its effective consumer count", match candidate.provenance { ReplacementProvenance::CseShare => "build once and share", ReplacementProvenance::CseRecompute => "recompute independently", - _ => unreachable!("SharedSubtreeStrategy only emits CSE candidates"), + _ => unreachable!("SharedSubDagStrategy only emits CSE candidates"), } ); group.add_candidate(candidate); @@ -6935,7 +6928,7 @@ fn walk( } /// `node`'s own **relational-skeleton** operator children — the same scope -/// `asap_types::pre_asap::cse::share_common_subtrees`/`rebuild_children` +/// `asap_types::pre_asap::cse::share_common_sub_dags`/`rebuild_children` /// itself uses (see that module's "Algorithm" section) and /// `tests::count_consumers` mirrors for its own fixtures. Exhaustive over /// every `QueryExpr` variant: a new variant fails to compile here until this @@ -7957,7 +7950,7 @@ mod tests { ); } - // ── SketchAlgorithmStrategy / SharedSubtreeStrategy fixtures ─────────── + // ── SketchAlgorithmStrategy / SharedSubDagStrategy fixtures ─────────── fn metric_scan(labels: &[&str]) -> QueryExpr { let mut columns = vec![ @@ -8272,7 +8265,7 @@ mod tests { } } - // ── SharedSubtreeStrategy ──────────────────────────────────────────── + // ── SharedSubDagStrategy ──────────────────────────────────────────── #[test] fn does_not_match_a_single_consumer_target() { @@ -8283,8 +8276,8 @@ mod tests { )); let target = TargetSubDAG::new(&q); assert_eq!(target.consumer_count, 1); - assert!(!SharedSubtreeStrategy.matches(&target)); - assert!(SharedSubtreeStrategy.replacements(&target).is_empty()); + assert!(!SharedSubDAGStrategy.matches(&target)); + assert!(SharedSubDAGStrategy.replacements(&target).is_empty()); } #[test] @@ -8295,9 +8288,9 @@ mod tests { metric_scan(&["job"]), )); let target = TargetSubDAG::with_consumer_count(&q, 2); - assert!(SharedSubtreeStrategy.matches(&target)); + assert!(SharedSubDAGStrategy.matches(&target)); - let replacements = SharedSubtreeStrategy.replacements(&target); + let replacements = SharedSubDAGStrategy.replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); let shared = match &replacements[0].replacement { @@ -8333,7 +8326,7 @@ mod tests { metric_scan(&["job"]), )); let target = TargetSubDAG::with_consumer_count(&q, 3); - let replacements = SharedSubtreeStrategy.replacements(&target); + let replacements = SharedSubDAGStrategy.replacements(&target); assert!(replacements[0].rationale.contains('3')); assert!(replacements[1].rationale.contains('3')); } @@ -8341,7 +8334,7 @@ mod tests { /// Builds realistic multi-consumer `TargetSubDAG`s the same way this /// module's own [`discover_targets`]/`walk` does: dedup by `Rc::as_ptr`, /// walking only the relational-skeleton operator children - /// `asap_types::pre_asap::cse::share_common_subtrees` itself scopes to, + /// `asap_types::pre_asap::cse::share_common_sub_dags` itself scopes to, /// so a shared node nested below another shared node is only ever /// counted at the highest (maximal) point sharing starts. Test-only: /// this module deliberately does not ship a workload-wide discovery @@ -8411,12 +8404,12 @@ mod tests { #[test] fn realistic_cse_output_produces_a_two_consumer_target() { - // Two workload roots that `share_common_subtrees` collapses onto one + // Two workload roots that `share_common_sub_dags` collapses onto one // Rc (mirrors `explanation`'s and `cse`'s own fixtures): a grouped // Sum aggregate over the same scan, built independently at each root. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let shared = asap_types::pre_asap::cse::share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = asap_types::pre_asap::cse::share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -8428,8 +8421,8 @@ mod tests { assert_eq!(count, 2); let target = TargetSubDAG::with_consumer_count(&roots[0], count); - assert!(SharedSubtreeStrategy.matches(&target)); - assert_eq!(SharedSubtreeStrategy.replacements(&target).len(), 2); + assert!(SharedSubDAGStrategy.matches(&target)); + assert_eq!(SharedSubDAGStrategy.replacements(&target).len(), 2); } // ── search_workload / CandidateLogicalASAPDAGs / TargetSubDAGCandidates (merged from search.rs) ── @@ -8586,10 +8579,10 @@ mod tests { #[test] fn shared_aggregate_across_two_roots_gets_both_strategies_candidates() { // Two independently-built, structurally identical Sum aggregates: - // share_common_subtrees (run inside search_workload) collapses them + // share_common_sub_dags (run inside search_workload) collapses them // onto one Rc with consumer_count 2, so this single group should // carry SketchAlgorithmStrategy's one ExactAggregate candidate *and* - // SharedSubtreeStrategy's share-vs-recompute pair. + // SharedSubDagStrategy's share-vs-recompute pair. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); @@ -8632,7 +8625,7 @@ mod tests { } #[test] - fn nested_shared_subtree_below_an_unshared_parent_is_still_discovered() { + fn nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered() { // A shared grouped Aggregate nested under two *different*, // unshared Filter parents — real consumer_count must come from // walking the whole DAG, not just root-level pointer identity @@ -8665,7 +8658,7 @@ mod tests { "2 distinct Filters + 1 shared Aggregate + 1 shared Scan" ); - // `share_common_subtrees` re-clones+re-interns anything that already + // `share_common_sub_dags` re-clones+re-interns anything that already // had more than one owner going in (see `cse.rs`'s own doc on // `intern_child`'s clone-fallback path) — so the post-CSE shared // node is a *fresh* Rc, structurally equal to (but not the same @@ -8696,7 +8689,7 @@ mod tests { .expect("shared node must be a discovered target"); assert_eq!(group.consumer_count, 2); assert!( - SharedSubtreeStrategy.matches(&TargetSubDAG::with_consumer_count( + SharedSubDAGStrategy.matches(&TargetSubDAG::with_consumer_count( post_cse_shared, group.consumer_count )) @@ -8707,7 +8700,7 @@ mod tests { #[test] fn add_candidate_rejects_a_true_rewrite_duplicate() { - // SharedSubtreeStrategy's `Replacement::Rewrite` candidates are + // SharedSubDagStrategy's `Replacement::Rewrite` candidates are // real `QueryExpr` values with `PartialEq`, so `add_candidate` can // (and must) actually reject a genuine repeat — unlike the // `Replacement::Summary` case (see the test below). @@ -8719,7 +8712,7 @@ mod tests { let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 2); let target = TargetSubDAG::with_consumer_count(&root, 2); let mut inserted = 0; - for candidate in SharedSubtreeStrategy.replacements(&target) { + for candidate in SharedSubDAGStrategy.replacements(&target) { if group.add_candidate(candidate) { inserted += 1; } @@ -8732,7 +8725,7 @@ mod tests { // structurally identical value, both already covered by // `is_duplicate_rewrite`. let mut re_inserted = 0; - for candidate in SharedSubtreeStrategy.replacements(&target) { + for candidate in SharedSubDAGStrategy.replacements(&target) { if group.add_candidate(candidate) { re_inserted += 1; } @@ -8812,7 +8805,7 @@ mod tests { // ── cost-based ranking ─────────────────────────────────────────────── #[test] - fn cost_sorted_orders_shared_subtree_candidates_by_cse_share_decision() { + fn cost_sorted_orders_shared_sub_dag_candidates_by_cse_share_decision() { // Many consumers of a cheap-to-recompute, cheap-to-maintain exact // accumulator: cse_share_decision should prefer Share (see // cost_model.rs's own `cse_share_decision_shares_when_recompute_dominates_maintenance`). @@ -8973,9 +8966,9 @@ mod tests { // ── global_selection (issue #271) ─────────────────────────────────── - /// A `CostModel` with a constant, `subtree`-independent recompute cost + /// A `CostModel` with a constant, `sub_dag`-independent recompute cost /// and shared-maintenance cost, chosen (40 recompute-per-use, 100 - /// maintenance) so that a `SharedSubtreeStrategy` group's + /// maintenance) so that a `SharedSubDagStrategy` group's /// `cse_share_decision` flips exactly between a consumer count of 2 /// (recompute total 80, below maintenance: `RecomputeIndependently`) /// and a consumer count of 3 (recompute total 120, above @@ -9176,7 +9169,7 @@ mod tests { vec!["recompute", "different rewrite strategy", "share"], "the preferred CSE choice must be ranked without losing the unrelated rewrite" ); - let chosen = pick_shared_subtree_candidate( + let chosen = pick_shared_sub_dag_candidate( &group, decide_with_effective_count(&group, 2, &ConstantCseCost).unwrap(), ) @@ -9186,9 +9179,9 @@ mod tests { #[test] fn effective_consumer_count_corrects_a_nested_groups_share_decision() { - // The interaction issue #271 describes: an outer shared subtree `a` + // The interaction issue #271 describes: an outer shared sub-DAG `a` // (referenced by 2 roots, so consumer_count == 2) wraps an inner - // shared subtree `c` (referenced once through `a`'s own child edge, + // shared sub-DAG `c` (referenced once through `a`'s own child edge, // plus once more directly by a third, separate root — so `c`'s own // *raw* structural consumer_count is also 2, independent of `a`). // @@ -9199,11 +9192,11 @@ mod tests { // // `a` and `c` are both non-`Aggregate` nodes (`Filter`/`Dedup`) so // neither is bindable — each group is a *clean* two-candidate - // SharedSubtreeStrategy share-vs-recompute pair, with no + // SharedSubDagStrategy share-vs-recompute pair, with no // SketchAlgorithmStrategy `Summary` candidate mixed in to complicate // ranking (see `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // for what a *mixed*-shape group looks like — deliberately avoided - // here to isolate the SharedSubtreeStrategy-only interaction). + // here to isolate the SharedSubDagStrategy-only interaction). // // Under ConstantCseCost, consumer_count == 2 loses to maintenance // (2 * 40 = 80 < 100 ⇒ RecomputeIndependently); consumer_count == 3 wins @@ -9237,7 +9230,7 @@ mod tests { // Fixture sanity: root1/root2 merged onto one shared `a`, and `c` // (root1/root2's shared child, and root3 itself) merged onto one // shared `c` with raw consumer_count 2, and both groups are clean - // (non-mixed) two-candidate SharedSubtreeStrategy pairs. + // (non-mixed) two-candidate SharedSubDagStrategy pairs. assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let a_rc = &space.roots[0].1; let QueryExpr::Filter { child: c_via_a, .. } = a_rc.as_ref() else { @@ -9691,7 +9684,7 @@ mod tests { #[test] fn topological_order_puts_a_later_discovered_parent_before_its_child() { - // Mirrors nested_shared_subtree_below_an_unshared_parent_is_still_discovered's + // Mirrors nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered's // diamond fixture: discover_targets's own `order` visits root_b (a // parent of `shared`) *after* `shared` itself, because `shared` was // already fully walked via root_a first. A naive "process @@ -9731,7 +9724,7 @@ mod tests { order: order.clone(), composition_plans: Vec::new(), }; - let graph = reference_graph(&space); + let dag = reference_dag(&space); // Discovery-order sanity: root_b comes after the shared child in // discover_targets's own order (the exact non-topological case this @@ -9752,7 +9745,7 @@ mod tests { "fixture sanity: discover_targets's own order must NOT already be topological here" ); - let topo = topological_order(&order, &graph); + let topo = topological_order(&order, &dag); let shared_topo_pos = topo.iter().position(|p| *p == shared_ptr).unwrap(); let root_b_topo_pos = topo.iter().position(|p| *p == root_b_ptr).unwrap(); assert!( @@ -10188,7 +10181,7 @@ mod tests { #[test] fn nested_aggregates_realize_per_node() { // quantile(0.9, sum by (job) (m)) — the realization decision - // fires per node over the nested tree: KLL over an exact Sum + // fires per node over the nested DAG: KLL over an exact Sum // accumulator. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); @@ -10259,7 +10252,7 @@ mod tests { } } - /// The update expression of the first `SummaryAgg` in the tree. + /// The update expression of the first `SummaryAgg` in the DAG. fn find_summary_input(node: &SummaryNode) -> Option { match &node.expr { SummaryExpr::SummaryAgg { input, .. } if input.item.is_none() => { @@ -10274,7 +10267,7 @@ mod tests { fn pass_through_intents_stay_logical() { // avg is exact but non-mergeable; histogram_quantile (classic // buckets, #79) is never sketchable; exact quantile is exact by - // decree. All three stay whole logical subtrees. + // decree. All three stay whole logical sub-DAGs. for intent in [ AggIntent::Avg { col: None }, AggIntent::HistogramQuantile { q: 0.99, le: 0 }, @@ -10296,7 +10289,7 @@ mod tests { #[test] fn logical_parent_subsumes_bindable_child() { // Filter over a bindable quantile: `KeepPreAsap` has no post-ASAP - // children, so the conservative fallback keeps the whole subtree + // children, so the conservative fallback keeps the whole sub-DAG // logical. use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; use asap_types::pre_asap::query_expr::Predicate; @@ -10880,7 +10873,7 @@ mod tests { rejection.error ); } - // Fallback keeps the whole subtree pre-ASAP — executed exactly. + // Fallback keeps the whole sub-DAG pre-ASAP — executed exactly. let realized = realize_child(&outer, &DefaultCostModel).unwrap(); assert!(matches!(realized.expr, SummaryExpr::KeepPreAsap(_))); assert!(realized diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 06d1724fd..4b2d1a472 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -12,12 +12,12 @@ //! comment on why: `Avg`/`StdDev`/`Variance` "need richer partial state" //! than a bare sketch/exact accumulator gives, so there is no summary //! realization for a bare `avg` node to bind to at all. A logical `avg` -//! node therefore can never be a [`SharedSubtreeStrategy`] target either: +//! node therefore can never be a [`SharedSubDagStrategy`] target either: //! CSE-style sharing needs *some* mergeable accumulator underneath, and //! `PassThrough` has none. //! //! `Sum` and `Count` are both ordinary mergeable accumulators -//! (`agg_is_mergeable`) — exactly the shape [`SharedSubtreeStrategy`] and a +//! (`agg_is_mergeable`) — exactly the shape [`SharedSubDagStrategy`] and a //! future sketch-family search already know how to reuse across a //! workload. Rewriting `Aggregate{ measures: [Avg{col}], .. }` into two //! independent single-measure `Sum` and `Count` aggregates, divided with a @@ -43,7 +43,7 @@ //! Both are follow-ups (issue #253 itself scopes to "the concrete case in //! Peilin's comment"), not correctness bugs in what ships here — a node //! outside this scope simply doesn't `match`, the same "safe but -//! uninformative" fallback [`SketchAlgorithmStrategy`]/[`SharedSubtreeStrategy`] +//! uninformative" fallback [`SketchAlgorithmStrategy`]/[`SharedSubDagStrategy`] //! already use for shapes they don't have an opinion on. //! //! ## Non-goals (mirrors [`replacement`]'s own discipline) @@ -112,7 +112,7 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { Some((by.keys().len(), *col)) } -/// Build the rewritten `Project{ cast(sum) } / Aggregate{ Count }` tree for +/// Build the rewritten `Project{ cast(sum) } / Aggregate{ Count }` DAG for /// `root`, or `None` if `root` isn't [`avg_rewrite_target`]'s shape. `Sum` and /// `Count` deliberately live in separate, single-measure aggregates so the /// replacement fixpoint discovers each as an independently bindable target. @@ -362,7 +362,7 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option Option< /// 4. `finer_output_schema` (the finer aggregate's own *output* schema, not /// the shared child's) carries a provable unique key /// ([`Schema::has_unique_key`]) — **the exact legality gate -/// `pre_asap::cse::share_common_subtrees` already applies to its own +/// `pre_asap::cse::share_common_sub_dags` already applies to its own /// sharing decisions**, reused verbatim here rather than re-invented: -/// `share_common_subtrees`'s own doc ("Legality: gated by +/// `share_common_sub_dags`'s own doc ("Legality: gated by /// `Schema::unique_keys`") states a producer's output is only safely /// reusable across consumers when its row identity is provably stable — /// exactly the property re-aggregating over `finer` as if it were a /// fresh source requires. /// 5. `coarser_by` is a **strict, proper** subset of `finer_by` (same /// `ColumnId`s, finer strictly more of them) — an *equal* `by` is -/// `SharedSubtreeStrategy`'s CSE-sharing question, not a roll-up, so +/// `SharedSubDagStrategy`'s CSE-sharing question, not a roll-up, so /// equality is deliberately excluded here, not treated as a degenerate /// roll-up. pub fn is_legal_rollup_source( @@ -511,7 +511,7 @@ mod tests { #[test] fn predicate_rejects_equal_by_sets() { - // Equality is `SharedSubtreeStrategy`'s question, not a roll-up. + // Equality is `SharedSubDagStrategy`'s question, not a roll-up. let finer_schema = Schema::with_time_index(vec![], 0, vec![vec![0]]); assert!(!is_legal_rollup_source( &GroupKeys::by(vec![2]), @@ -827,7 +827,7 @@ mod tests { #[test] fn equal_by_sets_do_not_roll_up() { - // Equal groupings are `SharedSubtreeStrategy`'s CSE-sharing + // Equal groupings are `SharedSubDagStrategy`'s CSE-sharing // question (build once and share, or build independently) — a // roll-up requires a *strict* superset, not equality. let scan = Rc::new(metric_scan()); diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs index de20a94a0..a7272980f 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs @@ -229,7 +229,7 @@ impl SummaryOperatorEvidence { } } -/// Non-aggregation work for a retained pre-ASAP subtree over the comparison +/// Non-aggregation work for a retained pre-ASAP sub-DAG over the comparison /// horizon. Bootstrap/source I/O belongs exclusively to the owning aggregate, /// and summary insertion belongs exclusively to its insert evidence. #[derive(Debug, Clone, PartialEq)] diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs index 129d06ed3..6a1d2d3c4 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs @@ -1,6 +1,6 @@ //! Serializable DAG export for a materialized summary-maintenance plan. //! -//! `asap-types::dag_export` owns the crate-neutral post-ASAP graph shape. This +//! `asap-types::dag_export` owns the crate-neutral post-ASAP DAG shape. This //! adapter lives in the mapping layer, where summary-maintenance lifecycle //! alternatives and their typed rejection reasons are available, and emits //! both views together. @@ -10,7 +10,7 @@ use std::rc::Rc; use serde::Serialize; -use asap_types::dag_export::{self, SummaryDagGraph}; +use asap_types::dag_export::{self, SummaryDAG}; use asap_types::post_asap::{ PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, @@ -22,7 +22,7 @@ use crate::summary_maintenance_lifecycle::{ #[derive(Debug, Clone, Serialize)] pub struct SummaryMaintenanceDagExport { - pub graph: SummaryDagGraph, + pub dag: SummaryDAG, pub deployments: Vec, pub horizon_seconds: Option, pub evaluation_rate_per_second: Option, @@ -86,7 +86,7 @@ pub fn export_summary_maintenance_plan( .collect(), }) .collect(); - let mut graph = dag_export::export_summary(&plan.root); + let mut dag = dag_export::export_summary(&plan.root); let deployment_by_summary: HashMap<_, _> = plan .deployments .iter() @@ -96,13 +96,13 @@ pub fn export_summary_maintenance_plan( let mut next_node_id = 0; annotate_lifecycle_deployments( &plan.root, - &mut graph, + &mut dag, &deployment_by_summary, &mut next_node_id, ); SummaryMaintenanceDagExport { - graph, + dag, deployments, horizon_seconds: plan.horizon.map(|horizon| horizon.0), evaluation_rate_per_second: plan.evaluation_rate.map(|rate| rate.0), @@ -118,22 +118,22 @@ pub fn export_summary_maintenance_plan( /// Walk in the same post-order as `dag_export::export_summary` and attach a /// deployment directly to every flattened occurrence of its state node. -/// This makes the decision visible to graph consumers without asking them to -/// reconstruct pointer identity from graph position. +/// This makes the decision visible to DAG consumers without asking them to +/// reconstruct pointer identity from DAG position. fn annotate_lifecycle_deployments( node: &SummaryNode, - graph: &mut SummaryDagGraph, + dag: &mut SummaryDAG, deployments: &HashMap<*const SummaryNode, &SummaryMaintenanceDeploymentExport>, next_node_id: &mut usize, ) { if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { for child in summary_children(&node.expr) { - annotate_lifecycle_deployments(child, graph, deployments, next_node_id); + annotate_lifecycle_deployments(child, dag, deployments, next_node_id); } } - let graph_node = &mut graph.nodes[*next_node_id]; + let dag_node = &mut dag.nodes[*next_node_id]; if let Some(deployment) = deployments.get(&(node as *const SummaryNode)) { - graph_node.detail["summary_maintenance"] = + dag_node.detail["summary_maintenance"] = serde_json::to_value(deployment).expect("lifecycle export is serializable"); } *next_node_id += 1; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index 6bc5b13c7..27fa99d98 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -21,7 +21,7 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; use asap_types::post_asap::{ - compile_post_asap_dag_with_node_ids, share_common_summary_subtrees, EvaluationSchedule, + compile_post_asap_dag_with_node_ids, share_common_summary_sub_dags, EvaluationSchedule, ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDag, PostAsapDagValidationError, PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, @@ -800,7 +800,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( // Intern every member once; members whose outermost state (the // `SummaryAgg` every other state of the candidate feeds) interns to the // same node share it. Classes are kept in first-member order. - let interned = share_common_summary_subtrees( + let interned = share_common_summary_sub_dags( members .iter() .enumerate() diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 5f6384659..fb0001772 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -72,7 +72,7 @@ See [the design](../../docs/design_docs/physical-planning-and-deployment.md). ## Module boundaries -- `plan`: immutable graph, operator interface, schemas and execution properties. +- `plan`: immutable DAG, operator interface, schemas and execution properties. - `runtime`: per-run streams, shared producers, memory reservations and cancellation. - `expressions`: scalar evaluation; typed builders and the Planner expression adapter. - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. @@ -111,7 +111,7 @@ schemas, input ordering, sharing and boundedness before deployment source access A deployment calls `CompiledPhysicalDag::instantiate` with exactly the declared inputs. This checks source schemas and execution properties and constructs the -runnable graph without repeating logical lowering. The graph executes through +runnable DAG without repeating logical lowering. The DAG executes through the shared runtime with independent per-run state. Window coverage, revision and maintenance-policy admission remain deployment/planning contracts; this compiler does not discover storage or silently change a selected maintenance strategy. diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index 7cac8ef2b..9ef714c7a 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -1,4 +1,4 @@ -//! Compile maintenance-selected frontiers without deployment-specific graph rewrites. +//! Compile maintenance-selected frontiers without deployment-specific DAG rewrites. use super::*; /// One computation realization; lifecycle/window/revision requirements accompany diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index af0bbe393..cb2e230c9 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -35,7 +35,7 @@ enum Node { /// Selected native operators and input slots. Rebinding never repeats lowering. /// Serde is format-agnostic; deployments choose the encoding and its versioning. -/// Deserialization validates the graph before it is usable. +/// Deserialization validates the DAG before it is usable. #[derive(Clone, serde::Serialize, serde::Deserialize)] #[serde(try_from = "UncheckedDag")] pub struct CompiledPhysicalDag { @@ -63,7 +63,7 @@ impl TryFrom for CompiledPhysicalDag { impl CompiledPhysicalDag { /// Link already-selected physical fragments without lowering operators again. /// Fragment keys and source keys share a namespace; repeated dependency IDs - /// therefore remain one producer in the composed graph. + /// therefore remain one producer in the composed DAG. pub fn compose( sources: BTreeMap, fragments: BTreeMap, Self)>, @@ -286,7 +286,7 @@ impl CompiledPhysicalDag { &self, mut sources: BTreeMap>, ) -> Result, Error> { - let mut graph = PhysicalDag::default(); + let mut dag = PhysicalDag::default(); for (&id, node) in &self.nodes { match node { Node::Input(contract) => { @@ -305,7 +305,7 @@ impl CompiledPhysicalDag { "physical input {id} violates its compiled contract" ))); } - graph.add_boxed( + dag.add_boxed( id, vec![], Box::new(CheckedSource { @@ -315,15 +315,15 @@ impl CompiledPhysicalDag { )?; } Node::Operator { inputs, operator } => { - graph.add(id, inputs.clone(), operator.clone())?; + dag.add(id, inputs.clone(), operator.clone())?; } } } if !sources.is_empty() { return Err(invalid("unexpected physical input binding")); } - graph.validate(&self.roots)?; - Ok(graph) + dag.validate(&self.roots)?; + Ok(dag) } } impl PhysicalOperator for InputContract { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 72aed2720..30fddf375 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -114,7 +114,7 @@ thread_local! { } /// Helper operators are numbered from their Planner node alone, above the u32 -/// Planner ID range, so every boundary choice yields a subgraph of the same +/// Planner ID range, so every boundary choice yields a sub-DAG of the same /// lowering and candidate cuts need not renumber operators. A node lowering to /// several helpers takes consecutive indices below its base. fn helper_id(node: NodeId, index: u64) -> NodeId { @@ -210,7 +210,7 @@ fn compile_internal( } } } - let mut graph = CompiledPhysicalDag::new(roots.to_vec()); + let mut physical_dag = CompiledPhysicalDag::new(roots.to_vec()); for id in ordered { let node = nodes[&id]; let mut auxiliary = helper_id(id, 0); @@ -220,7 +220,7 @@ fn compile_internal( if source.schema != output { return Err(invalid("frontier does not have the declared schema")); } - graph.add_input(id, source)?; + physical_dag.add_input(id, source)?; } else { #[cfg(test)] LOWERED_NODES.with(|count| count.set(count.get() + 1)); @@ -233,7 +233,7 @@ fn compile_internal( if schemas.iter().any(|s| s != &schemas[0]) { return Err(invalid("summary merge inputs have different schemas")); } - graph.add( + physical_dag.add( auxiliary, inputs, Operator::union(schemas[0].clone(), schemas.len())?, @@ -260,7 +260,7 @@ fn compile_internal( let slot = promql_fallback::raw_series_input(id, i); match sources.remove(&slot) { Some(contract) if &contract.schema == schema => { - graph.add_input(slot, contract)? + physical_dag.add_input(slot, contract)? } Some(_) => { return Err(invalid(format!( @@ -289,11 +289,11 @@ fn compile_internal( .collect::>() }; for (operator, inputs) in steps { - graph.add(auxiliary, resolve(inputs, &ids), operator)?; + physical_dag.add(auxiliary, resolve(inputs, &ids), operator)?; ids.push(auxiliary); auxiliary -= 1; } - graph.add( + physical_dag.add( id, resolve(last_inputs, &ids), last.with_output_schema(output)?, @@ -328,7 +328,7 @@ fn compile_internal( let value = named_column(input, &ColumnRef::SampleValue)?; let lookback = i64::try_from(spec.lookback_ms) .map_err(|_| invalid("current-series lookback overflows"))?; - graph.add( + physical_dag.add( id, inputs, Operator::current_series(input.clone(), identity, coordinate, value, lookback)? @@ -369,11 +369,11 @@ fn compile_internal( let last = chain.pop().expect("nonempty chain"); let mut inputs = inputs; for operator in chain { - graph.add(auxiliary, inputs, operator)?; + physical_dag.add(auxiliary, inputs, operator)?; inputs = vec![auxiliary]; auxiliary -= 1; } - graph.add(id, inputs, last.with_output_schema(output)?)?; + physical_dag.add(id, inputs, last.with_output_schema(output)?)?; continue; }; let groups = spec @@ -382,7 +382,7 @@ fn compile_internal( .map(|name| named_column(&input, &ColumnRef::Named(name.clone()))) .collect::, _>>()?; let value = named_column(&input, &ColumnRef::SampleValue)?; - graph.add( + physical_dag.add( auxiliary, inputs, Operator::sort( @@ -395,7 +395,7 @@ fn compile_internal( groups.clone(), )?, )?; - graph.add( + physical_dag.add( id, vec![auxiliary], Operator::limit(input, *k as u64, 0, groups)?.with_output_schema(output)?, @@ -454,8 +454,8 @@ fn compile_internal( groups, )?; let compact = build.schema(); - graph.add(auxiliary, inputs, build)?; - graph.add( + physical_dag.add(auxiliary, inputs, build)?; + physical_dag.add( id, vec![auxiliary], Operator::scope_timestamp(compact, output)?, @@ -490,8 +490,8 @@ fn compile_internal( let [l, r] = sides; let binary = Operator::series_binary(l, r, operator.clone(), scalars) .map_err(|error| invalid(format!("node {id}: {error}")))?; - graph.add(auxiliary, vec![], scalar)?; - graph.add(id, operands, binary.with_output_schema(output)?)?; + physical_dag.add(auxiliary, vec![], scalar)?; + physical_dag.add(id, operands, binary.with_output_schema(output)?)?; auxiliary -= 1; continue; } @@ -520,7 +520,7 @@ fn compile_internal( [scalar(&inputs[0]), scalar(&inputs[1])], ) .map_err(|error| invalid(format!("node {id}: {error}")))?; - graph.add(id, inputs, binary.with_output_schema(output)?)?; + physical_dag.add(id, inputs, binary.with_output_schema(output)?)?; continue; } } @@ -555,16 +555,16 @@ fn compile_internal( .collect(); let project = Operator::project(actual, columns)?.with_output_schema(output.clone())?; - graph.add(auxiliary, inputs, readout)?; + physical_dag.add(auxiliary, inputs, readout)?; if temporal_readout_drops_name(node) { - graph.add(auxiliary - 1, vec![auxiliary], project)?; - graph.add( + physical_dag.add(auxiliary - 1, vec![auxiliary], project)?; + physical_dag.add( id, vec![auxiliary - 1], Operator::series_without_name(output)?, )?; } else { - graph.add(id, vec![auxiliary], project)?; + physical_dag.add(id, vec![auxiliary], project)?; } auxiliary -= 1; continue; @@ -600,15 +600,15 @@ fn compile_internal( } } if temporal_readout_drops_name(node) { - graph.add(auxiliary, inputs, operator)?; - graph.add(id, vec![auxiliary], Operator::series_without_name(output)?)?; + physical_dag.add(auxiliary, inputs, operator)?; + physical_dag.add(id, vec![auxiliary], Operator::series_without_name(output)?)?; } else { - graph.add(id, inputs, operator)?; + physical_dag.add(id, inputs, operator)?; } } } - graph.validate()?; - Ok(graph) + physical_dag.validate()?; + Ok(physical_dag) } // Temporal summary readouts produce PromQL vectors, whose range functions drop diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 0674ff4c2..7d2976211 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -226,7 +226,7 @@ pub fn compile( continue; } if node.output_state.timing != ExecutionTiming::IngestionTime { - return Err(invalid("precompute graph contains a query-time operation")); + return Err(invalid("precompute DAG contains a query-time operation")); } let inputs = dependencies.get(&id).cloned().unwrap_or_default(); let schemas = inputs @@ -238,13 +238,18 @@ pub fn compile( .ok_or_else(|| invalid("missing precompute input")) }) .collect::, _>>()?; - let graph = fragment( + let physical_dag = fragment( node, &schemas, &inputs.iter().map(|id| nodes[id]).collect::>(), )?; - outputs.insert(id, graph.output_contract(graph.roots()[0])?.schema); - fragments.insert(id, (inputs, graph)); + outputs.insert( + id, + physical_dag + .output_contract(physical_dag.roots()[0])? + .schema, + ); + fragments.insert(id, (inputs, physical_dag)); } CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) } diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index 3d92f0278..7f6ac49c1 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -1,4 +1,4 @@ -//! Compile a retained PromQL subtree (`Fallback`) from its typed expression. +//! Compile a retained PromQL sub-DAG (`Fallback`) from its typed expression. //! The deployment supplies the raw series of each selector; the Planner //! computes selection, range functions, subqueries, matching and aggregation. use super::*; diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs index 0dce4dd8e..18016043e 100644 --- a/crates/asap-physical-operators/src/plan/mod.rs +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -1,4 +1,4 @@ -//! Immutable physical graph, operator contracts and pre-execution validation. +//! Immutable physical DAG, operator contracts and pre-execution validation. use crate::{ runtime::{Input, OutputStream, RunContext}, Error, diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index f63004401..d89b75e7b 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -17,18 +17,18 @@ pub fn evaluate_batch( operators: Vec, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); - graph.add( + let mut dag = PhysicalDag::default(); + dag.add( 0, vec![], Operator::source(input.schema().clone(), vec![input])?, )?; let mut root = 0; for operator in operators { - graph.add(root + 1, vec![root], operator)?; + dag.add(root + 1, vec![root], operator)?; root += 1; } - evaluate_graph(graph, root, context) + evaluate_dag(dag, root, context) } /// Bind the ordered in-memory inputs of a native multi-input operator. @@ -37,17 +37,17 @@ pub fn evaluate_inputs( operator: Operator, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); + let mut dag = PhysicalDag::default(); let root = inputs.len() as u64; for (id, input) in inputs.into_iter().enumerate() { - graph.add( + dag.add( id as u64, vec![], Operator::source(input.schema().clone(), vec![input])?, )?; } - graph.add(root, (0..root).collect(), operator)?; - evaluate_graph(graph, root, context) + dag.add(root, (0..root).collect(), operator)?; + evaluate_dag(dag, root, context) } /// Evaluate a native in-memory source, including scalar sources, in the caller's scope. @@ -55,17 +55,17 @@ pub fn evaluate_source( source: Operator, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); - graph.add(0, vec![], source)?; - evaluate_graph(graph, 0, context) + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source)?; + evaluate_dag(dag, 0, context) } -fn evaluate_graph( - graph: PhysicalDag<'_, Batch, crate::values::Schema>, +fn evaluate_dag( + dag: PhysicalDag<'_, Batch, crate::values::Schema>, root: crate::plan::NodeId, context: RunContext, ) -> Result>, Error> { - let mut output = graph.execute(&[root], context)?.remove(0); + let mut output = dag.execute(&[root], context)?.remove(0); let mut batches = Vec::new(); loop { match output.next().now_or_never() { diff --git a/crates/asap-physical-operators/src/runtime/tests.rs b/crates/asap-physical-operators/src/runtime/tests.rs index f683041c6..1e4aea749 100644 --- a/crates/asap-physical-operators/src/runtime/tests.rs +++ b/crates/asap-physical-operators/src/runtime/tests.rs @@ -203,9 +203,9 @@ fn retained_outputs_count_against_budget() { assert_eq!(run.retained_bytes(), 0); } -// Invalid graphs fail before even starting a source. +// Invalid DAGs fail before even starting a source. #[test] -fn invalid_graphs_do_not_start_sources() { +fn invalid_dags_do_not_start_sources() { let (source, starts, _) = source(false); let mut dag = PhysicalDag::default(); dag.add(0, vec![], source).unwrap(); diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 322bd6fee..c8bc3a351 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -36,7 +36,7 @@ fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, @@ -51,7 +51,7 @@ fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result Err(error), Ok(mut streams) => block_on(streams.remove(0).next()).unwrap().map(|_| ()), }; diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 3a92e4d96..cbe66c721 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -171,7 +171,7 @@ fn execute_relabeled( ) }) .collect(); - let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let physical_dag = program.instantiate(sources).map_err(|e| e.to_string())?; let context = RunContext::new( Scope::Query { evaluation_time_ms: end, @@ -181,7 +181,7 @@ fn execute_relabeled( ) .unwrap(); block_on(async { - let mut stream = graph + let mut stream = physical_dag .execute(program.roots(), context) .map_err(|e| e.to_string())? .remove(0); @@ -694,7 +694,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { }) .collect(); let batch = Batch::try_new(schema.clone(), vec![row]).unwrap(); - let graph = program + let physical_dag = program .instantiate(BTreeMap::from([( u64::from(state.id.0), Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, @@ -709,7 +709,10 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { ) .unwrap(); let values = block_on(async { - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let mut values = Vec::new(); while let Some(batch) = stream.next().await { values.extend(batch.unwrap().rows().iter().map(|row| row[0].clone())); diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 652880e04..01bf7f743 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -1266,7 +1266,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { }), }, }; - let graph = CompiledPhysicalDag::from_operators( + let dag = CompiledPhysicalDag::from_operators( [ (0, InputContract::bounded(schema.clone())), (1, InputContract::bounded(schema.clone())), @@ -1283,11 +1283,10 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { vec![2], ) .unwrap(); - let graph = - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); + let dag = serde_json::from_slice::(&serde_json::to_vec(&dag).unwrap()) + .unwrap(); assert_eq!( - graph.certified_pruning_keys(2), + dag.certified_pruning_keys(2), certified.then_some(&[(0, 0)][..]) ); for complete in [false, true] { @@ -1315,11 +1314,11 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { ) }) .collect::>(); - let bound = graph.instantiate(sources).unwrap(); + let bound = dag.instantiate(sources).unwrap(); let result = block_on(async { let mut stream = bound .execute( - graph.roots(), + dag.roots(), RunContext::new(query(), Limits::default()).unwrap(), ) .unwrap() @@ -1422,9 +1421,9 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ) }) .collect::>(); - let graph = program.instantiate(sources).unwrap(); + let dag = program.instantiate(sources).unwrap(); let result = block_on(async { - let mut stream = graph + let mut stream = dag .execute( program.roots(), RunContext::new(query(), Limits::default()).unwrap(), diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index fa124c686..89562c6d4 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -1,4 +1,4 @@ -//! Persisted precompute graphs preserve group/window identity and execute state-to-state computation. +//! Persisted precompute DAGs preserve group/window identity and execute state-to-state computation. use asap_physical_operators::{ factory::create_planner_accumulator, operators::Operator, @@ -175,7 +175,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) as Source<'_>, )]); - let graph = program.instantiate(sources).unwrap(); + let physical_dag = program.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Ingestion { window_start_ms: 0, @@ -186,7 +186,10 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { ) .unwrap(); let output = block_on(async { - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let batch = stream.next().await.unwrap().unwrap(); assert!(stream.next().await.is_none()); batch @@ -221,7 +224,7 @@ fn logical_schema(family: SummaryFamilyType) -> SummarySchema { time_index: None, } } -fn state_graph( +fn state_dag( family: SummaryFamilyType, target: Option, merge: bool, @@ -308,13 +311,13 @@ fn native_run( }) .collect(); let input = Batch::try_new(precompute::population_schema(family), rows)?; - let graph = program.instantiate(BTreeMap::from([( + let physical_dag = program.instantiate(BTreeMap::from([( 0, Box::new(Operator::source(input.schema().clone(), vec![input])?) as Source<'_>, )]))?; block_on(async { let mut rows = Vec::new(); - let mut stream = graph.execute(program.roots(), context)?.remove(0); + let mut stream = physical_dag.execute(program.roots(), context)?.remove(0); while let Some(batch) = stream.next().await { rows.extend(batch?.rows().iter().cloned()); } @@ -350,7 +353,7 @@ fn sum_state(value: f64) -> Arc { fn explicit_merge_changes_pane_cardinality() { let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { - let program = state_graph(family.clone(), None, merge); + let program = state_dag(family.clone(), None, merge); let rows = native_run( &program, family.clone(), @@ -381,7 +384,7 @@ fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { ), GroupingStrategy::default(), ); - let program = state_graph(source.clone(), Some(target), false); + let program = state_dag(source.clone(), Some(target), false); assert!(native_run( &program, source.clone(), @@ -405,7 +408,7 @@ fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { fn precompute_count_conversion_checks_precision() { use asap_physical_operators::summary_kernels::exact::ExactAccumulator; let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); - let program = state_graph(family.clone(), None, false); + let program = state_dag(family.clone(), None, false); for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { let mut state = serde_json::to_value(ExactAccumulator::new(family.clone(), false).unwrap()).unwrap(); @@ -421,11 +424,11 @@ fn precompute_count_conversion_checks_precision() { } } -// Graph execution retains terminal cancellation and shared workspace limits. +// DAG execution retains terminal cancellation and shared workspace limits. #[test] -fn precompute_graph_enforces_cancellation_and_budget() { +fn precompute_dag_enforces_cancellation_and_budget() { let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let program = state_graph(family.clone(), None, true); + let program = state_dag(family.clone(), None, true); let context = ingestion_context(Limits::default()); context.cancel(); let error = native_run(&program, family.clone(), vec![sum_state(1.)], context).unwrap_err(); diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index 822bde4ea..ba86f7224 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -66,7 +66,7 @@ fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { guarantee: None, }; let operator = compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(); - let graph = CompiledPhysicalDag::from_operators( + let physical_dag = CompiledPhysicalDag::from_operators( BTreeMap::from([ (0, InputContract::bounded(schema.clone())), (1, InputContract::bounded(schema)), @@ -75,7 +75,8 @@ fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { vec![2], ) .unwrap(); - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()).unwrap() + serde_json::from_slice::(&serde_json::to_vec(&physical_dag).unwrap()) + .unwrap() } fn evaluate( left: Vec>, @@ -84,7 +85,7 @@ fn evaluate( evaluate_with(program(), left, right) } fn evaluate_with( - graph: CompiledPhysicalDag, + physical_dag: CompiledPhysicalDag, left: Vec>, right: Vec>, ) -> Result>, asap_physical_operators::Error> { @@ -99,7 +100,7 @@ fn evaluate_with( ) }) .collect(); - let bound = graph.instantiate(sources)?; + let bound = physical_dag.instantiate(sources)?; let ctx = RunContext::new( Scope::Query { evaluation_time_ms: 1, @@ -150,7 +151,7 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { use asap_physical_operators::physical_planner::promql_values; use planner_types::pre_asap::CompareOpKind; for return_bool in [false, true] { - let graph = promql_values::compile_binary( + let physical_dag = promql_values::compile_binary( &BinaryOperator { kind: BinaryOpKind::Compare(CompareOpKind::Lt), vector_match: None, @@ -162,9 +163,10 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { false, ) .unwrap(); - let graph = - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); + let physical_dag = serde_json::from_slice::( + &serde_json::to_vec(&physical_dag).unwrap(), + ) + .unwrap(); let scalar = promql_values::scalar_schema(); let vector = promql_values::vector_schema(); let sources = BTreeMap::from([ @@ -193,7 +195,7 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { ) as Source<'_>, ), ]); - let bound = graph.instantiate(sources).unwrap(); + let bound = physical_dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 1, @@ -232,7 +234,7 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { #[test] fn binary_obeys_memory_and_cancellation() { for cancel in [false, true] { - let graph = program(); + let physical_dag = program(); let sources = (0..2) .map(|id| { ( @@ -247,7 +249,7 @@ fn binary_obeys_memory_and_cancellation() { ) }) .collect(); - let bound = graph.instantiate(sources).unwrap(); + let bound = physical_dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 1, @@ -392,7 +394,7 @@ fn stored_series_readouts_support_filters_and_sets() { edges, root: PostAsapNodeId(4), }; - let graph = compile( + let physical_dag = compile( &dag, BTreeMap::from([ (0, InputContract::bounded(state_schema.clone())), @@ -401,8 +403,8 @@ fn stored_series_readouts_support_filters_and_sets() { &[4], ) .unwrap(); - let graph: CompiledPhysicalDag = - serde_json::from_slice(&serde_json::to_vec(&graph).unwrap()).unwrap(); + let physical_dag: CompiledPhysicalDag = + serde_json::from_slice(&serde_json::to_vec(&physical_dag).unwrap()).unwrap(); let sources = [(0, "a", 6.), (1, "b", 2.)] .into_iter() .map(|(id, name, value)| { @@ -437,7 +439,7 @@ fn stored_series_readouts_support_filters_and_sets() { ) }) .collect(); - let bound = graph.instantiate(sources).unwrap(); + let bound = physical_dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 1, diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index 88fe3690f..a696df7f0 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -1,4 +1,4 @@ -//! A retained PromQL subtree (`Fallback`) compiles from its typed expression. +//! A retained PromQL sub-DAG (`Fallback`) compiles from its typed expression. //! The deployment supplies only its selector's raw series; expected values are //! hand-computed with Prometheus semantics. use asap_physical_operators::{ @@ -168,7 +168,7 @@ fn evaluate_dag_with_range( Box::new(Operator::source(schema, vec![batch]).unwrap()) as _, ); } - let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let physical_dag = program.instantiate(sources).map_err(|e| e.to_string())?; let context = RunContext::new( Scope::Query { evaluation_time_ms: at * 1000, @@ -184,7 +184,7 @@ fn evaluate_dag_with_range( None => context, }; block_on(async { - let mut stream = graph + let mut stream = physical_dag .execute(program.roots(), context) .map_err(|e| e.to_string())? .remove(0); diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index 886b7d00d..f9742ebfd 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -21,15 +21,15 @@ fn row(labels: &[(&str, &str)], value: f64) -> Vec { Value::Float64(value), ] } -fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { - run_inputs(graph, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() +fn run(dag: CompiledPhysicalDag, rows: Vec>) -> Vec> { + run_inputs(dag, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() } fn run_inputs( - graph: CompiledPhysicalDag, + dag: CompiledPhysicalDag, batches: Vec, ) -> Result>, asap_physical_operators::Error> { - let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); + let dag = + serde_json::from_slice::(&serde_json::to_vec(&dag).unwrap()).unwrap(); let sources = batches .into_iter() .enumerate() @@ -41,7 +41,7 @@ fn run_inputs( ) }) .collect::>(); - let bound = graph.instantiate(sources).unwrap(); + let bound = dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 0, @@ -51,7 +51,7 @@ fn run_inputs( ) .unwrap(); block_on(async { - let mut stream = bound.execute(graph.roots(), context).unwrap().remove(0); + let mut stream = bound.execute(dag.roots(), context).unwrap().remove(0); let mut rows = Vec::new(); while let Some(batch) = stream.next().await { rows.extend(batch?.rows().iter().cloned()); @@ -138,10 +138,10 @@ fn empty_vector_aggregation_stays_empty() { assert!(matches!(scalar[0][0],Value::Float64(v) if v.is_nan())); } -// One persisted temporal graph accepts different request windows and detects resets. +// One persisted temporal DAG accepts different request windows and detects resets. #[test] -fn temporal_graph_uses_bound_window_without_recompilation() { - let graph = compile_temporal(&AggIntent::Rate, false).unwrap(); +fn temporal_dag_uses_bound_window_without_recompilation() { + let dag = compile_temporal(&AggIntent::Rate, false).unwrap(); for start in [0, 60_000] { let labels = row(&[("__name__", "counter"), ("job", "api")], 0.)[0].clone(); let samples = [(0, 5.), (30_000, 1.), (60_000, 7.)]; @@ -158,7 +158,7 @@ fn temporal_graph_uses_bound_window_without_recompilation() { }) .collect(); let output = run_inputs( - graph.clone(), + dag.clone(), vec![Batch::try_new(matrix_schema(), rows).unwrap()], ) .unwrap(); @@ -181,20 +181,20 @@ fn temporal_graph_uses_bound_window_without_recompilation() { Value::Timestamp(2000), ], ]; - assert!(run_inputs(graph, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); + assert!(run_inputs(dag, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); } // The quantile is an ordinary scalar input, and bucket labels are native computation. #[test] fn histogram_quantile_keeps_each_label_group() { - let graph = compile_histogram_quantile().unwrap(); + let dag = compile_histogram_quantile().unwrap(); let buckets = vec![ row(&[("job", "api"), ("le", "1")], 2.), row(&[("job", "api"), ("le", "2")], 4.), row(&[("job", "api"), ("le", "+Inf")], 4.), ]; let output = run_inputs( - graph, + dag, vec![ Batch::try_new(scalar_schema(), vec![vec![Value::Float64(0.75)]]).unwrap(), Batch::try_new(vector_schema(), buckets).unwrap(), @@ -263,7 +263,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { false, ) .unwrap(); - let graph = CompiledPhysicalDag::compose( + let dag = CompiledPhysicalDag::compose( BTreeMap::from([(0, InputContract::bounded(vector_schema()))]), BTreeMap::from([ (10, (vec![0], aggregate)), @@ -273,9 +273,9 @@ fn composed_ensemble_shares_a_producer_across_roots() { vec![20, 30], ) .unwrap(); - let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); - assert_eq!(graph.input_contracts().count(), 1); + let dag = + serde_json::from_slice::(&serde_json::to_vec(&dag).unwrap()).unwrap(); + assert_eq!(dag.input_contracts().count(), 1); let starts = std::rc::Rc::new(std::cell::Cell::new(0)); for _ in 0..2 { let input = Batch::try_new(vector_schema(), vec![row(&[("job", "api")], 3.)]).unwrap(); @@ -283,7 +283,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { source: Operator::source(vector_schema(), vec![input]).unwrap(), starts: starts.clone(), }; - let bound = graph + let bound = dag .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) .unwrap(); let context = RunContext::new( @@ -300,7 +300,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { let results = block_on(futures::future::join_all( bound - .execute(graph.roots(), context) + .execute(dag.roots(), context) .unwrap() .into_iter() .map(|mut stream| async move { @@ -315,9 +315,9 @@ fn composed_ensemble_shares_a_producer_across_roots() { #[test] fn compiled_constant_needs_no_deployment_source() { - let graph = compile_scalar(3.).unwrap(); - assert_eq!(graph.input_contracts().count(), 0); - let result = run_inputs(graph, vec![]).unwrap(); + let dag = compile_scalar(3.).unwrap(); + assert_eq!(dag.input_contracts().count(), 0); + let result = run_inputs(dag, vec![]).unwrap(); assert!(matches!(result[0][0], Value::Float64(3.))); } @@ -335,7 +335,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { (BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), false), (BinaryOpKind::Compare(CompareOpKind::Gt), true), ] { - let graph = compile_binary( + let dag = compile_binary( &BinaryOperator { kind, vector_match: None, @@ -358,7 +358,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { let scalar = Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(); let result = run_inputs( - graph, + dag, if left_scalar { vec![scalar, vector] } else { @@ -369,7 +369,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { } } } - let graph = compile_binary( + let dag = compile_binary( &BinaryOperator { kind: BinaryOpKind::Compare(CompareOpKind::Gt), vector_match: None, @@ -387,7 +387,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { ]; equal_rows( run_inputs( - graph, + dag, vec![ Batch::try_new(vector_schema(), rows.clone()).unwrap(), Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(), @@ -398,7 +398,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { ); } -// Persisted exact readout graphs, rather than the storage adapter, merge panes, +// Persisted exact readout DAGs, rather than the storage adapter, merge panes, // finalize each population, and preserve the requested metric-name semantics. #[test] fn exact_state_readouts_recover_and_finalize_panes() { diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 78e4c8c02..60b73f244 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -264,13 +264,13 @@ fn schema_drift_and_memory_limits_fail_the_scan() { batch: bad, })); let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); - let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let physical_dag = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); block_on(async { - let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut s = physical_dag.execute(&[0], context()).unwrap().remove(0); assert!(s.next().await.unwrap().is_err()); }); let sources = registry(Arc::new(MemorySource::new(schema, batches).unwrap())); - let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let physical_dag = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); let ctx = RunContext::new( Scope::Query { evaluation_time_ms: 0, @@ -283,7 +283,7 @@ fn schema_drift_and_memory_limits_fail_the_scan() { ) .unwrap(); block_on(async { - let mut s = graph.execute(&[0], ctx.clone()).unwrap().remove(0); + let mut s = physical_dag.execute(&[0], ctx.clone()).unwrap().remove(0); assert!(s.next().await.unwrap().is_err()); }); assert_eq!(ctx.retained_bytes(), 0); @@ -318,9 +318,9 @@ fn empty_sources_and_three_valued_predicates() { ) .unwrap(); let plan = plan(scan.clone(), &schema, ExecutionDataState::QUERY_ROWS); - let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let physical_dag = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); block_on(async { - let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut s = physical_dag.execute(&[0], context()).unwrap().remove(0); let mut count = 0; while let Some(b) = s.next().await { count += b.unwrap().rows().len(); @@ -351,8 +351,8 @@ fn compile_without_readers_and_rebind_inputs() { 0, Box::new(Operator::source(schema.clone(), batches.clone()).unwrap()) as Source<'_>, )]); - let graph = compiled.instantiate(sources).unwrap(); - let mut outputs = graph.execute(compiled.roots(), context()).unwrap(); + let physical_dag = compiled.instantiate(sources).unwrap(); + let mut outputs = physical_dag.execute(compiled.roots(), context()).unwrap(); let result = block_on(outputs.remove(0).collect::>()); assert!(result.iter().all(Result::is_ok)); assert_eq!( diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index a61a596cb..72664b527 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -131,7 +131,7 @@ fn post_asap_summary_projection_survives_recovery() { ]], ) .unwrap(); - let graph = program + let physical_dag = program .instantiate(BTreeMap::from([( 0, Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, @@ -146,7 +146,7 @@ fn post_asap_summary_projection_survives_recovery() { ) .unwrap(); block_on(async { - let mut output = graph.execute(&[1], context).unwrap().remove(0); + let mut output = physical_dag.execute(&[1], context).unwrap().remove(0); let batch = output.next().await.unwrap().unwrap(); assert!(matches!(&batch.rows()[0][0], Value::Utf8(label) if label.as_ref() == "api")); let Value::Summary { diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index ef086b1bf..d08e60e6b 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -159,13 +159,13 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S &[dag.root.0 as u64], ) .unwrap(); - let graph = compiled + let physical_dag = compiled .instantiate(BTreeMap::from([(rate_id.0 as u64, source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let output = block_on(async { let mut output = Vec::new(); - let mut stream = graph + let mut stream = physical_dag .execute(&[dag.root.0 as u64], context) .unwrap() .remove(0); @@ -468,13 +468,13 @@ fn check_direct_rate_topk(dynamic: bool) { let source = Box::new( Operator::source(raw_schema.clone(), vec![raw_batch.clone()]).unwrap(), ) as Source<'static>; - let graph = raw_compiled + let physical_dag = raw_compiled .instantiate(BTreeMap::from([(u64::from(raw.id.0), source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let mut raw_scores = block_on(async { let mut scores = Vec::new(); - let mut stream = graph + let mut stream = physical_dag .execute(&[u64::from(dag.root.0)], context) .unwrap() .remove(0); @@ -584,13 +584,13 @@ fn check_direct_rate_topk(dynamic: bool) { let source = Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) as Source<'static>; - let graph = compiled + let physical_dag = compiled .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let mut scores = block_on(async { let mut scores = vec![]; - let mut stream = graph + let mut stream = physical_dag .execute(&[u64::from(dag.root.0)], context) .unwrap() .remove(0); @@ -707,7 +707,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { }) .collect(); let batch = Batch::try_new(schema.clone(), rows).unwrap(); - let graph = program + let physical_dag = program .instantiate(BTreeMap::from([( u64::from(raw.id.0), Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, @@ -722,7 +722,10 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { Limits::default(), ) .unwrap(); - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let mut result = Vec::new(); while let Some(batch) = stream.next().await { let batch = batch.unwrap(); @@ -928,9 +931,9 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { let id = plan.input_contracts().next().unwrap().0; let source = Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) as Source<'static>; - let graph = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); + let physical_dag = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); block_on(async { - let mut stream = graph + let mut stream = physical_dag .execute( plan.roots(), RunContext::new(scope, Limits::default()).unwrap(), diff --git a/crates/devtools/src/bin/analyze_corpora.rs b/crates/devtools/src/bin/analyze_corpora.rs index 503fb1f04..2c4bacefb 100644 --- a/crates/devtools/src/bin/analyze_corpora.rs +++ b/crates/devtools/src/bin/analyze_corpora.rs @@ -529,7 +529,7 @@ async fn run_sql_corpora(out_dir: PathBuf) { serde_json::to_vec_pretty(&summary).unwrap(), ) .expect("failed to write summary.json"); - let notes = "- **Manual review result:** the reviewed SQL trees preserve `COUNT(*)` versus `COUNT(column)`, `COUNT(DISTINCT)`/`uniqExact`, `HAVING`, CTE-derived projections, `LAG`/`lagInFrame`, grouping keys, and explicit window frames. No concrete semantic collapse was found in this pass.\n- **Apparently intentional omission:** ClickHouse `FORMAT Null` is absent from the IR; it is an output/transport directive rather than query semantics.\n- **Failure boundaries:** unsupported ClickHouse functions and unsupported grammar are retained in the per-query error files rather than being converted into partial IR.\n"; + let notes = "- **Manual review result:** the reviewed SQL DAGs preserve `COUNT(*)` versus `COUNT(column)`, `COUNT(DISTINCT)`/`uniqExact`, `HAVING`, CTE-derived projections, `LAG`/`lagInFrame`, grouping keys, and explicit window frames. No concrete semantic collapse was found in this pass.\n- **Apparently intentional omission:** ClickHouse `FORMAT Null` is absent from the IR; it is an output/transport directive rather than query semantics.\n- **Failure boundaries:** unsupported ClickHouse functions and unsupported grammar are retained in the per-query error files rather than being converted into partial IR.\n"; std::fs::write( out_dir.join("anomalies.md"), anomaly_report(&all, "SQL", notes), diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index 3f7cd8850..acf424031 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -3,7 +3,7 @@ // --promql "topk(5, rate(http_requests_total[5m]))" --name q2 // // Lowers each given SQL/PromQL query to pre-ASAP IR and prints a single -// `asap_types::dag_export::WorkloadGraph` as JSON on stdout — the input format +// `asap_types::dag_export::WorkloadDAG` as JSON on stdout — the input format // for `tools/dag-viewer` (issue #133). Redirect to a file and load it there: // cargo run -p asap-lower --bin dag_export -- --sql "..." --name q1 > /tmp/dag.json // @@ -31,23 +31,23 @@ // candidate per group feeds two additive outputs: // // - one `asap_types::dag_export::TargetReplacement` per group on whichever -// query's `NamedGraph.replacements` contains that target node (matched +// query's `NamedDAG.replacements` contains that target node (matched // by `DagNode::hash` + structural equality, the same collision-safe // pattern `annotate_with_explanations` below already uses for notes) — // a small, self-contained "before -> after" pair per replacement site; -// - one merged `NamedGraph.post_graph`: a single flattened graph per +// - one merged `NamedDAG.post_dag`: a single flattened DAG per // query with every winning candidate spliced directly into the query's // own pre-ASAP shape in place, built via // `asap_types::dag_export::export_post_asap`. // // Together these surface every one of the four concrete replacement kinds: // the sketch family `SketchAlgorithmStrategy`/`HydraGroupingStrategy` bound, -// the CSE share/recompute choice `SharedSubtreeStrategy` found, the +// the CSE share/recompute choice `SharedSubDagStrategy` found, the // workload-aware roll-up `RollupStrategy` derived, and the `avg -> // sum/count` rewrite `AvgToSumOverCountStrategy` proposes. Without // `--post-asap`, every existing invocation of this binary produces -// byte-identical output to before (`NamedGraph.replacements` is empty and -// `post_graph` is `None`, both skipped from the JSON entirely in that +// byte-identical output to before (`NamedDAG.replacements` is empty and +// `post_dag` is `None`, both skipped from the JSON entirely in that // case). E.g.: // cargo run -p asap-lower --bin dag_export -- \ // --post-asap --default-cost --epsilon 0.01 \ @@ -94,8 +94,8 @@ use asap_aware_mapping::replacement::{ use asap_aware_mapping::{AccuracyEvidenceProvider, PropagationStats}; use asap_types::cost::{BaselineRef, CostAnnotation, CostInput, CostSource, CostUnit}; use asap_types::dag_export::{ - self, DagDecision, DagGraph, DagNote, NamedGraph, PostAsapSubstitution, TargetRejection, - TargetReplacement, TargetReplacementAfter, WorkloadGraph, + self, DagDecision, DagNote, ExportDAG, NamedDAG, PostAsapSubstitution, TargetRejection, + TargetReplacement, TargetReplacementAfter, WorkloadDAG, }; use asap_types::post_asap::SummaryExpr; use asap_types::post_asap::SummaryNode; @@ -249,7 +249,7 @@ impl CandidatePhysicalEvidence { /// Compare complete exported-plan identity while tolerating the one-ULP /// decimal round trip that `serde_json::Value` can introduce for derived -/// floating-point guarantees. Integer configuration and graph identity stay +/// floating-point guarantees. Integer configuration and DAG identity stay /// exact; no guarantee field is dropped or otherwise normalized away. fn plan_values_match(actual: &serde_json::Value, expected: &serde_json::Value) -> bool { plan_values_match_inner(actual, expected, false) @@ -960,19 +960,19 @@ fn parse_args_from(argv: impl Iterator) -> ParsedArgs { } } -/// Attach workload-wide replacement explanations to their exact graph nodes. +/// Attach workload-wide replacement explanations to their exact DAG nodes. /// `node_hash` is only a narrowing filter; `source_expr == Some(target)` is /// the collision-safe identity check (`source_expr` is `None` only for a -/// post-ASAP-originated node inside a `--post-asap` `post_graph`, which this +/// post-ASAP-originated node inside a `--post-asap` `post_dag`, which this /// function is never called on — every node it sees, from an ordinary /// [`dag_export::export`], carries `Some`). fn annotate_with_explanations( - graph: &mut DagGraph, + dag: &mut ExportDAG, explanations: &[asap_aware_mapping::ReplacementExplanation], matched: &mut [bool], ) { for (i, explanation) in explanations.iter().enumerate() { - for node in graph.nodes.iter_mut() { + for node in dag.nodes.iter_mut() { if node.hash == Some(explanation.node_hash) && node.source_expr.as_ref() == Some(explanation.target.as_ref()) { @@ -988,7 +988,7 @@ fn annotate_with_explanations( /// One `TargetSubDAGCandidates`'s best-ranked candidate, kept alongside its own `target` /// — the unit both [`PostAsapResults::replacements`] and -/// [`PostAsapResults::post_graphs`] are built from, so the two outputs can +/// [`PostAsapResults::post_dags`] are built from, so the two outputs can /// never disagree about which candidate won for a given target. #[allow(dead_code)] struct Winner<'a> { @@ -1012,15 +1012,15 @@ fn decision_rationale(winner: &Winner<'_>) -> String { "Composes compatible nested aggregates using their declared algebraic intent while preserving the output schema." .to_string() } - "SharedSubtreeStrategy" => match winner.candidate.provenance { + "SharedSubDAGStrategy" => match winner.candidate.provenance { asap_aware_mapping::replacement::ReplacementProvenance::CseShare => { - "Builds the repeated subtree once and shares it across consumers.".to_string() + "Builds the repeated sub-DAG once and shares it across consumers.".to_string() } asap_aware_mapping::replacement::ReplacementProvenance::CseRecompute => { - "Recomputes the subtree per consumer because that has the lower estimated cost." + "Recomputes the sub-DAG per consumer because that has the lower estimated cost." .to_string() } - _ => "Chooses the lowest-cost handling of the repeated subtree.".to_string(), + _ => "Chooses the lowest-cost handling of the repeated sub-DAG.".to_string(), }, "RollupStrategy" => { "Answers this aggregate from a compatible finer-grained aggregate.".to_string() @@ -1067,15 +1067,15 @@ fn lookup_winner( } /// One `(decision.id, baseline_cost, selected_cost)` triple per *distinct* -/// [`DagDecision`] carried anywhere in `graph` — collapsing every node that +/// [`DagDecision`] carried anywhere in `dag` — collapsing every node that /// shares one `decision.id` (a replacement region can span many nodes, all /// carrying an identical clone of the same decision) down to a single /// entry, so a caller summing these never counts one decision's cost once /// per node it happens to touch. -fn decision_cost_entries(graph: &DagGraph) -> Vec<(u32, CostAnnotation, CostAnnotation)> { +fn decision_cost_entries(dag: &ExportDAG) -> Vec<(u32, CostAnnotation, CostAnnotation)> { let mut seen = std::collections::HashSet::new(); let mut entries = Vec::new(); - for node in &graph.nodes { + for node in &dag.nodes { let Some(decision) = &node.decision else { continue; }; @@ -1135,44 +1135,44 @@ fn target_replacement( /// usage doc for what each is for. struct PostAsapResults { /// One `(query_name, TargetReplacement)` pair per discovered replacement - /// site whose target node is found in that query's own exported graph. A + /// site whose target node is found in that query's own exported DAG. A /// target can in principle be reachable from more than one query's root - /// after CSE (a shared subtree), in which case it yields one pair per + /// after CSE (a shared sub-DAG), in which case it yields one pair per /// matching query, each with that query's own `target_pre_id`. replacements: Vec<(String, TargetReplacement)>, - /// One merged, whole-query [`DagGraph`] per query, built via + /// One merged, whole-query [`ExportDAG`] per query, built via /// [`dag_export::export_post_asap`] — every winning candidate spliced /// directly into that query's own pre-ASAP shape in place. - post_graphs: Vec<(String, DagGraph)>, + post_dags: Vec<(String, ExportDAG)>, /// One `(query_name, TargetRejection)` per accuracy-illegal candidate /// the search refused (`TargetSubDAGCandidates::rejected`, issue #172) whose target - /// node is found in that query's own exported graph. + /// node is found in that query's own exported DAG. rejections: Vec<(String, TargetRejection)>, } fn raw_only_post_asap_results() -> PostAsapResults { PostAsapResults { replacements: Vec::new(), - post_graphs: Vec::new(), + post_dags: Vec::new(), rejections: Vec::new(), } } /// Assign collision-free, explicit identities to structurally equal nodes -/// across a set of exported query graphs. The full canonical subtree string +/// across a set of exported query DAGs. The full canonical sub-DAG string /// is the equality key; the compact integer is what JSON consumers receive. /// Consequently the viewer never needs to guess identity from labels, /// hashes, or a client-side node signature. -fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { - fn key_for(id: u32, graph: &DagGraph, memo: &mut HashMap) -> String { +fn assign_workload_node_ids(dags: &mut [&mut ExportDAG]) { + fn key_for(id: u32, dag: &ExportDAG, memo: &mut HashMap) -> String { if let Some(key) = memo.get(&id) { return key.clone(); } - let node = &graph.nodes[id as usize]; + let node = &dag.nodes[id as usize]; let child_keys: Vec<_> = node .children .iter() - .map(|child| key_for(*child, graph, memo)) + .map(|child| key_for(*child, dag, memo)) .collect(); let key = serde_json::to_string(&(node.kind, &node.detail, &node.schema, child_keys)) .expect("exported DAG node content is serializable"); @@ -1182,14 +1182,14 @@ fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { let mut ids = HashMap::::new(); let mut next_id = 0_u32; - for graph in graphs.iter_mut() { + for dag in dags.iter_mut() { let mut memo = HashMap::new(); - let keys: Vec<_> = graph + let keys: Vec<_> = dag .nodes .iter() - .map(|node| key_for(node.id, graph, &mut memo)) + .map(|node| key_for(node.id, dag, &mut memo)) .collect(); - for (node, key) in graph.nodes.iter_mut().zip(keys) { + for (node, key) in dag.nodes.iter_mut().zip(keys) { let id = *ids.entry(key).or_insert_with(|| { let id = next_id; next_id += 1; @@ -1206,7 +1206,7 @@ fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { /// needed) over every lowered query, rank each discovered `TargetSubDAGCandidates` via /// `CandidateLogicalASAPDAGs::global_selection`, and build both `--post-asap` outputs from the /// exact same set of winning candidates (see [`Winner`]), so the flat -/// `replacements` list and the merged `post_graph` can never disagree about +/// `replacements` list and the merged `post_dag` can never disagree about /// which candidate won for a given target. #[allow(dead_code)] fn run_post_asap_with_progress( @@ -1298,7 +1298,7 @@ fn run_post_asap_with_progress( ); } - // ---- Merged, whole-query `post_graph`, one per query --------------- + // ---- Merged, whole-query `post_dag`, one per query --------------- // // Built *before* the flat `replacements` pass below, not after: a // winner whose target only exists inside another winner's own @@ -1306,10 +1306,10 @@ fn run_post_asap_with_progress( // `AvgToSumOverCountStrategy`'s rewrite exposes, which // `default_strategies()` — #282 — now discovers and independently // sketch-ranks in the same search pass) can never appear in any query's - // *original*, pre-rewrite `graph` — there's nothing wrong with that - // winner, it's just nested. `post_graph` is where it's expected to + // *original*, pre-rewrite `dag` — there's nothing wrong with that + // winner, it's just nested. `post_dag` is where it's expected to // surface instead (`export_post_asap`'s recursive `find_winner` - // threading walks straight through a rewritten subtree and re-checks + // threading walks straight through a rewritten sub-DAG and re-checks // every node inside it too), so the flat-`replacements` pass below // checks there before deciding a miss is a real anomaly worth a // warning. @@ -1317,9 +1317,9 @@ fn run_post_asap_with_progress( eprintln!("[4/4] Post-ASAP DAG generation is running…"); } let post_started = Instant::now(); - let mut post_graph_cache = HashCache::new(); + let mut post_dag_cache = HashCache::new(); let mut find_winner = |expr: &QueryExpr| -> Option { - let i = lookup_winner(&by_hash, &winners, &mut post_graph_cache, expr)?; + let i = lookup_winner(&by_hash, &winners, &mut post_dag_cache, expr)?; let winner = &winners[i]; let (baseline_cost, selected_cost, benefit) = winner.costs.clone(); // Derived from `selected_cost`; see `target_replacement`'s identical @@ -1350,7 +1350,7 @@ fn run_post_asap_with_progress( } }) }; - let post_graphs: Vec<(String, DagGraph)> = lowered_queries + let post_dags: Vec<(String, ExportDAG)> = lowered_queries .iter() .map(|(name, _, qe)| { ( @@ -1362,21 +1362,21 @@ fn run_post_asap_with_progress( // ---- Flat per-target `replacements`, one list per query ----------- // - // Independently re-export every query's own graph for matching — a + // Independently re-export every query's own DAG for matching — a // fresh `export` per query, not reused from `main`'s own already-built - // `NamedGraph`s, so this function stays self-contained and callable on + // `NamedDAG`s, so this function stays self-contained and callable on // its own (see this file's `#[cfg(test)]` module). Deliberately anchored - // to the *original* `graph` only (never `post_graph`) — `target_pre_id` - // is documented as an id into `NamedGraph.graph.nodes`, so a nested + // to the *original* `dag` only (never `post_dag`) — `target_pre_id` + // is documented as an id into `NamedDAG.DAG.nodes`, so a nested // secondary target (see above) never gets a flat entry of its own here: // it's already visible, in place, inside its parent's own `after` - // subtree and inside `post_graph` as a whole. + // sub-DAG and inside `post_dag` as a whole. let mut lookup_cache = HashCache::new(); let mut replacements = Vec::new(); let mut rejections = Vec::new(); let mut matched = vec![false; winners.len()]; // Groups with accuracy-refused candidates (issue #172): matched to a - // query's graph nodes the same hash-then-structural-equality way. + // query's DAG nodes the same hash-then-structural-equality way. let rejected_groups: Vec<_> = space .target_subdag_candidates() .filter(|group| !group.rejected.is_empty()) @@ -1387,8 +1387,8 @@ fn run_post_asap_with_progress( rejected_by_hash.entry(hash).or_default().push(i); } for (name, _, qe) in lowered_queries { - let graph = dag_export::export(qe); - for node in &graph.nodes { + let dag = dag_export::export(qe); + for node in &dag.nodes { let Some(source_expr) = node.source_expr.as_ref() else { continue; // never true for a plain `export` — defensive only. }; @@ -1425,12 +1425,12 @@ fn run_post_asap_with_progress( // similarly `RollupStrategy`'s) can expose a brand-new `sum`/`count` // descendant that the *same* search pass then independently discovers // and ranks — a real winner, but one with no node anywhere in any - // query's original, pre-rewrite `graph` to attach a flat entry to - // (`target_pre_id` is documented as an id into `graph.nodes` + // query's original, pre-rewrite `dag` to attach a flat entry to + // (`target_pre_id` is documented as an id into `DAG.nodes` // specifically). This isn't a data loss: `export_post_asap` still - // splices that winner in, in place, inside `post_graph` — see this - // function's own construction of `post_graphs` above, which walks - // straight through a rewritten subtree and resolves every nested + // splices that winner in, in place, inside `post_dag` — see this + // function's own construction of `post_dags` above, which walks + // straight through a rewritten sub-DAG and resolves every nested // winner too, recursively. So an unmatched winner here is expected, // not necessarily a bug, whenever it's downstream of some other // winner's own `Replacement::Rewrite` — logged as an FYI rather than a @@ -1438,15 +1438,15 @@ fn run_post_asap_with_progress( // reimplementing `search`'s own private descendant-discovery walk // (`discover_new_descendant_targets` in `asap_aware_mapping::replacement`, // not exposed) a second time here just to double-check something - // `post_graph`'s own construction already handled correctly. + // `post_dag`'s own construction already handled correctly. for (winner, matched) in winners.iter().zip(&matched) { if !matched { let strategy = winner.candidate.strategy; eprintln!( "dag_export: post-asap replacement ({strategy}) has no node in any query's \ - original graph — expected for a winner exposed only inside another winner's \ + original DAG — expected for a winner exposed only inside another winner's \ own rewrite output (e.g. a sum/count descendant of an avg rewrite); still \ - present in that query's post_graph" + present in that query's post_dag" ); } } @@ -1459,7 +1459,7 @@ fn run_post_asap_with_progress( PostAsapResults { replacements, - post_graphs, + post_dags, rejections, } } @@ -1536,14 +1536,14 @@ async fn main() { let mut matched = vec![false; explanations.len()]; let mut queries = Vec::new(); for (name, source, qe) in &lowered_queries { - let mut graph = dag_export::export(qe); - annotate_with_explanations(&mut graph, &explanations, &mut matched); - queries.push(NamedGraph { + let mut dag = dag_export::export(qe); + annotate_with_explanations(&mut dag, &explanations, &mut matched); + queries.push(NamedDAG { name: name.clone(), source: Some(source.clone()), - graph, + dag, replacements: Vec::new(), - post_graph: None, + post_dag: None, workload_cost: None, rejections: Vec::new(), }); @@ -1606,9 +1606,9 @@ async fn main() { named.replacements.push(replacement); } } - for (query_name, post_graph) in results.post_graphs { + for (query_name, post_dag) in results.post_dags { if let Some(named) = queries.iter_mut().find(|q| q.name == query_name) { - named.post_graph = Some(post_graph); + named.post_dag = Some(post_dag); } } for (query_name, rejection) in results.rejections { @@ -1619,15 +1619,15 @@ async fn main() { } { - let mut pre_graphs: Vec<_> = queries.iter_mut().map(|query| &mut query.graph).collect(); - assign_workload_node_ids(&mut pre_graphs); + let mut pre_dags: Vec<_> = queries.iter_mut().map(|query| &mut query.dag).collect(); + assign_workload_node_ids(&mut pre_dags); } { - let mut post_graphs: Vec<_> = queries + let mut post_dags: Vec<_> = queries .iter_mut() - .filter_map(|query| query.post_graph.as_mut()) + .filter_map(|query| query.post_dag.as_mut()) .collect(); - assign_workload_node_ids(&mut post_graphs); + assign_workload_node_ids(&mut post_dags); } // Whole selected-workload cost/benefit (issue #286) — per query, and @@ -1641,10 +1641,10 @@ async fn main() { // needed. let mut workload_entries = Vec::new(); for query in &mut queries { - let Some(post_graph) = &query.post_graph else { + let Some(post_dag) = &query.post_dag else { continue; }; - let entries = decision_cost_entries(post_graph); + let entries = decision_cost_entries(post_dag); if entries.is_empty() { continue; } @@ -1679,7 +1679,7 @@ async fn main() { } }; - let workload = WorkloadGraph { + let workload = WorkloadDAG { queries, workload_cost, }; @@ -2698,7 +2698,7 @@ mod tests { fn missing_physical_evidence_keeps_the_export_raw_only() { let results = raw_only_post_asap_results(); assert!(results.replacements.is_empty()); - assert!(results.post_graphs.is_empty()); + assert!(results.post_dags.is_empty()); } fn argv(args: &[&str]) -> impl Iterator { @@ -2850,12 +2850,12 @@ mod tests { .any(|e| { e.kind == asap_aware_mapping::ExplanationKind::CommonSubexpressionReuse })); let mut matched = vec![false; explanations.len()]; - let mut graph_a = dag_export::export(&a); - let mut graph_b = dag_export::export(&b); - annotate_with_explanations(&mut graph_a, &explanations, &mut matched); - annotate_with_explanations(&mut graph_b, &explanations, &mut matched); - assert!(graph_a.nodes.iter().any(|n| !n.notes.is_empty())); - assert!(graph_b.nodes.iter().any(|n| !n.notes.is_empty())); + let mut dag_a = dag_export::export(&a); + let mut dag_b = dag_export::export(&b); + annotate_with_explanations(&mut dag_a, &explanations, &mut matched); + annotate_with_explanations(&mut dag_b, &explanations, &mut matched); + assert!(dag_a.nodes.iter().any(|n| !n.notes.is_empty())); + assert!(dag_b.nodes.iter().any(|n| !n.notes.is_empty())); } #[test] @@ -2873,13 +2873,13 @@ mod tests { let unrelated = lower_promql("sum(rate(other_metric[5m]))", AccuracyTarget::Epsilon(0.01)).unwrap(); - let mut graph = dag_export::export(&unrelated); - for node in &mut graph.nodes { + let mut dag = dag_export::export(&unrelated); + for node in &mut dag.nodes { node.hash = Some(explanation.node_hash); } let mut matched = vec![false; explanations.len()]; - annotate_with_explanations(&mut graph, &explanations, &mut matched); - assert!(graph.nodes.iter().all(|n| n.notes.is_empty())); + annotate_with_explanations(&mut dag, &explanations, &mut matched); + assert!(dag.nodes.iter().all(|n| n.notes.is_empty())); } /// The `--post-asap` code path, exercised directly (not through the CLI): @@ -2888,7 +2888,7 @@ mod tests { /// `after: TargetReplacementAfter::Summary(..)` (the bound sketch) and /// at least one with `after: TargetReplacementAfter::Rewrite(..)` (the /// `avg -> sum/count` rewrite) — see [`run_post_asap`]. Also checks that - /// a non-empty `post_graph` comes back for every query, since that's the + /// a non-empty `post_dag` comes back for every query, since that's the /// other `--post-asap` output `main` wires up. #[tokio::test] async fn post_asap_run_produces_both_summary_and_rewrite_replacements() { @@ -2957,18 +2957,18 @@ mod tests { "AvgToSumOverCountStrategy" | "SemanticEquivalentRewriteStrategy" ))); - assert_eq!(results.post_graphs.len(), 2, "one post_graph per query"); - for (name, graph) in &results.post_graphs { + assert_eq!(results.post_dags.len(), 2, "one post_dag per query"); + for (name, dag) in &results.post_dags { assert!( - !graph.nodes.is_empty(), - "post_graph for {name:?} must not be empty" + !dag.nodes.is_empty(), + "post_dag for {name:?} must not be empty" ); } let avg_post = &results - .post_graphs + .post_dags .iter() .find(|(name, _)| name == "avg") - .expect("avg post graph") + .expect("avg post DAG") .1; assert_eq!( avg_post @@ -2980,13 +2980,13 @@ mod tests { "AVG's SUM and COUNT branches must retain their shared input as one DAG node" ); let decisions: Vec<_> = results - .post_graphs + .post_dags .iter() - .flat_map(|(_, graph)| graph.nodes.iter().filter_map(|node| node.decision.as_ref())) + .flat_map(|(_, dag)| dag.nodes.iter().filter_map(|node| node.decision.as_ref())) .collect(); assert!( !decisions.is_empty(), - "post_graph nodes must carry explicit strategy metadata" + "post_dag nodes must carry explicit strategy metadata" ); for decision in decisions { assert!(!decision.strategy.is_empty()); @@ -3021,21 +3021,17 @@ mod tests { ), ]; let mut results = run_post_asap(&lowered); - let mut graph_refs: Vec<_> = results - .post_graphs - .iter_mut() - .map(|(_, graph)| graph) - .collect(); - assign_workload_node_ids(&mut graph_refs); + let mut dag_refs: Vec<_> = results.post_dags.iter_mut().map(|(_, dag)| dag).collect(); + assign_workload_node_ids(&mut dag_refs); let q3 = &results - .post_graphs + .post_dags .iter() .find(|(name, _)| name == "q3") .unwrap() .1; let q4 = &results - .post_graphs + .post_dags .iter() .find(|(name, _)| name == "q4") .unwrap() @@ -3068,26 +3064,26 @@ mod tests { ) .await .unwrap(); - let mut q1_graph = dag_export::export(&q1); - let mut q6_graph = dag_export::export(&q6); - assign_workload_node_ids(&mut [&mut q1_graph, &mut q6_graph]); - let q1_scan = q1_graph + let mut q1_dag = dag_export::export(&q1); + let mut q6_dag = dag_export::export(&q6); + assign_workload_node_ids(&mut [&mut q1_dag, &mut q6_dag]); + let q1_scan = q1_dag .nodes .iter() .find(|node| node.label == "Scan(metrics)") .unwrap(); - let q6_scan = q6_graph + let q6_scan = q6_dag .nodes .iter() .find(|node| node.label == "Scan(metrics)") .unwrap(); assert_eq!(q1_scan.workload_node_id, q6_scan.workload_node_id); - assert!(q6_graph.nodes.iter().any(|node| node.kind == "Join")); + assert!(q6_dag.nodes.iter().any(|node| node.kind == "Join")); } /// The `--default-cost` contract, end to end on the code path `main` /// takes for it (`DefaultCostModel` ranking, `export_model: None`): the - /// structure must be real — replacements found, a merged `post_graph` + /// structure must be real — replacements found, a merged `post_dag` /// per query — while every cost stays `Unavailable` with no value, so /// the viewer shows "Not estimated" and the structural ranking number /// never escapes as if it were a measured cost. @@ -3114,9 +3110,9 @@ mod tests { "the search itself must still run under --default-cost" ); assert!(results - .post_graphs + .post_dags .iter() - .all(|(_, graph)| !graph.nodes.is_empty())); + .all(|(_, dag)| !dag.nodes.is_empty())); for (_, replacement) in &results.replacements { for annotation in [ @@ -3139,8 +3135,8 @@ mod tests { // Absent a value, the per-query aggregation `main` runs degrades to // an unavailable summary rather than a number or an error. - for (_, graph) in &results.post_graphs { - for (_, baseline, selected) in decision_cost_entries(graph) { + for (_, dag) in &results.post_dags { + for (_, baseline, selected) in decision_cost_entries(dag) { assert_eq!(baseline.source, CostSource::Unavailable); assert_eq!(selected.source, CostSource::Unavailable); } @@ -3195,7 +3191,7 @@ mod tests { .map(|(name, r)| (name.clone(), r.strategy.clone())) .collect::>() ); - assert_eq!(results.post_graphs.len(), 1); - assert!(!results.post_graphs[0].1.nodes.is_empty()); + assert_eq!(results.post_dags.len(), 1); + assert!(!results.post_dags[0].1.nodes.is_empty()); } } diff --git a/crates/devtools/src/bin/sketch_coverage.rs b/crates/devtools/src/bin/sketch_coverage.rs index e9f0ccb10..58fa25348 100644 --- a/crates/devtools/src/bin/sketch_coverage.rs +++ b/crates/devtools/src/bin/sketch_coverage.rs @@ -9,7 +9,7 @@ // - a `SketchApproximation` candidate (a genuine sketch alternative was // found for at least one aggregate in the query — the KLL-vs-DDSketch // kind of degree of freedom), and/or -// - a `CommonSubexpressionReuse` candidate (the query shares a subtree, +// - a `CommonSubexpressionReuse` candidate (the query shares a sub-DAG, // inside itself or with another query in the same corpus, that a // build-once-and-share candidate was found for). // diff --git a/crates/devtools/src/bin/variant_coverage.rs b/crates/devtools/src/bin/variant_coverage.rs index a96042a2e..290d5584d 100644 --- a/crates/devtools/src/bin/variant_coverage.rs +++ b/crates/devtools/src/bin/variant_coverage.rs @@ -1,7 +1,7 @@ // cargo run -p asap-lower --bin variant_coverage // // Lowers every query in every corpus we have (PromQL + SQL), walks the -// resulting QueryExpr trees, and reports which enum variants show up — per +// resulting QueryExpr DAGs, and reports which enum variants show up — per // corpus, then rolled up globally. Used to find the minimal QueryExpr node set. use asap_devtools::lower_promql_with_data_ingestion_interval; diff --git a/crates/frontend-metricsql/tests/lowering.rs b/crates/frontend-metricsql/tests/lowering.rs index 133fc328b..3add8ee2a 100644 --- a/crates/frontend-metricsql/tests/lowering.rs +++ b/crates/frontend-metricsql/tests/lowering.rs @@ -13,13 +13,13 @@ fn lower(query: &str) -> QueryExpr { #[test] fn selector_range_aggregate_and_call_share_the_canonical_shape() { let query = r#"sum by (job) (rate(http_requests_total{status=~"5.."}[5m]))"#; - let tree = lower(query); + let dag = lower(query); let QueryExpr::Aggregate { reduction, measures, child, .. - } = tree + } = dag else { panic!("expected outer aggregate"); }; @@ -43,13 +43,13 @@ fn selector_range_aggregate_and_call_share_the_canonical_shape() { #[test] fn default_rollup_with_explicit_range_is_last_over_time() { - let tree = lower("default_rollup(cpu_usage[5m])"); + let dag = lower("default_rollup(cpu_usage[5m])"); let QueryExpr::Aggregate { reduction, measures, child, .. - } = tree + } = dag else { panic!("expected aggregate"); }; diff --git a/crates/frontend-promql/src/error.rs b/crates/frontend-promql/src/error.rs index 1885a5f42..6f996b11d 100644 --- a/crates/frontend-promql/src/error.rs +++ b/crates/frontend-promql/src/error.rs @@ -1,12 +1,12 @@ use std::fmt; -use asap_types::pre_asap::ResolveTreeError; +use asap_types::pre_asap::ResolveDAGError; use asap_types::workload::WorkloadError; /// Errors from lowering a PromQL query (parse → the canonical, unresolved -/// tree, built directly → +/// DAG, built directly → /// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved tree, issue #179). +/// resolved DAG, issue #179). /// /// Carries no DataFusion type — the PromQL front end never depends on the SQL /// stack. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` @@ -31,9 +31,9 @@ pub enum PromqlError { InvalidParameter(String), /// The workload's query language is not PromQL. WrongLanguage(String), - /// Resolving the canonical unresolved tree failed (name resolution + /// Resolving the canonical unresolved DAG failed (name resolution /// against the bound schema). - Convert(ResolveTreeError), + Convert(ResolveDAGError), } impl fmt::Display for PromqlError { @@ -54,8 +54,8 @@ impl fmt::Display for PromqlError { impl std::error::Error for PromqlError {} -impl From for PromqlError { - fn from(e: ResolveTreeError) -> Self { +impl From for PromqlError { + fn from(e: ResolveDAGError) -> Self { Self::Convert(e) } } diff --git a/crates/frontend-promql/src/lib.rs b/crates/frontend-promql/src/lib.rs index 60254d3d9..31b169359 100644 --- a/crates/frontend-promql/src/lib.rs +++ b/crates/frontend-promql/src/lib.rs @@ -4,7 +4,7 @@ //! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the //! canonical `QueryExpr`, generic over an unresolved //! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational tree; `resolve_root` runs the +//! separate per-language relational DAG; `resolve_root` runs the //! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. //! Depends on the PromQL parser only — never on the SQL / DataFusion stack. diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index e1061683f..83251053e 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -6,7 +6,7 @@ //! - **Lowering** builds *directly in canonical shape* here (issue #179): the //! walk interprets PromQL semantics (range vectors, aggregate operators, //! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved -//! `ColumnRef`s — the same tree shape +//! `ColumnRef`s — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to //! canonical, positional `QueryExpr`. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter @@ -17,7 +17,7 @@ //! the schema-*dependent* work: binding every `ColumnRef` to its //! positional `ColumnId`. //! -//! # PromQL → canonical unresolved-tree mapping (summary) +//! # PromQL → canonical unresolved-DAG mapping (summary) //! //! | PromQL | Canonical shape | //! |---|---| @@ -79,7 +79,7 @@ use crate::error::PromqlError as LoweringError; type Result = std::result::Result; -/// Parses and lowers (→ the canonical, unresolved tree) a PromQL query string. +/// Parses and lowers (→ the canonical, unresolved DAG) a PromQL query string. pub(crate) struct PromqlLowerer; #[derive(Debug, Clone)] @@ -357,8 +357,8 @@ fn walk(expr: &Expr) -> Result { /// `extract_matrix` can't accept it. Lower the sub-query recursively and reduce /// it per series (issue #27). fn walk_call(call: &Call) -> Result { - if let Some(tree) = range_fn_over_subquery(call)? { - return Ok(tree); + if let Some(dag) = range_fn_over_subquery(call)? { + return Ok(dag); } build(lower_inner_call(call)?, vec![], Outer::None) } @@ -491,7 +491,7 @@ fn walk_aggregate(agg: &AggregateExpr) -> Result { // exclusion form when the modifier was `without(...)`. let built = match lower_inner(&agg.expr) { Ok(inner) => build(inner, keys, outer)?, - Err(_) => build_over_subtree(outer, keys, walk(&agg.expr)?)?, + Err(_) => build_over_sub_dag(outer, keys, walk(&agg.expr)?)?, }; Ok(mark_without(built, without)) } @@ -555,13 +555,13 @@ fn outer_kind(agg: &AggregateExpr) -> Result { }) } -/// Wrap an already-lowered Unresolved subtree in the outer aggregation. This is the +/// Wrap an already-lowered Unresolved sub-DAG in the outer aggregation. This is the /// general-nesting counterpart to [`build`]: where `build` assembles the /// two-level shape from a flat [`Inner`], this composes the outer operator over /// an arbitrary child (`max(sum by (job) (…))`, `sum(a + b)`, …). /// /// A heavy-hitter `TopK` is only recognised on the flat `count_over_time` shape -/// (handled in `build`); over a general subtree, `topk`/`bottomk` is a generic +/// (handled in `build`); over a general sub-DAG, `topk`/`bottomk` is a generic /// order-by-value + limit — the same `Sort{partition_by} → Limit` pair `build` /// emits for any non-heavy-hitter ranking. /// Flip the outer `Aggregate` produced for a `without(...)` grouping into the @@ -576,7 +576,7 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// build this node) decides `PerEntity` vs `Reduce(by)` *without* knowing /// about `without` yet — it only ever sees `by`-mode keys, since `without`'s /// excluded-labels list is applied here, after the fact, exactly like the -/// pre-#179 legacy `relational::QueryExpr` tree's own `mark_without` did (its +/// pre-#179 legacy `relational::QueryExpr` DAG's own `mark_without` did (its /// converter read `without` only after this front-end step had already set /// it). Whether /// `reduction_for` picked `PerEntity` (only possible when `keys` was empty) @@ -584,11 +584,11 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// `Reduce(without(keys))`: a `without` grouping is never label-preserving — /// per-entity requires `!by.is_without()` — so this both re-tags an existing /// `Reduce` and upgrades a wrongly-early `PerEntity` guess, uniformly. -fn mark_without(tree: Unresolved, without: bool) -> Unresolved { +fn mark_without(dag: Unresolved, without: bool) -> Unresolved { if !without { - return tree; + return dag; } - match tree { + match dag { Unresolved::Aggregate { reduction, measures, @@ -614,7 +614,7 @@ fn mark_without(tree: Unresolved, without: bool) -> Unresolved { } } -fn build_over_subtree(outer: Outer, keys: Vec, child: Unresolved) -> Result { +fn build_over_sub_dag(outer: Outer, keys: Vec, child: Unresolved) -> Result { Ok(match outer { // `walk_aggregate` always passes a real aggregator; `None` can't occur. Outer::None => child, @@ -750,7 +750,7 @@ fn classic_histogram_quantile(q: f64, output_name: &str, child: Unresolved) -> U /// (`HistogramQuantile`), native histograms / raw samples take the sketch-able /// `Quantile` (issues #43 / #79) — so the two functions cannot diverge. /// -/// The vector argument is lowered once per branch, duplicating the subtree — +/// The vector argument is lowered once per branch, duplicating the sub-DAG — /// a future workload-level reuse pass could hoist it back into a single /// producer. /// @@ -906,7 +906,7 @@ fn is_presence_fn(name: &str) -> bool { /// `absent(v)` / `absent_over_time(m[w])` / `present_over_time(m[w])` — lowered /// to an `Aggregate{[Absent/…]}` over the (instant or range) argument. The /// empty-result → synthesized-1-sample logic is a post-ASAP/runtime concern; -/// the canonical tree only marks the operation (issue #47). +/// the canonical DAG only marks the operation (issue #47). fn walk_presence(call: &Call) -> Result { let func = match call.func.name { "absent" => AggIntent::Absent, @@ -1470,7 +1470,7 @@ fn lower_inner_call(call: &Call) -> Result { } } -/// Assemble the Layer-2 tree from a lowered inner vector, the resolved group +/// Assemble the Layer-2 DAG from a lowered inner vector, the resolved group /// keys, and the enclosing aggregator shape. fn build(inner: Inner, keys: Vec, outer: Outer) -> Result { match outer { @@ -1558,7 +1558,7 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result }; let additive_ranking = measure.is_supported(descending); if additive_ranking { - // Preserve the ranked aggregate intent in the canonical tree so the + // Preserve the ranked aggregate intent in the canonical DAG so the // intent algebra is explicit about what is being computed. // Post-ASAP binding may fuse the Count and TopK into a // single-pass heavy-hitter sketch (SpaceSaving / @@ -1622,7 +1622,7 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result /// Decide `PerEntity` vs `Reduce(by)` for a canonical `Aggregate`, entirely /// from local PromQL semantics: the keys and whether this operation preserves -/// each input series. It never infers entity reduction from the child tree's +/// each input series. It never infers entity reduction from the child DAG's /// temporal shape. `without()` is applied /// separately, post-hoc, by `mark_without` — see its doc for why that's still /// correct here. @@ -1676,7 +1676,7 @@ fn windowed_aggregate( } } -/// `Aggregate{reduction, [intent]}` directly over an existing Unresolved subtree — the +/// `Aggregate{reduction, [intent]}` directly over an existing Unresolved sub-DAG — the /// OUTER level of a two-level aggregation such as `sum(rate(…))` or the /// `Aggregate{[Quantile]}` that wraps a `histogram_quantile` argument. fn outer_aggregate( diff --git a/crates/frontend-promql/tests/histogram_metadata.rs b/crates/frontend-promql/tests/histogram_metadata.rs index 60ff5a60e..55f35ddeb 100644 --- a/crates/frontend-promql/tests/histogram_metadata.rs +++ b/crates/frontend-promql/tests/histogram_metadata.rs @@ -11,7 +11,7 @@ use asap_types::pre_asap::{AggIntent, QueryExpr}; use asap_types::types::AccuracyTarget; use support::{lower_promql, lower_promql_with_histograms}; -/// The histogram/quantile intent kind in the lowered tree: `"HQ"` for the +/// The histogram/quantile intent kind in the lowered DAG: `"HQ"` for the /// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. fn quantile_kind(qe: &QueryExpr) -> &'static str { fn walk(e: &QueryExpr) -> Option<&'static str> { diff --git a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs index 12afb94da..1b37cdee4 100644 --- a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs +++ b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs @@ -5,7 +5,7 @@ //! host/hardware, node-exporter, databases, message brokers, Kubernetes, and //! more (`tests/data/awesome_prometheus_alerts.txt`). //! -//! We *lower* (parse → the canonical tree), we do not execute. Two guarantees: +//! We *lower* (parse → the canonical DAG), we do not execute. Two guarantees: //! 1. **Totality** — every real-world query returns `Ok` or a clean //! `LoweringError` and never panics. //! 2. **Parseability** — none of them fail at the *parse* stage; the private @@ -49,7 +49,7 @@ fn ok(q: &str) -> QueryExpr { .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } -/// Every `AggIntent` in the tree. +/// Every `AggIntent` in the DAG. fn intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); fn go(e: &QueryExpr, out: &mut Vec) { diff --git a/crates/frontend-promql/tests/observability/o11y_bench_promql.rs b/crates/frontend-promql/tests/observability/o11y_bench_promql.rs index e178d8981..cd371be69 100644 --- a/crates/frontend-promql/tests/observability/o11y_bench_promql.rs +++ b/crates/frontend-promql/tests/observability/o11y_bench_promql.rs @@ -12,7 +12,7 @@ //! verbatim into a differently-licensed test suite is fine as-is is an open //! question — tracked in issue #135, not resolved by this file's existence. //! -//! We *lower* (parse → the canonical tree), we do not execute. Totality: every query +//! We *lower* (parse → the canonical DAG), we do not execute. Totality: every query //! returns `Ok` or a clean `LoweringError` and never panics. Given the corpus //! is small and hand-picked from realistic incident-response queries, we also //! assert full lowering coverage — a regression here means a real pattern diff --git a/crates/frontend-promql/tests/observability/promql_corpus.rs b/crates/frontend-promql/tests/observability/promql_corpus.rs index f265edb45..1bc7e2166 100644 --- a/crates/frontend-promql/tests/observability/promql_corpus.rs +++ b/crates/frontend-promql/tests/observability/promql_corpus.rs @@ -27,7 +27,7 @@ use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::types::AccuracyTarget; use support::lower_promql; -/// This crate has no "bind me one tree" public API any more — +/// This crate has no "bind me one DAG" public API any more — /// `SketchAlgorithmStrategy::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so [`bind_tally`] @@ -104,10 +104,10 @@ struct BindTally { fn bind_tally(corpus: &str, accuracy: AccuracyTarget) -> BindTally { let mut t = BindTally::default(); for q in queries(corpus) { - let Ok(tree) = lower_promql(q, accuracy.clone()) else { + let Ok(dag) = lower_promql(q, accuracy.clone()) else { continue; }; - match bind(&tree) { + match bind(&dag) { Ok(bound) if matches!(bound.expr, SummaryExpr::KeepPreAsap(_)) => t.unchanged += 1, Ok(_) => t.transformed += 1, Err(_) => t.errored += 1, diff --git a/crates/frontend-promql/tests/promql_binding_regressions.rs b/crates/frontend-promql/tests/promql_binding_regressions.rs index b5edf6049..26d1be06d 100644 --- a/crates/frontend-promql/tests/promql_binding_regressions.rs +++ b/crates/frontend-promql/tests/promql_binding_regressions.rs @@ -34,8 +34,8 @@ fn irate_and_rate_have_distinct_canonical_intents() { #[test] fn count_is_row_count_not_distinct_sample_value_count() { use asap_types::pre_asap::{AggIntent, QueryExpr}; - let tree = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { measures, .. } = tree else { + let dag = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); + let QueryExpr::Aggregate { measures, .. } = dag else { panic!("expected aggregate") }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); diff --git a/crates/frontend-promql/tests/promql_conformance.rs b/crates/frontend-promql/tests/promql_conformance.rs index 69fbb2230..a7b405f0d 100644 --- a/crates/frontend-promql/tests/promql_conformance.rs +++ b/crates/frontend-promql/tests/promql_conformance.rs @@ -1,8 +1,8 @@ -//! PromQL **semantic conformance** for the parse-to-canonical-tree lowering. +//! PromQL **semantic conformance** for the parse-to-canonical-DAG lowering. //! //! We *lower* PromQL to the intent algebra; we do not *execute* it. So "same //! semantic job as Prometheus" here means: for each canonical query, does the -//! canonical tree encode the **documented PromQL meaning** — and where we knowingly +//! canonical DAG encode the **documented PromQL meaning** — and where we knowingly //! diverge (reject, approximate, or drop a modifier), is that pinned by a test //! so it stays visible? //! @@ -55,11 +55,11 @@ fn ok(q: &str) -> QueryExpr { fn rejected(q: &str) -> LoweringError { match lower_promql(q, AccuracyTarget::Exact) { Err(e) => e, - Ok(tree) => panic!("expected {q:?} to be rejected, but it lowered to: {tree:?}"), + Ok(dag) => panic!("expected {q:?} to be rejected, but it lowered to: {dag:?}"), } } -/// Every `AggIntent` anywhere in the tree, root-to-leaf. +/// Every `AggIntent` anywhere in the DAG, root-to-leaf. fn intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); collect(e, &mut out); @@ -148,7 +148,7 @@ fn has bool>(e: &QueryExpr, pred: F) -> bool { intents(e).iter().any(pred) } -/// Whether the tree contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary +/// Whether the DAG contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary /// negation lowers to (issue #36). fn negates_via_scalar(e: &QueryExpr) -> bool { let is_neg_one = |q: &QueryExpr| { @@ -239,7 +239,7 @@ fn name_regex_matcher_is_rejected__GAP() { #[test] fn range_vector_selector_is_time_range() { // SEMANTICS: `[5m]` turns an instant vector into a range vector, - // represented in the canonical tree as a dedicated `TimeRange` node. + // represented in the canonical DAG as a dedicated `TimeRange` node. let qe = ok("node_cpu_seconds_total[5m]"); let QueryExpr::TimeRange { range, .. } = &qe else { panic!("expected TimeRange for a range-vector selector, got {qe:?}"); @@ -549,7 +549,7 @@ fn histogram_quantile_over_rate() { fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { // SEMANTICS: the standard pattern — bucket rates summed by `le`, then the // quantile. The `sum by (le)` aggregation must survive into the - // canonical tree. + // canonical DAG. let qe = ok( "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", ); @@ -635,7 +635,7 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { "sum(-node_cpu_seconds_total)", ] { let qe = ok(q); - // A `Mul`-by-`-1` against a `PromqlScalarBridge(-1)` appears somewhere in every tree. + // A `Mul`-by-`-1` against a `PromqlScalarBridge(-1)` appears somewhere in every DAG. assert!( negates_via_scalar(&qe), "no `* -1` negation found in {q}: {qe:?}" @@ -859,7 +859,7 @@ fn outer_aggregate_over_nested_aggregate_nests() { // `max(sum by (job) (rate(m[5m])))` — an outer cross-series reduction over a // nested per-group reduction over a per-series rate: three stacked levels the // flat two-level template rejected. Each level survives into the - // canonical tree (issue #27). + // canonical DAG (issue #27). let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); let QueryExpr::Aggregate { measures, child, .. @@ -955,7 +955,7 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // Issue #52: an outer aggregate's group key that appears in *neither* side of // a binary op — the metric-name label `__name__`, or a plain `job` — must // still resolve. Each `or` side is bound independently against its own - // sub-tree, so the key is seeded as an inherited column on both sides. + // sub-DAG, so the key is seeded as an inherited column on both sides. let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); let QueryExpr::Aggregate { reduction, child, .. @@ -1833,7 +1833,7 @@ fn no_arg_calendar_function_reads_the_eval_time() { #[test] fn timestamp_composes_under_an_outer_aggregation() { // `sum by (job) (timestamp(up))` — the per-series `timestamp` transform sits - // below an ordinary grouped sum. Both intents must appear in the tree. + // below an ordinary grouped sum. Both intents must appear in the DAG. let qe = ok("sum by (job) (timestamp(up))"); assert!(has(&qe, |i| *i == AggIntent::TimeFn(TimeFunc::Timestamp))); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); diff --git a/crates/frontend-promql/tests/promql_equivalence.rs b/crates/frontend-promql/tests/promql_equivalence.rs index d1177cc59..9d0cab2ad 100644 --- a/crates/frontend-promql/tests/promql_equivalence.rs +++ b/crates/frontend-promql/tests/promql_equivalence.rs @@ -1,10 +1,10 @@ -//! PromQL **semantic-equivalence proving** for the parse-to-canonical-tree lowering. +//! PromQL **semantic-equivalence proving** for the parse-to-canonical-DAG lowering. //! //! The lowering is a *normalizer*: it should map a whole class of -//! semantically-equivalent PromQL strings to **one** canonical tree, and +//! semantically-equivalent PromQL strings to **one** canonical DAG, and //! must keep semantically-*distinct* queries distinct. This suite proves: //! -//! 1. Equivalence classes collapse to an identical canonical tree (`assert_equiv`). +//! 1. Equivalence classes collapse to an identical canonical DAG (`assert_equiv`). //! 2. Distinct meanings stay distinct (`assert_distinct`). //! 3. The lowering never *wrongly* equates distinct semantics — the cases it //! cannot faithfully distinguish are **rejected**, not silently merged. @@ -26,25 +26,25 @@ fn lo(q: &str) -> QueryExpr { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("{q:?} should lower: {e}")) } -/// Every member of an equivalence class must lower to the *same* canonical tree. +/// Every member of an equivalence class must lower to the *same* canonical DAG. fn assert_equiv(class: &[&str]) { let first = lo(class[0]); for q in &class[1..] { assert_eq!( lo(q), first, - "expected {q:?} ≡ {:?}, but they lowered to different trees", + "expected {q:?} ≡ {:?}, but they lowered to different DAGs", class[0] ); } } -/// Two semantically-distinct queries must lower to *different* canonical trees. +/// Two semantically-distinct queries must lower to *different* canonical DAGs. fn assert_distinct(a: &str, b: &str) { assert_ne!( lo(a), lo(b), - "{a:?} and {b:?} must not collapse to the same tree" + "{a:?} and {b:?} must not collapse to the same DAG" ); } @@ -79,7 +79,7 @@ fn whitespace_is_irrelevant() { #[test] fn label_matcher_order_is_equivalent() { - // A matcher set is unordered: same series, so same canonical tree (FIX: + // A matcher set is unordered: same series, so same canonical DAG (FIX: // predicates are now canonicalised by (name, value) at lowering time). assert_equiv(&[r#"up{job="a",env="prod"}"#, r#"up{env="prod",job="a"}"#]); } @@ -153,7 +153,7 @@ fn changes_and_resets_are_not_count_over_time() { // PromQL: count_over_time = #samples, changes = #value-changes, // resets = #counter-resets. They previously all collapsed to `Count`; now // each lowers to its own intent (issue #44), so all three are pairwise - // distinct canonical trees rather than being rejected or merged. + // distinct canonical DAGs rather than being rejected or merged. assert_distinct("changes(m[5m])", "count_over_time(m[5m])"); assert_distinct("resets(m[5m])", "count_over_time(m[5m])"); assert_distinct("changes(m[5m])", "resets(m[5m])"); @@ -163,7 +163,7 @@ fn changes_and_resets_are_not_count_over_time() { fn group_is_not_sum() { // PromQL `group` returns a constant 1 per group; it previously collapsed // onto `sum` (sum of values). It now lowers to its own `Group` intent - // (issue #49) — a distinct canonical tree from `sum`, not merged. + // (issue #49) — a distinct canonical DAG from `sum`, not merged. assert_distinct("group(up)", "sum(up)"); assert_distinct("group by (job) (up)", "sum by (job) (up)"); } diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 9d3b80de9..619d066c2 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -1,4 +1,4 @@ -//! End-to-end tests for PromQL → unresolved → canonical tree lowering. +//! End-to-end tests for PromQL → unresolved → canonical DAG lowering. use std::time::Duration; @@ -56,14 +56,14 @@ fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { "sum by(job)(distinct_over_time(cpu_usage[5m]))", ] { for accuracy in [AccuracyTarget::Exact, AccuracyTarget::Epsilon(0.02)] { - let tree = lower_promql(query, accuracy.clone()).unwrap(); + let dag = lower_promql(query, accuracy.clone()).unwrap(); let mut intents = Vec::new(); - collect_intents(&tree, &mut intents); + collect_intents(&dag, &mut intents); assert!( intents.iter().any(|intent| matches!( intent, AggIntent::Cardinality { accuracy: actual, .. } if actual == &accuracy )), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert!(!intents .iter() @@ -264,7 +264,7 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { // The canonical Prometheus histogram pattern. Previously returned // UnsupportedFeature because `extract_matrix` couldn't see through the // `sum by (le)` aggregate; now the `le` grouping survives into the - // canonical tree. + // canonical DAG. let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); let QueryExpr::Aggregate { measures, child, .. @@ -486,19 +486,19 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { Reduction::by(vec![2]), ), ] { - let tree = lower(query); + let dag = lower(query); let QueryExpr::Aggregate { measures, reduction: actual, child, .. - } = &tree + } = &dag else { - panic!("expected outer Count: {tree:?}"); + panic!("expected outer Count: {dag:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::Count { .. }]), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert_eq!(actual, &reduction, "{query}"); let QueryExpr::Aggregate { @@ -508,11 +508,11 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { .. } = child.as_ref() else { - panic!("expected inner per-series Cardinality: {tree:?}"); + panic!("expected inner per-series Cardinality: {dag:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::Cardinality { .. }]), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert_eq!(reduction, &Reduction::PerEntity, "{query}"); assert!( @@ -534,17 +534,17 @@ fn count_never_lowers_to_distinct_sample_values() { "count(count_over_time(up[5m]))", "count_over_time(up[5m])", ] { - let tree = lower(query); - let intents = all_intents(&tree); + let dag = lower(query); + let intents = all_intents(&dag); assert!( intents.iter().any(|i| matches!(i, AggIntent::Count { .. })), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert!( !intents .iter() .any(|i| matches!(i, AggIntent::Cardinality { .. })), - "{query}: {tree:?}" + "{query}: {dag:?}" ); } } @@ -829,7 +829,7 @@ fn binary_op_binds_each_branch_against_its_own_schema() { ); } -/// Collect every `AggIntent` in the tree, root-to-leaf. +/// Collect every `AggIntent` in the DAG, root-to-leaf. fn all_intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); collect_intents(e, &mut out); @@ -856,7 +856,7 @@ fn collect_intents(e: &QueryExpr, out: &mut Vec) { } } -/// True if any `AggIntent` anywhere in the tree satisfies `pred`. +/// True if any `AggIntent` anywhere in the DAG satisfies `pred`. fn has_intent bool>(e: &QueryExpr, pred: F) -> bool { all_intents(e).iter().any(pred) } @@ -1328,7 +1328,7 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { } } -// A subquery's `offset`/`@` shift the whole subquery, so the tree keeps them. +// A subquery's `offset`/`@` shift the whole subquery, so the DAG keeps them. #[test] fn subquery_time_shift_is_retained() { let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index d34768e8f..1c7083464 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -8,7 +8,7 @@ use asap_aware_mapping::replacement::{default_strategies, search_workload_with_t use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; mod support; use asap_types::post_asap::{ - compile_post_asap_dag, cse::share_common_summary_subtrees, AccuracyError, BoundExpr, + compile_post_asap_dag, cse::share_common_summary_sub_dags, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchQuery, SummaryExpr, SummaryFamilyType, SummaryInputExpr, SummaryNode, }; @@ -76,7 +76,7 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { .enumerate() .map(|(id, (query, accuracy))| (id, candidate(query, accuracy))) .collect(); - let roots = share_common_summary_subtrees(roots); + let roots = share_common_summary_sub_dags(roots); let mut first_state = None; for (index, root) in &roots { let SummaryExpr::SummaryEstimate { diff --git a/crates/frontend-sql/src/error.rs b/crates/frontend-sql/src/error.rs index 5342f5a3f..04117352d 100644 --- a/crates/frontend-sql/src/error.rs +++ b/crates/frontend-sql/src/error.rs @@ -1,11 +1,11 @@ use std::fmt; -use asap_types::pre_asap::ResolveTreeError; +use asap_types::pre_asap::ResolveDAGError; /// Errors from lowering a SQL query (parse + plan via DataFusion → the -/// canonical, unresolved tree, built directly → +/// canonical, unresolved DAG, built directly → /// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved tree, issue #179). +/// resolved DAG, issue #179). /// /// Carries no PromQL type — the SQL front end never depends on the PromQL /// parser. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` @@ -28,9 +28,9 @@ pub enum SqlError { UnsupportedFeature(String), /// The workload's query language is not SQL. WrongLanguage(String), - /// Resolving the canonical unresolved tree failed (name resolution + /// Resolving the canonical unresolved DAG failed (name resolution /// against the bound schema). - Convert(ResolveTreeError), + Convert(ResolveDAGError), } impl fmt::Display for SqlError { @@ -50,8 +50,8 @@ impl fmt::Display for SqlError { impl std::error::Error for SqlError {} -impl From for SqlError { - fn from(e: ResolveTreeError) -> Self { +impl From for SqlError { + fn from(e: ResolveDAGError) -> Self { Self::Convert(e) } } diff --git a/crates/frontend-sql/src/lib.rs b/crates/frontend-sql/src/lib.rs index 58474e2eb..4ec61e69d 100644 --- a/crates/frontend-sql/src/lib.rs +++ b/crates/frontend-sql/src/lib.rs @@ -4,7 +4,7 @@ //! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the //! canonical `QueryExpr`, generic over an unresolved //! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational tree; `resolve_root` runs the +//! separate per-language relational DAG; `resolve_root` runs the //! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. //! Depends on DataFusion only — never on the PromQL parser. @@ -24,7 +24,7 @@ pub use sql::{SqlCatalog, SqlLowerer}; /// /// The `catalog` supplies table schemas (used both to plan the SQL with /// DataFusion and to carry positional column identity into the resolved -/// tree). `accuracy` is threaded onto every approximate intent as it's built. +/// DAG). `accuracy` is threaded onto every approximate intent as it's built. pub async fn lower_sql( query: &str, catalog: &SqlCatalog, diff --git a/crates/frontend-sql/src/sql/collection_planning.rs b/crates/frontend-sql/src/sql/collection_planning.rs index 92fda701d..eba207751 100644 --- a/crates/frontend-sql/src/sql/collection_planning.rs +++ b/crates/frontend-sql/src/sql/collection_planning.rs @@ -164,7 +164,7 @@ impl ScalarUDFImpl for CollectionPlanningFunction { } } fn invoke_batch(&self, _args: &[ColumnarValue], _number_rows: usize) -> Result { - Err(DataFusionError::NotImplemented("collection planning adapter cannot execute; use a capable query engine or external exact subtree".into())) + Err(DataFusionError::NotImplemented("collection planning adapter cannot execute; use a capable query engine or external exact sub-DAG".into())) } } diff --git a/crates/frontend-sql/src/sql/expr.rs b/crates/frontend-sql/src/sql/expr.rs index 27ca85beb..06b682062 100644 --- a/crates/frontend-sql/src/sql/expr.rs +++ b/crates/frontend-sql/src/sql/expr.rs @@ -24,7 +24,7 @@ pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { } } -/// Translate a DataFusion `Expr` to the canonical, unresolved tree. +/// Translate a DataFusion `Expr` to the canonical, unresolved DAG. /// Returns `UnsupportedFeature` for anything not needed in v1. pub(super) fn df_expr_to_unresolved(expr: &Expr) -> Result { match expr { diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index e147e0bf3..52c065e9a 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -4,7 +4,7 @@ //! //! Parses SQL via DataFusion (over the catalog's registered tables), then //! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with -//! unresolved `ColumnRef`s directly (issue #179) — the same tree shape +//! unresolved `ColumnRef`s directly (issue #179) — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, //! positional `QueryExpr`. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit @@ -114,7 +114,7 @@ fn current_accuracy() -> AccuracyTarget { /// Lowers SQL strings to the canonical [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) /// over a table [`SqlCatalog`]. Call /// [`resolve_root`](asap_types::pre_asap::resolve_root) on the result for -/// the canonical, resolved tree. +/// the canonical, resolved DAG. pub struct SqlLowerer<'a> { catalog: &'a SqlCatalog, dialect: SqlDialect, @@ -2116,7 +2116,7 @@ fn expand_grouping_set(gs: &logical_expr::GroupingSet) -> Vec> { struct DerivedCols { cols: Vec>, /// Whether any column is genuinely derived. Without one the aggregate keeps - /// its original child, so trees that lower today keep their exact shape. + /// its original child, so DAGs that lower today keep their exact shape. any: bool, /// First same-name-different-value collision, reported only if the /// projection is actually inserted (see [`Self::wrap`]). @@ -2215,7 +2215,7 @@ impl DerivedCols { /// Wrap `input` in the materializing `Project`, or return it untouched when /// nothing needed deriving — so a query that lowers today keeps its exact - /// tree, and a name collision that the projection would have flattened only + /// DAG, and a name collision that the projection would have flattened only /// matters once the projection exists. fn wrap(self, input: Unresolved) -> Result { if !self.any { diff --git a/crates/frontend-sql/src/sql/types.rs b/crates/frontend-sql/src/sql/types.rs index 5382c4171..22611e6d0 100644 --- a/crates/frontend-sql/src/sql/types.rs +++ b/crates/frontend-sql/src/sql/types.rs @@ -1,6 +1,6 @@ //! Type bridges between DataFusion's Arrow types and the canonical `DataType`, plus //! the SQL table catalog used to register tables with DataFusion and to carry -//! resolved leaf schemas into the canonical, unresolved tree. +//! resolved leaf schemas into the canonical, unresolved DAG. use std::collections::HashMap; diff --git a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs index 273597e7d..189ff0ff7 100644 --- a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs +++ b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs @@ -65,9 +65,9 @@ fn queries() -> Vec { .collect() } -// ── tree helpers ────────────────────────────────────────────────────────────── +// ── DAG helpers ────────────────────────────────────────────────────────────── -/// Every `AggIntent` in the tree, root-to-leaf. +/// Every `AggIntent` in the DAG, root-to-leaf. fn intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); fn go(e: &QueryExpr, out: &mut Vec) { @@ -114,7 +114,7 @@ fn intents(e: &QueryExpr) -> Vec { | QueryExpr::CurrentTimestamp => {} // Scalar expression variants (issue #205): `AggIntent` only ever // lives in `Aggregate.measures`, never nested inside a scalar - // expression tree, so there's nothing to recurse into here. + // expression DAG, so there's nothing to recurse into here. QueryExpr::Column(_) | QueryExpr::Literal(_) | QueryExpr::Compare { .. } diff --git a/crates/frontend-sql/tests/maintained_population.rs b/crates/frontend-sql/tests/maintained_population.rs index 0019b3b83..5d73d7e90 100644 --- a/crates/frontend-sql/tests/maintained_population.rs +++ b/crates/frontend-sql/tests/maintained_population.rs @@ -5,7 +5,7 @@ use asap_types::{ post_asap::{ compile_post_asap_dag, maintained_population::{MaintainedPopulation, PopulationInput}, - share_common_summary_subtrees, SummaryExpr, ValueOperation, + share_common_summary_sub_dags, SummaryExpr, ValueOperation, }, pre_asap::{Column, DataType, QueryExpr, Schema}, types::AccuracyTarget, @@ -59,7 +59,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { aggregate("SELECT approx_percentile_cont(latency, 0.99) FROM samples").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() @@ -120,7 +120,7 @@ async fn sql_scalar_readouts_share_membership() { roots.push(aggregate(&format!("SELECT {function} FROM samples")).await); } let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() @@ -167,7 +167,7 @@ async fn sql_topk_limits_share_maximum_k() { aggregate("SELECT * FROM samples ORDER BY latency DESC LIMIT 5").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() diff --git a/crates/frontend-sql/tests/netflow/netflow.rs b/crates/frontend-sql/tests/netflow/netflow.rs index 09d742b8b..dbe315381 100644 --- a/crates/frontend-sql/tests/netflow/netflow.rs +++ b/crates/frontend-sql/tests/netflow/netflow.rs @@ -303,7 +303,7 @@ fn visit(qe: &QueryExpr, f: &mut impl FnMut(&QueryExpr)) { | QueryExpr::EvalTimestamp | QueryExpr::CurrentTimestamp => {} // Scalar expression variants (issue #205) aren't relational nodes; - // this visitor only walks the relational tree, so stop here. + // this visitor only walks the relational DAG, so stop here. QueryExpr::Column(_) | QueryExpr::Literal(_) | QueryExpr::Compare { .. } diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index 80f25fbc1..e3e065e23 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -1,9 +1,9 @@ -//! End-to-end SQL → unresolved → canonical tree lowering tests (positional IR). +//! End-to-end SQL → unresolved → canonical DAG lowering tests (positional IR). //! //! Validates the DataFusion front end: SQL parses + plans, lowers directly to //! the canonical, unresolved shape (`QueryExpr`, issue #179), and //! the shared `resolve_root` produces the positional, resolved canonical -//! tree (the same resolver the PromQL path uses). +//! DAG (the same resolver the PromQL path uses). use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; @@ -235,7 +235,7 @@ async fn select_star_with_where_folds_predicate_onto_scan() { async fn multi_aggregate_group_by_binds_columns_positionally() { // SUM(bytes)=col 3, AVG(latency)=col 2, GROUP BY service=col 1. let qe = lower("SELECT service, SUM(bytes), AVG(latency) FROM metrics GROUP BY service").await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate in the tree"); + let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate in the DAG"); assert_eq!(by, &vec![1], "GROUP BY service → column 1"); assert!( measures.contains(&AggIntent::Sum { col: Some(3) }), @@ -330,7 +330,7 @@ async fn count_ranked_topk_is_heavy_hitter() { async fn count_ranked_topk_via_alias_is_also_heavy_hitter() { // Regression for #20: aliasing `COUNT(*)` in the ORDER BY used to defeat the // SQL front-end gate. The positional `canonicalize` pass now promotes it too, - // so the aliased and inline forms produce an identical canonical tree. + // so the aliased and inline forms produce an identical canonical DAG. let inline = lower( "SELECT service, COUNT(*) FROM metrics GROUP BY service ORDER BY COUNT(*) DESC LIMIT 10", ) @@ -441,7 +441,7 @@ async fn inner_join_lowers_to_join_over_two_scans() { FROM metrics JOIN hosts ON metrics.service = hosts.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); let QueryExpr::Join { kind, left, right, .. } = join @@ -489,7 +489,7 @@ async fn join_predicate_disambiguates_shared_column_name() { FROM metrics JOIN hosts ON metrics.service = hosts.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); assert_eq!( join_eq_columns(join), [1, 4], @@ -510,7 +510,7 @@ async fn derived_table_join_disambiguates_via_alias() { JOIN (SELECT service, region FROM hosts) b ON a.service = b.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); assert_eq!( join_eq_columns(join), [0, 2], @@ -528,7 +528,7 @@ async fn derived_table_select_star_join_disambiguates_via_alias() { ON a.service = b.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); let [l, r] = join_eq_columns(join); assert_ne!( l, r, @@ -546,7 +546,7 @@ async fn self_join_disambiguates_via_aliases() { FROM metrics a JOIN metrics b ON a.service = b.service", ) .await; - let join = find_join(&qe).expect("expected a self-Join in the tree"); + let join = find_join(&qe).expect("expected a self-Join in the DAG"); assert_eq!( join_eq_columns(join), [1, 5], @@ -938,7 +938,7 @@ async fn groups_frame_is_rejected() { // ── Nested query functions: derived tables / inline views (issue #27) ─────────── -/// Collect every `AggIntent` in the tree, root-to-leaf. +/// Collect every `AggIntent` in the DAG, root-to-leaf. fn all_intents(qe: &QueryExpr) -> Vec { let mut out = Vec::new(); fn go(qe: &QueryExpr, out: &mut Vec) { @@ -981,7 +981,7 @@ fn all_intents(qe: &QueryExpr) -> Vec { async fn derived_table_aggregate_over_aggregate_nests() { // `MAX(s)` over a derived table `(SELECT service, SUM(bytes) AS s … GROUP BY // service)` — the SQL counterpart of PromQL function nesting (issue #27). - // Both reductions survive into the canonical tree: an outer `Max` over + // Both reductions survive into the canonical DAG: an outer `Max` over // the inner `Sum`. let qe = lower( "SELECT MAX(s) FROM \ @@ -997,7 +997,7 @@ async fn derived_table_aggregate_over_aggregate_nests() { intents.iter().any(|i| matches!(i, AggIntent::Sum { .. })), "inner SUM survives, got {intents:?}" ); - // The whole nested tree's output schema derives without error (positional + // The whole nested DAG's output schema derives without error (positional // resolution is total across the derived-table boundary). assert_eq!(qe.output_schema().unwrap().columns.len(), 1); } @@ -1185,8 +1185,8 @@ async fn count_distinct_carries_its_input_column() { #[tokio::test] async fn quantile_and_count_distinct_over_an_expression_bind_the_derived_column() { // A SQL aggregate has no "sample value" to fall back on, so an expression - // argument must never reach the canonical tree as `col: None` (#115). - // Since #110 it reaches the canonical tree as `col: Some(derived)` + // argument must never reach the canonical DAG as `col: None` (#115). + // Since #110 it reaches the canonical DAG as `col: Some(derived)` // instead of being rejected. for q in [ "SELECT approx_percentile_cont(bytes * 8, 0.95) FROM metrics", @@ -1333,7 +1333,7 @@ async fn time_bucketing_keeps_the_scan_predicate() { #[tokio::test] async fn a_plain_group_by_inserts_no_projection() { - // Queries that lowered before #110 must keep their exact tree shape — the + // Queries that lowered before #110 must keep their exact DAG shape — the // projection appears only when something actually needs materializing. for q in [ "SELECT service, SUM(bytes) FROM metrics GROUP BY service", diff --git a/crates/integration-tests/src/lib.rs b/crates/integration-tests/src/lib.rs index 99fb24226..59602ac07 100644 --- a/crates/integration-tests/src/lib.rs +++ b/crates/integration-tests/src/lib.rs @@ -8,7 +8,7 @@ //! are in scope. //! //! `fixtures` provides column/schema constructors used across test files. -//! Expected IR trees are always hand-constructed inside each test — nothing +//! Expected IR DAGs are always hand-constructed inside each test — nothing //! here derives or computes expected outputs. pub mod fixtures { diff --git a/crates/integration-tests/tests/cse.rs b/crates/integration-tests/tests/cse.rs index 094ffd658..b6913c110 100644 --- a/crates/integration-tests/tests/cse.rs +++ b/crates/integration-tests/tests/cse.rs @@ -2,13 +2,13 @@ //! #223). //! //! Drives the full staged pipeline this issue lands: two independently -//! lowered `QueryExpr` trees → `share_common_subtrees` (stage 1, +//! lowered `QueryExpr` DAGs → `share_common_sub_dags` (stage 1, //! `asap-types::pre_asap::cse`, run internally by `search_workload`) → //! `search_workload` (stage 2, `asap-aware-mapping`) — and asserts the //! sharing that stage 1 decides survives into stage 2's discovered //! `CandidateLogicalASAPDAGs` as one genuinely shared `TargetSubDAGCandidates`, not just one shared //! `Rc`. This is the "real caller" the issue's landing plan -//! requires before `share_common_subtrees` is allowed to exist at all (its +//! requires before `share_common_sub_dags` is allowed to exist at all (its //! predecessor, `asap-plan::cse::dedupe_subtrees`, was deleted in #192 for //! being unwired dead code). //! @@ -32,7 +32,7 @@ use asap_types::types::AccuracyTarget; /// Two workload entries that happen to submit the exact same query (a /// realistic case — two dashboards, or a query fired both standalone and as /// part of a larger batch) collapse onto one shared `Rc` after -/// `search_workload`'s internal `share_common_subtrees` pass, and onto one +/// `search_workload`'s internal `share_common_sub_dags` pass, and onto one /// genuinely-shared [`TargetSubDAGCandidates`](asap_aware_mapping::TargetSubDAGCandidates) — carrying /// every candidate discovered for it exactly once, not once per root — no /// second structural-equality pass at the post-ASAP layer needed for this @@ -40,7 +40,7 @@ use asap_types::types::AccuracyTarget; #[test] fn duplicate_workload_queries_collapse_onto_one_memo_group() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema carries - // a provable unique key — the legality gate `share_common_subtrees` + // a provable unique key — the legality gate `share_common_sub_dags` // enforces (see `asap-types::pre_asap::cse`'s module doc) — and its // `ExactAggregate(Sum)` realization is deterministic regardless of the // accuracy target, so this pins the sharing mechanism itself rather than @@ -51,7 +51,7 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // Independently lowered: not yet sharing any `Rc`, even though they are // structurally identical (`resolve_root` gives each call its own fresh - // tree). + // DAG). assert_eq!( a, b, "fixture sanity: identical query text lowers identically" @@ -60,7 +60,7 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); // roots[0] and roots[1] must have merged onto the same Rc — the - // `share_common_subtrees` pass `search_workload` runs internally. + // `share_common_sub_dags` pass `search_workload` runs internally. assert!( Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1), "search_workload must collapse the two identical roots onto one Rc" @@ -68,7 +68,7 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // The single shared root is one discovered TargetSubDAG, holding one // TargetSubDAGCandidates with consumer_count 2 — SketchAlgorithmStrategy's one - // ExactAggregate candidate *and* SharedSubtreeStrategy's share-vs- + // ExactAggregate candidate *and* SharedSubDagStrategy's share-vs- // recompute pair, exactly as `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // (asap-aware-mapping::replacement's own equivalent, internal test) // pins for the same fixture shape. @@ -141,7 +141,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { /// Single-query CSE (a repeated sub-expression within one query) also /// survives through `search_workload`: the two grouped-`Aggregate` branches /// of a `BinaryOp` collapse to one shared `Rc` in the internal -/// `share_common_subtrees` pass, and to one shared `TargetSubDAGCandidates` (with +/// `share_common_sub_dags` pass, and to one shared `TargetSubDAGCandidates` (with /// `consumer_count == 2`, one per branch) here. #[test] fn single_query_repeated_subexpression_shares_one_memo_group() { diff --git a/crates/integration-tests/tests/exact_composition.rs b/crates/integration-tests/tests/exact_composition.rs index fe72aea26..7e2e7c12a 100644 --- a/crates/integration-tests/tests/exact_composition.rs +++ b/crates/integration-tests/tests/exact_composition.rs @@ -768,8 +768,8 @@ fn dag_export_carries_explicit_stage_and_plain_schema_for_a_composed_plan() { .assemble_selected_dag(root) .unwrap() .unwrap(); - let graph = dag_export::export_summary(&composed); - let node = &graph.nodes[graph.root as usize]; + let dag = dag_export::export_summary(&composed); + let node = &dag.nodes[dag.root as usize]; assert_eq!(node.kind, "ValueOperation"); assert_eq!(node.detail["timing"], "query_time"); assert!(node.detail["operation"] diff --git a/crates/integration-tests/tests/nested.rs b/crates/integration-tests/tests/nested.rs index af25af3be..1e3e34ebd 100644 --- a/crates/integration-tests/tests/nested.rs +++ b/crates/integration-tests/tests/nested.rs @@ -101,13 +101,13 @@ fn q23_sum_by_job_over_filtered_scan() { ); } -// #25 — binary op over two complex subtrees +// #25 — binary op over two complex sub-DAGs // LHS: sum by (job) over rate over filtered scan // schema [ts, value, job, status]; outer by=[2] (job) // RHS: sum by (job) over rate over bare scan // schema [ts, value, job]; outer by=[2] (job) #[test] -fn q25_div_over_complex_subtrees() { +fn q25_div_over_complex_sub_dags() { let lhs_scan = QueryExpr::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), @@ -229,7 +229,7 @@ fn q53_outer_group_key_absent_from_nested_aggregate() { // #52 — an outer group key referenced by neither binary-op side (`__name__`) // still resolves. Each `or` side is bound independently against its own -// sub-tree, so `__name__` is seeded as an inherited column on both. Each side +// sub-DAG, so `__name__` is seeded as an inherited column on both. Each side // references only `env` (its matcher), so its schema is [ts, value, env, // __name__] (referenced `env` first, inherited `__name__` appended) → the // outer `by (__name__)` resolves to col 3 on both sides. The `or` carries the diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index 2df295725..38197eea2 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -1,4 +1,4 @@ -//! Planner-selected summaries over raw samples compile as precompute graphs +//! Planner-selected summaries over raw samples compile as precompute DAGs //! and produce the same estimates as feeding their kernel sample by sample. use std::{collections::BTreeMap, collections::BTreeSet, rc::Rc, sync::Arc}; @@ -143,7 +143,7 @@ fn execute( source, Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, )]); - let graph = program.instantiate(sources).unwrap(); + let physical_dag = program.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Ingestion { window_start_ms: 0, @@ -154,7 +154,10 @@ fn execute( ) .unwrap(); block_on(async { - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let mut result = Vec::new(); while let Some(batch) = stream.next().await { for row in batch.unwrap().rows() { diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 1a8d364e0..73f781443 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -29,7 +29,7 @@ use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; use asap_types::pre_asap::schema::DataType; use asap_types::types::AccuracyTarget; -/// This crate has no "bind me one tree" public API any more — +/// This crate has no "bind me one DAG" public API any more — /// `SketchAlgorithmStrategy::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the @@ -643,7 +643,7 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { ); let shared = - asap_types::post_asap::share_common_summary_subtrees(vec![("ratio", node.clone())]); + asap_types::post_asap::share_common_summary_sub_dags(vec![("ratio", node.clone())]); let SummaryExpr::BinaryOp { lhs, rhs, .. } = &shared[0].1.expr else { panic!("expected binary ratio") }; @@ -892,7 +892,7 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { /// └─ KeepPreAsap(TimeRange{5m} → Scan) → {ts, value} /// ``` /// -/// The nested tree exercises both realizations: the approximate quantile +/// The nested DAG exercises both realizations: the approximate quantile /// binds a KLL sketch + readout; the per-series `rate` binds the exact /// counter-reset-aware accumulator (no estimate — its state is the value). #[test] @@ -1011,7 +1011,7 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { /// An exact workload binds zero sketches: `sum by (job) (m)` at /// `AccuracyTarget::Exact` still gets its mergeable exact accumulator, and -/// `avg(m)` (non-mergeable) passes through as a whole logical subtree. +/// `avg(m)` (non-mergeable) passes through as a whole logical sub-DAG. #[test] fn promql_exact_workload_binds_accumulators_not_sketches() { let pre_asap = lower_promql("sum by (job) (http_requests_total)", AccuracyTarget::Exact) diff --git a/crates/integration-tests/tests/schema.rs b/crates/integration-tests/tests/schema.rs index 6024619cc..7d7e2570a 100644 --- a/crates/integration-tests/tests/schema.rs +++ b/crates/integration-tests/tests/schema.rs @@ -1,7 +1,7 @@ //! `Schema::closed` propagation — open/closed invariant tests. //! //! Verifies that `QueryExpr::output_schema()` propagates the open/closed -//! completeness flag correctly through a lowered query tree. +//! completeness flag correctly through a lowered query DAG. //! //! Key invariant: a PromQL scan is always `closed: false` (open) because its //! label set is runtime-only. The schema freezes to `closed: true` exactly at diff --git a/crates/integration-tests/tests/sql_to_post_asap.rs b/crates/integration-tests/tests/sql_to_post_asap.rs index d2d418222..5d10fb2f5 100644 --- a/crates/integration-tests/tests/sql_to_post_asap.rs +++ b/crates/integration-tests/tests/sql_to_post_asap.rs @@ -11,9 +11,9 @@ //! //! `lower_promql` returns a *bare* `QueryExpr::Aggregate` for a top-level //! aggregation (`sum by (job) (m)`, `quantile(0.99, …)`), so [`realize`] can -//! bind it directly at the tree root. `lower_sql` never does: DataFusion's +//! bind it directly at the DAG root. `lower_sql` never does: DataFusion's //! planner always wraps even a single, unaliased aggregate in an identity -//! `Project` (confirmed below), so a SQL tree's *root* is normally `Project { +//! `Project` (confirmed below), so a SQL DAG's *root* is normally `Project { //! child: Aggregate { .. } }`. Final materialization retains that projection //! as a query-time value operation and independently plans its child, keeping //! both SELECT-list semantics and the summary-bound aggregate visible. @@ -37,7 +37,7 @@ use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -/// This crate has no "bind me one tree" public API any more — +/// This crate has no "bind me one DAG" public API any more — /// `SketchAlgorithmStrategy::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the @@ -752,7 +752,7 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { /// An exact workload binds zero sketches: `SUM(bytes) GROUP BY service` at /// `AccuracyTarget::Exact` still gets its mergeable exact accumulator, and -/// `AVG(bytes)` (non-mergeable) stays a whole logical subtree untouched. SQL +/// `AVG(bytes)` (non-mergeable) stays a whole logical sub-DAG untouched. SQL /// counterpart of `promql_to_post_asap.rs`'s /// `promql_exact_workload_binds_accumulators_not_sketches`. #[tokio::test] diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index d3d73121b..2dead4098 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -174,8 +174,8 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() alternative["lifecycle"]["kind"] == "shared" && alternative["rejection"] == "unsupported_by_runtime" })); - assert!(exported["graph"]["nodes"].as_array().is_some()); - let summary_node = exported["graph"]["nodes"] + assert!(exported["dag"]["nodes"].as_array().is_some()); + let summary_node = exported["dag"]["nodes"] .as_array() .unwrap() .iter() @@ -982,7 +982,7 @@ fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { (dag, ingestion_binary) } -/// Execute a timed DAG's precompute and query graphs over `samples` +/// Execute a timed DAG's precompute and query DAGs over `samples` /// (`(metric, job, seconds, value)`) at 300s; returns the root's values. fn execute_timed( dag: &asap_types::post_asap::PostAsapDag, diff --git a/crates/planner/tests/summary_sharing.rs b/crates/planner/tests/summary_sharing.rs index 829c38a9c..069bda8ff 100644 --- a/crates/planner/tests/summary_sharing.rs +++ b/crates/planner/tests/summary_sharing.rs @@ -21,7 +21,7 @@ use asap_frontend_promql::lower_promql_workload; use asap_frontend_sql::SqlCatalog; use asap_planner::{e2e_plan, FrontendInput, UserInput}; use asap_types::post_asap::{ - share_common_summary_subtrees, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, + share_common_summary_sub_dags, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, SketchQuery, }; use asap_types::post_asap::{ @@ -542,7 +542,7 @@ fn certified_frequency_readouts_share_one_univmon_state() { }) .collect(); let mut states: Vec> = Vec::new(); - for (_, root) in share_common_summary_subtrees(assembled) { + for (_, root) in share_common_summary_sub_dags(assembled) { assert!(root.guarantee.is_some(), "{:?}", root.expr); let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { panic!("summary readout: {:?}", root.expr); diff --git a/crates/types/Cargo.toml b/crates/types/Cargo.toml index 6b3536575..940ecd90b 100644 --- a/crates/types/Cargo.toml +++ b/crates/types/Cargo.toml @@ -10,7 +10,7 @@ edition = "2021" [dependencies] # "rc" — QueryExpr's child fields are Rc> (issue #212, #222: # shared sub-expressions), and Rc's Serialize/Deserialize impls live behind -# this feature flag. dag_export.rs / DagNode already flatten the tree to a +# this feature flag. dag_export.rs / DagNode already flatten the DAG to a # node+edge list for JSON export, so this does not change wire format — a # shared Rc still (de)serializes as an ordinary inline value, once per # reference, exactly like the old Box. diff --git a/crates/types/src/cost.rs b/crates/types/src/cost.rs index 66f4febc1..660d3bb73 100644 --- a/crates/types/src/cost.rs +++ b/crates/types/src/cost.rs @@ -420,8 +420,8 @@ where /// Whole-selected-workload cost/benefit — one query's (or one workload /// batch's) aggregate baseline, selected, and benefit, built from /// [`sum_workload_costs`] over that scope's own per-decision node -/// annotations. See [`crate::dag_export::NamedGraph::workload_cost`] / -/// [`crate::dag_export::WorkloadGraph::workload_cost`]. +/// annotations. See [`crate::dag_export::NamedDAG::workload_cost`] / +/// [`crate::dag_export::WorkloadDAG::workload_cost`]. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct WorkloadCostSummary { pub baseline_cost: CostAnnotation, diff --git a/crates/types/src/dag_export.rs b/crates/types/src/dag_export.rs index 0fe6977f9..11055d4a8 100644 --- a/crates/types/src/dag_export.rs +++ b/crates/types/src/dag_export.rs @@ -1,25 +1,25 @@ -//! Export the pre-ASAP [`QueryExpr`] tree as a generic node/edge graph, for tools +//! Export the pre-ASAP [`QueryExpr`] DAG as a generic node/edge DAG, for tools //! that need to render or diff the IR (the `dag_export` example + the //! `tools/dag-viewer` viewer — see issue #133) rather than walk it in Rust. //! -//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged tree +//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged DAG //! (`Rc` children nested inside each variant's own field). This module //! flattens that into an explicit node list + child-id edges — the shape a -//! generic graph renderer wants — and additionally tags each node with +//! generic DAG renderer wants — and additionally tags each node with //! [`structural_hash`](crate::pre_asap::cse::structural_hash), so a caller -//! with several exported queries can spot identical subtrees (a +//! with several exported queries can spot identical sub-DAGs (a //! shared `Scan`, a repeated `Aggregate` shape, …) by comparing hashes //! rather than re-implementing `QueryExpr: PartialEq` structural comparison //! client-side. //! //! This is literally the same hashing -//! [`share_common_subtrees`](crate::pre_asap::cse::share_common_subtrees) +//! [`share_common_sub_dags`](crate::pre_asap::cse::share_common_sub_dags) //! uses to bucket candidates in its `InternTable` (issue #223 stage 3) — not -//! a parallel reimplementation. `tools/dag-viewer`'s "shared subtree" +//! a parallel reimplementation. `tools/dag-viewer`'s "shared sub-DAG" //! highlighting is still a *proxy* for real CSE, though: a hash match here //! only means two nodes are legal `InternTable` bucket-mates (same coarse //! hash), the same candidate-narrowing step `structural_hash` performs -//! inside `InternTable::intern` — it does not mean `share_common_subtrees` +//! inside `InternTable::intern` — it does not mean `share_common_sub_dags` //! actually ran on this data and merged them onto one `Rc` (that also //! requires the `PartialEq` check `InternTable::intern` performs, and the //! `Schema::has_unique_key` legality gate, neither of which this export @@ -30,11 +30,11 @@ //! //! [`DagNode`] also carries `notes: Vec<`[`DagNote`]`>`, always empty coming //! out of [`export`]. It exists so a *higher* layer — one that depends on -//! `asap_types`, never the reverse — can annotate an already-exported graph +//! `asap_types`, never the reverse — can annotate an already-exported DAG //! after the fact without this module needing to know anything about that //! layer's concepts. Concretely: `asap-aware-mapping`'s `explanation` module //! (issue #257) computes `structural_hash` over the same `QueryExpr` -//! subtrees this module does (via the identical function). The devtools +//! sub-DAGs this module does (via the identical function). The devtools //! exporter uses that hash to narrow candidates, then compares //! `ReplacementExplanation::target` with [`DagNode::source_expr`] for a //! collision-safe match before pushing a [`DagNote`] onto the node. @@ -59,7 +59,7 @@ pub struct DagNode { pub id: u32, /// The `QueryExpr` variant name (e.g. `"Aggregate"`). pub kind: &'static str, - /// Short human-readable summary for a node's collapsed on-graph label. + /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, /// Output schema carried by every exported node. Edge renderers use the @@ -75,16 +75,16 @@ pub struct DagNode { #[serde(skip_serializing_if = "Option::is_none")] pub workload_node_id: Option, /// [`structural_hash`](crate::pre_asap::cse::structural_hash) of the - /// subtree rooted at this node — the exact same function `cse`'s + /// sub-DAG rooted at this node — the exact same function `cse`'s /// `InternTable` uses to bucket CSE candidates, so two nodes hash /// equally here iff they would land in the same `InternTable` bucket. /// See the module doc for what a hash match here does and doesn't /// guarantee. /// /// `None` for the same reason `source_expr` is `None` — a post-ASAP- - /// originated node in an [`export_post_asap`] merged graph has no + /// originated node in an [`export_post_asap`] merged DAG has no /// `QueryExpr` to hash. Omitted from JSON entirely (rather than, say, - /// serialized as `0`) so a consumer's shared-subtree-by-hash pass can + /// serialized as `0`) so a consumer's shared-sub-DAG-by-hash pass can /// tell "no hash" apart from a real hash that happens to collide with a /// placeholder — `0` is a legal `structural_hash` output, not a safe /// sentinel. @@ -97,7 +97,7 @@ pub struct DagNode { /// /// `None` for a node with no corresponding pre-ASAP `QueryExpr` at all — /// only possible for a post-ASAP-originated node inside a merged - /// [`export_post_asap`] graph (a `SummaryAgg`/`SummaryJoin`/… node has no + /// [`export_post_asap`] DAG (a `SummaryAgg`/`SummaryJoin`/… node has no /// single `QueryExpr` it corresponds to). Every node [`export`] itself /// produces is pre-ASAP by construction and always carries `Some`. #[serde(skip)] @@ -117,8 +117,8 @@ pub struct DagNode { #[serde(default, skip_serializing_if = "Vec::is_empty")] pub notes: Vec, /// Explicit workload-level decision that produced or carried this node. - /// Present only in `post_graph`; consumers must read this rather than - /// infer strategy provenance from labels, hashes, or graph similarity. + /// Present only in `post_dag`; consumers must read this rather than + /// infer strategy provenance from labels, hashes, or DAG similarity. #[serde(skip_serializing_if = "Option::is_none")] pub decision: Option, } @@ -164,19 +164,19 @@ pub struct DagDecision { #[serde(skip_serializing_if = "Option::is_none")] pub selected_cost: Option, /// `baseline_cost.value - selected_cost.value` under `baseline_cost`'s - /// own baseline — for a winning `SharedSubtreeStrategy`/`CseShare` + /// own baseline — for a winning `SharedSubDagStrategy`/`CseShare` /// decision this *is* "avoided recomputation for a shared sub-DAG" (one /// of `dag_export`'s issue #286 granularity items): the baseline is - /// exactly the cost of recomputing this subtree independently at every + /// exactly the cost of recomputing this sub-DAG independently at every /// consumer, so the benefit is exactly what sharing avoided. #[serde(skip_serializing_if = "Option::is_none")] pub benefit: Option, } -/// A cost/benefit annotation attributed to one specific graph edge (`from` +/// A cost/benefit annotation attributed to one specific DAG edge (`from` /// -> `to`, in [`DagNode::children`]'s direction) rather than to a node — /// issue #286's "edge cost only when genuinely attributable to the edge" -/// granularity item. Graph structure alone cannot determine transfer, +/// granularity item. DAG structure alone cannot determine transfer, /// materialization, or read cost. A higher layer may attach this annotation /// only when physical evidence attributes cost to this exact edge; this /// module never derives one from structural node counts. @@ -187,15 +187,15 @@ pub struct EdgeCostAnnotation { pub cost: CostAnnotation, } -/// One query's exported graph. `nodes[root as usize]` is the tree's root. +/// One query's exported DAG. `nodes[root as usize]` is the DAG's root. #[derive(Debug, Clone, Serialize)] -pub struct DagGraph { +pub struct ExportDAG { pub nodes: Vec, pub root: u32, /// See [`EdgeCostAnnotation`]. Always empty unless a higher layer /// explicitly populated it (same layering rule as [`DagNode::notes`]); /// omitted from JSON entirely when empty, so every existing producer of - /// [`DagGraph`] (every call to [`export`]/[`export_summary`]) is + /// [`ExportDAG`] (every call to [`export`]/[`export_summary`]) is /// unaffected. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub edge_annotations: Vec, @@ -203,57 +203,57 @@ pub struct DagGraph { /// A single named query within a multi-query export. #[derive(Debug, Clone, Serialize)] -pub struct NamedGraph { +pub struct NamedDAG { pub name: String, - /// The original query text (SQL or PromQL) this graph was lowered from, - /// for display alongside the graph — not used by `export` itself, since + /// The original query text (SQL or PromQL) this DAG was lowered from, + /// for display alongside the DAG — not used by `export` itself, since /// that only sees the already-lowered `QueryExpr`. Optional because not - /// every producer of a `NamedGraph` has the source text on hand. + /// every producer of a `NamedDAG` has the source text on hand. #[serde(skip_serializing_if = "Option::is_none")] pub source: Option, - pub graph: DagGraph, + pub dag: ExportDAG, /// Concrete post-ASAP replacement sites discovered for this query — see /// [`TargetReplacement`]. Always empty coming out of anything in this /// module (same layering rule as [`DagNode::notes`]: `asap_types` never /// runs `asap-aware-mapping`'s search itself); a higher layer populates /// this after the fact, e.g. the `dag_export` devtools binary's /// `--post-asap` flag. Omitted from the JSON entirely when empty, so - /// every existing producer/consumer of `NamedGraph` (in particular every + /// every existing producer/consumer of `NamedDAG` (in particular every /// invocation of `dag_export` without `--post-asap`) keeps emitting and /// parsing exactly the same shape it always has. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub replacements: Vec, - /// One merged "whole query, but post-ASAP" graph — see + /// One merged "whole query, but post-ASAP" DAG — see /// [`export_post_asap`] for how a higher layer builds this. Unlike /// [`TargetReplacement::before`]/`::after` (small, self-contained /// before/after pairs, one per independently-discovered replacement - /// site), this is a single flattened [`DagGraph`] spanning the whole + /// site), this is a single flattened [`ExportDAG`] spanning the whole /// query: every node that has no winning replacement renders as an /// ordinary pre-ASAP [`DagNode`] (same shape [`export`] itself /// produces), and every node that does splices in its winning - /// candidate's shape instead — a rewritten [`QueryExpr`] subtree, or a - /// bound `SummaryNode` subtree, rendered inline in the very same node + /// candidate's shape instead — a rewritten [`QueryExpr`] sub-DAG, or a + /// bound `SummaryNode` sub-DAG, rendered inline in the very same node /// list. `None` unless a higher layer explicitly built one (e.g. the /// `dag_export` devtools binary's `--post-asap` flag); omitted from the /// JSON entirely when absent, so every existing producer/consumer of - /// `NamedGraph` is unaffected. + /// `NamedDAG` is unaffected. #[serde(default, skip_serializing_if = "Option::is_none")] - pub post_graph: Option, + pub post_dag: Option, /// This query's own selected-workload cost/benefit — one of issue /// #286's granularity items. Built by summing *this query's own* - /// `post_graph` decision-node cost annotations, deduplicated by + /// `post_dag` decision-node cost annotations, deduplicated by /// `decision.id` **within this one query only** (a decision spanning /// several nodes in this query's own replacement region is still /// counted once here). `None` unless a higher layer built one (same - /// `--post-asap`-gated pattern as `post_graph`); omitted from JSON when + /// `--post-asap`-gated pattern as `post_dag`); omitted from JSON when /// absent. /// /// This does **not** dedupe across queries: a target shared by two /// queries (e.g. a common `Scan` after workload-wide CSE) is counted /// once in *each* query's own `workload_cost` — summing several - /// `NamedGraph.workload_cost` values by hand double-counts any decision + /// `NamedDAG.workload_cost` values by hand double-counts any decision /// shared between them. For a cross-query total that dedupes correctly, - /// use [`WorkloadGraph::workload_cost`] instead, which is built + /// use [`WorkloadDAG::workload_cost`] instead, which is built /// specifically to cover every query in one pass. #[serde(default, skip_serializing_if = "Option::is_none")] pub workload_cost: Option, @@ -266,12 +266,12 @@ pub struct NamedGraph { } /// A batch of named queries — the shape the viewer's multi-query / compare -/// mode reads (each query starts its own `DagGraph`; shared-subtree +/// mode reads (each query starts its own `ExportDAG`; shared-sub-DAG /// highlighting is done by the viewer, matching `DagNode::hash` across /// queries). #[derive(Debug, Clone, Serialize)] -pub struct WorkloadGraph { - pub queries: Vec, +pub struct WorkloadDAG { + pub queries: Vec, /// The selected multi-query workload's own cost/benefit, deduplicated /// across every query in `queries` (not just within one) — the /// "Selecting ... multiple queries ... display correct Pre/Post-ASAP @@ -298,17 +298,17 @@ pub struct WorkloadGraph { // `tools/dag-viewer` to render without needing to know anything about // `asap-aware-mapping`'s own vocabulary. // -// A single whole-query "post-ASAP tree" isn't attempted here, and isn't +// A single whole-query "post-ASAP DAG" isn't attempted here, and isn't // representable in the current type system either: `SummaryExpr` has no // variant letting a `SummaryNode` be embedded back inside a plain // `QueryExpr`'s child slot (`QueryExpr`'s own children are always // `Rc`, never `Rc`), so there is no way to splice a -// post-ASAP binding back into its original pre-ASAP tree in place. Inventing +// post-ASAP binding back into its original pre-ASAP DAG in place. Inventing // a bridge type for that is a real `asap_types`/`asap-aware-mapping` IR // design decision, well beyond what a devtools visualization export should // decide unilaterally. Instead, each independently-discovered replacement // target gets its own small, self-contained `before`/`after` pair — the -// target's own pre-ASAP subtree, and either the winning `SummaryNode` or the +// target's own pre-ASAP sub-DAG, and either the winning `SummaryNode` or the // winning rewritten `QueryExpr`, both of which *are* fully representable // today via [`export`]/[`export_summary`] as-is. @@ -323,7 +323,7 @@ pub struct WorkloadGraph { /// way `DagNode::hash` lets a higher layer re-identify a pre-ASAP node (a /// `SummaryNode` is always freshly exported for exactly one /// [`TargetReplacementAfter::Summary`] site, never matched back against a -/// separately-exported graph the way pre-ASAP notes are). +/// separately-exported DAG the way pre-ASAP notes are). /// /// Several of `SummaryExpr`'s own fields (`SummaryFamilyType`, /// `GroupingStrategy`, `SketchQuery`) derive neither `Serialize` nor @@ -343,7 +343,7 @@ pub struct SummaryDagNode { pub id: u32, /// The `SummaryExpr` variant name (e.g. `"SummaryAgg"`). pub kind: &'static str, - /// Short human-readable summary for a node's collapsed on-graph label. + /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, /// Child node ids, in the variant's field order (e.g. `SummaryJoin` is @@ -363,11 +363,11 @@ pub struct SummaryDagNode { /// target (issue #172) — `asap_aware_mapping::replacement::RejectedCandidate` /// re-shaped into this crate's own crate-agnostic vocabulary, the same /// layering rule as [`TargetReplacement`]. Carried on -/// [`NamedGraph::rejections`] so a renderer can explain *why* a target kept +/// [`NamedDAG::rejections`] so a renderer can explain *why* a target kept /// its raw/pre-ASAP form, not only what won elsewhere. #[derive(Debug, Clone, Serialize)] pub struct TargetRejection { - /// Id of the [`DagNode`] in this query's own `graph.nodes` the refused + /// Id of the [`DagNode`] in this query's own `DAG.nodes` the refused /// candidate targeted. pub target_pre_id: u32, /// Which strategy considered the candidate. @@ -378,10 +378,10 @@ pub struct TargetRejection { pub error: AccuracyError, } -/// One post-ASAP `SummaryNode` tree, flattened the same way [`DagGraph`] -/// flattens a pre-ASAP `QueryExpr` tree. +/// One post-ASAP `SummaryNode` DAG, flattened the same way [`ExportDAG`] +/// flattens a pre-ASAP `QueryExpr` DAG. #[derive(Debug, Clone, Serialize)] -pub struct SummaryDagGraph { +pub struct SummaryDAG { pub nodes: Vec, pub root: u32, } @@ -390,22 +390,22 @@ pub struct SummaryDagGraph { /// — post-order, one [`SummaryDagNode`] per [`SummaryExpr`] variant, no /// memoization of repeated `Rc` references (a shared /// sub-expression reachable through two parents is flattened twice, into two -/// separate node entries — the same "this is a flattened tree view, not a -/// pointer-identity-preserving graph" behavior [`build`] already has for +/// separate node entries — the same "this is a flattened DAG view, not a +/// pointer-identity-preserving DAG" behavior [`build`] already has for /// `QueryExpr`). /// -/// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP subtree beneath it -/// as a nested [`DagGraph`] (via [`export(inner)`](export)) inside its own -/// `detail` field (`{"pre_asap_subgraph": }`) rather than trying to +/// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP sub-DAG beneath it +/// as a nested [`ExportDAG`] (via [`export(inner)`](export)) inside its own +/// `detail` field (`{"pre_asap_sub_dag": }`) rather than trying to /// flatten it into this same node list — [`DagNode`] and [`SummaryDagNode`] /// are different types with different id spaces, so mixing them into one /// `Vec` isn't type-safe; nesting is. `label` for a `KeepPreAsap` node is -/// `format!("KeepPreAsap({kind})")`, where `kind` is the inner subtree's own +/// `format!("KeepPreAsap({kind})")`, where `kind` is the inner sub-DAG's own /// top-level `DagNode::kind`. -pub fn export_summary(node: &SummaryNode) -> SummaryDagGraph { +pub fn export_summary(node: &SummaryNode) -> SummaryDAG { let mut nodes = Vec::new(); let root = build_summary(node, &mut nodes); - SummaryDagGraph { nodes, root } + SummaryDAG { nodes, root } } fn push_summary_node( @@ -452,9 +452,9 @@ fn family_label(family: &crate::post_asap::SummaryFamilyType) -> String { /// `DagNode` of its own (see [`build_summary`]/[`build_summary_hybrid`], its /// only two callers, both of which special-case it before ever reaching /// this function). Factored out so [`build_summary`] (nests a `KeepPreAsap` -/// leaf's pre-ASAP subtree as its own [`SummaryDagGraph`]) and -/// [`build_summary_hybrid`] (splices that same subtree directly into a -/// shared [`DagGraph`] node list — see [`export_post_asap`]) can't drift +/// leaf's pre-ASAP sub-DAG as its own [`SummaryDAG`]) and +/// [`build_summary_hybrid`] (splices that same sub-DAG directly into a +/// shared [`ExportDAG`] node list — see [`export_post_asap`]) can't drift /// apart on how every *other* variant's own shape is described, since /// nothing about that description differs between the two. macro_rules! define_summary_kind_tags { @@ -590,10 +590,10 @@ fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { /// file's own exhaustive style for `QueryExpr` in [`build`]. fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - let pre_asap_subgraph = export(inner); - let inner_kind = pre_asap_subgraph.nodes[pre_asap_subgraph.root as usize].kind; + let pre_asap_sub_dag = export(inner); + let inner_kind = pre_asap_sub_dag.nodes[pre_asap_sub_dag.root as usize].kind; let label = format!("KeepPreAsap({inner_kind})"); - let detail = serde_json::json!({ "pre_asap_subgraph": pre_asap_subgraph }); + let detail = serde_json::json!({ "pre_asap_sub_dag": pre_asap_sub_dag }); return push_summary_node( nodes, "KeepPreAsap", @@ -620,16 +620,16 @@ fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { #[derive(Debug, Clone, Serialize)] pub struct TargetReplacement { /// Stable id of this workload-level winning decision. Nodes in - /// [`NamedGraph::post_graph`] produced by this decision carry the same id, + /// [`NamedDAG::post_dag`] produced by this decision carry the same id, /// so renderers can explain a clicked post-ASAP node without guessing by - /// label, hash, or graph shape. + /// label, hash, or DAG shape. pub decision_id: u32, - /// Id of the [`DagNode`] (in this query's own `graph.nodes`, i.e. the - /// [`NamedGraph`] this `TargetReplacement` is attached to) this - /// replacement's `before` subtree is rooted at. + /// Id of the [`DagNode`] (in this query's own `DAG.nodes`, i.e. the + /// [`NamedDAG`] this `TargetReplacement` is attached to) this + /// replacement's `before` sub-DAG is rooted at. pub target_pre_id: u32, /// Human label for which strategy proposed the winning candidate — - /// e.g. `"Sketch"` / `"HydraGrouping"` / `"SharedSubtree"` / + /// e.g. `"Sketch"` / `"HydraGrouping"` / `"SharedSubDAG"` / /// `"AvgToSumCountRewrite"` / `"Rollup"`. The higher layer derives this from /// `ReplacementProvenance` plus which strategy's shape actually /// produced the winning candidate; `asap_types` has no opinion on the @@ -648,9 +648,9 @@ pub struct TargetReplacement { /// doesn't estimate a numeric cost for this candidate shape (see that /// field's own doc upstream). pub cost: f64, - /// The target's own pre-ASAP subtree, before replacement — literally + /// The target's own pre-ASAP sub-DAG, before replacement — literally /// `export(target)` for the `TargetSubDAGCandidates`'s own `target`, reused as-is. - pub before: DagGraph, + pub before: ExportDAG, pub after: TargetReplacementAfter, /// Structured baseline/selected/benefit cost annotations for this one /// replacement region — issue #286's "replacement-region baseline @@ -658,7 +658,7 @@ pub struct TargetReplacement { /// consistent with `cost` above: `selected_cost.value == Some(cost)` /// whenever `cost` is finite, `None`/`Unavailable` whenever it is /// `NaN`. Baseline and selected values require complete, scope-matched - /// physical evidence; neither is inferred from logical graph structure. + /// physical evidence; neither is inferred from logical DAG structure. #[serde(skip_serializing_if = "Option::is_none")] pub baseline_cost: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -671,25 +671,25 @@ pub struct TargetReplacement { /// or a still-pre-ASAP-shaped structural rewrite, mirroring /// `asap_aware_mapping::replacement::Replacement`'s own two variants. /// -/// Serializes as `{"kind": "Summary"|"Rewrite", "graph": {...}}` (serde's +/// Serializes as `{"kind": "Summary"|"Rewrite", "DAG": {...}}` (serde's /// adjacently-tagged representation for a `#[serde(tag = "kind", content = -/// "graph")]` enum) — this exact shape is a cross-team contract with +/// "DAG")]` enum) — this exact shape is a cross-team contract with /// `tools/dag-viewer`'s fixture data, so it isn't incidental: changing it /// needs coordinating with that side, not just a local refactor here. #[derive(Debug, Clone, Serialize)] -#[serde(tag = "kind", content = "graph")] +#[serde(tag = "kind", content = "dag")] pub enum TargetReplacementAfter { /// A `Replacement::Summary` candidate — a genuine post-ASAP binding. - Summary(SummaryDagGraph), + Summary(SummaryDAG), /// A `Replacement::Rewrite` candidate — still pre-ASAP shaped (CSE /// share/recompute, `AvgToSumOverCountStrategy`, and `RollupStrategy` - /// all produce this kind), so this reuses [`DagGraph`]/[`export`] too, + /// all produce this kind), so this reuses [`ExportDAG`]/[`export`] too, /// not a new type. - Rewrite(DagGraph), + Rewrite(ExportDAG), } -/// Flatten `expr` into a [`DagGraph`]. -pub fn export(expr: &QueryExpr) -> DagGraph { +/// Flatten `expr` into a [`ExportDAG`]. +pub fn export(expr: &QueryExpr) -> ExportDAG { let mut nodes = Vec::new(); // One cache for the whole export — persisted across every `build`/ // `push_node` call, not reset per node, so `structural_hash` memoizes @@ -701,7 +701,7 @@ pub fn export(expr: &QueryExpr) -> DagGraph { // callback regardless (so `export_post_asap` can share this exact // per-variant traversal instead of duplicating it). let root = build(expr, &mut nodes, &mut cache, &mut |_| None); - DagGraph { + ExportDAG { nodes, root, edge_annotations: Vec::new(), @@ -709,7 +709,7 @@ pub fn export(expr: &QueryExpr) -> DagGraph { } /// What a higher layer found for one specific pre-ASAP node when building a -/// merged post-ASAP graph via [`export_post_asap`] — see that function's own +/// merged post-ASAP DAG via [`export_post_asap`] — see that function's own /// doc for the full design. `asap_types` has no opinion on *how* this is /// decided (that's `asap_aware_mapping::replacement::search_workload_with` + /// `CandidateLogicalASAPDAGs::cost_sorted`'s job, a higher layer, exactly the layering rule @@ -735,16 +735,16 @@ pub enum PostAsapSubstitution { }, } -/// Build one merged "whole query, but post-ASAP" [`DagGraph`] by walking +/// Build one merged "whole query, but post-ASAP" [`ExportDAG`] by walking /// `root`'s ordinary pre-ASAP shape and, at every node, asking `find_winner` /// whether *that exact node* has a winning replacement — if so, splicing /// the replacement's own shape in at that position instead, in the very -/// same flattened node list (not a nested sub-graph the way +/// same flattened node list (not a nested sub-DAG the way /// [`TargetReplacement::before`]/`::after` — small, independent, per-site /// before/after pairs — already do; see this file's "Post-ASAP replacement /// export" section doc for why *that* design doesn't attempt a single /// whole-query composite, and why this one can: this is a synthetic -/// id/edge list, the same kind of thing [`DagGraph`] already is for the +/// id/edge list, the same kind of thing [`ExportDAG`] already is for the /// pre-ASAP side, not a real `QueryExpr`/`SummaryNode` value with a type /// system to satisfy). /// @@ -762,7 +762,7 @@ pub enum PostAsapSubstitution { /// substitution's own immediate top level (only on that substitution's /// *descendants*, which get an ordinary fresh call same as any other node). /// This matters for correctness, not just efficiency: -/// `SharedSubtreeStrategy`'s own "build once and share" candidate is +/// `SharedSubDagStrategy`'s own "build once and share" candidate is /// `Replacement::Rewrite(Rc::clone(target))` — literally the *same* value /// as the target it's a candidate for. Re-querying `find_winner` on that /// candidate's own top level would find the identical group and its @@ -773,14 +773,14 @@ pub enum PostAsapSubstitution { pub fn export_post_asap( root: &QueryExpr, find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> DagGraph { +) -> ExportDAG { let mut nodes = Vec::new(); let mut cache = HashCache::new(); let root_id = build(root, &mut nodes, &mut cache, find_winner); deduplicate_pointer_shared_nodes(nodes, root_id) } -fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> DagGraph { +fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> ExportDAG { let mut by_source_ptr = HashMap::::new(); let mut old_to_new = vec![0_u32; nodes.len()]; let mut deduplicated = Vec::with_capacity(nodes.len()); @@ -807,7 +807,7 @@ fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> DagGraph deduplicated.push(node); } - DagGraph { + ExportDAG { nodes: deduplicated, root: old_to_new[root as usize], edge_annotations: Vec::new(), @@ -868,11 +868,11 @@ define_query_kind_tags! { QueryExpr::BinaryOp { .. } => "BinaryOp", } -/// Push one flattened node for `expr`. `expr` is the *whole* subtree this +/// Push one flattened node for `expr`. `expr` is the *whole* sub-DAG this /// node represents (not just its own fields) — `hash` is /// [`structural_hash(expr)`](structural_hash), the identical function and /// the identical input `InternTable::intern` would hash for this same -/// subtree, so this node's `hash` matches what `cse::share_common_subtrees` +/// sub-DAG, so this node's `hash` matches what `cse::share_common_sub_dags` /// would bucket it under. `kind` is [`kind_tag(expr)`](kind_tag), not a /// caller-supplied argument — see that function's doc for why. fn push_node( @@ -907,7 +907,7 @@ fn push_node( /// Push one flattened node with no corresponding pre-ASAP `QueryExpr` at /// all — a post-ASAP-originated node inside [`export_post_asap`]'s merged -/// graph (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). +/// DAG (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). /// `hash`/`source_expr`-based re-identification (see [`DagNode::hash`]'s own /// doc) has no meaning for a node with no `QueryExpr` behind it, so this /// pushes a fixed placeholder hash (`0`) and `source_expr: None` rather than @@ -942,12 +942,12 @@ fn push_summary_originated_node( /// The [`build_summary`]/[`build_summary_hybrid`] counterpart of [`build`] /// for a bound [`SummaryNode`] reached while building -/// [`export_post_asap`]'s merged graph: appends into the *same* `nodes: +/// [`export_post_asap`]'s merged DAG: appends into the *same* `nodes: /// Vec` list `build` itself is filling, instead of a separate -/// [`SummaryDagGraph`]. A `KeepPreAsap(inner)` leaf recurses back into +/// [`SummaryDAG`]. A `KeepPreAsap(inner)` leaf recurses back into /// [`build`] on `inner` (the general pre-ASAP entry, `find_winner` included) -/// rather than nesting a `{"pre_asap_subgraph": ...}` blob the way -/// [`build_summary`] does — so the merged graph reads as one seamless graph +/// rather than nesting a `{"pre_asap_sub_dag": ...}` blob the way +/// [`build_summary`] does — so the merged DAG reads as one seamless DAG /// with no dead ends, and so a target reachable underneath a `KeepPreAsap` /// wrapper (a nested aggregate a strategy independently found a /// replacement for, say) still gets spliced in correctly. @@ -965,7 +965,7 @@ fn build_summary_hybrid( .map(|child| build_summary_hybrid(child, nodes, cache, find_winner)) .collect(); let (kind, label, mut detail) = summary_shape(&node.expr); - // The merged graph's `DagNode` has no dedicated guarantee field (it is + // The merged DAG's `DagNode` has no dedicated guarantee field (it is // the pre-ASAP node shape); the guarantee rides in `detail` under the // same key/shape `SummaryDagNode::guarantee` uses, additively. if let Some(guarantee) = &node.guarantee { @@ -1007,7 +1007,7 @@ fn source_label(source: &Source) -> String { /// arm that carries one (`Filter.pred`, `Project.cols`, `Aggregate.having`, …) /// serializes it as opaque `detail` JSON via `Predicate`/`ProjectItem`/ /// `AggIntent`'s own `Serialize` impl, same as before the merge — a scalar -/// subtree was never a separate DAG node, so this doesn't change that. +/// sub-DAG was never a separate DAG node, so this doesn't change that. /// /// `find_winner` is [`export_post_asap`]'s substitution seam, threaded /// through every recursive call (including [`export`]'s own, which always @@ -1434,11 +1434,11 @@ mod tests { #[test] fn leaf_scan_is_a_single_node() { - let graph = export(&scan("metrics", value_col())); - assert_eq!(graph.nodes.len(), 1); - assert_eq!(graph.root, 0); - assert_eq!(graph.nodes[0].kind, "Scan"); - assert!(graph.nodes[0].children.is_empty()); + let dag = export(&scan("metrics", value_col())); + assert_eq!(dag.nodes.len(), 1); + assert_eq!(dag.root, 0); + assert_eq!(dag.nodes[0].kind, "Scan"); + assert!(dag.nodes[0].children.is_empty()); } /// `export` itself never populates higher-layer annotations. Empty @@ -1446,12 +1446,12 @@ mod tests { /// exports retain their existing shape. #[test] fn export_omits_empty_higher_layer_annotations() { - let graph = export(&scan("metrics", value_col())); - assert!(graph.nodes[0].notes.is_empty()); - assert!(graph.nodes[0].decision.is_none()); - assert!(graph.nodes[0].schema.is_some()); - assert!(graph.edge_annotations.is_empty()); - let json = serde_json::to_string(&graph.nodes[0]).unwrap(); + let dag = export(&scan("metrics", value_col())); + assert!(dag.nodes[0].notes.is_empty()); + assert!(dag.nodes[0].decision.is_none()); + assert!(dag.nodes[0].schema.is_some()); + assert!(dag.edge_annotations.is_empty()); + let json = serde_json::to_string(&dag.nodes[0]).unwrap(); assert!( !json.contains("notes"), "empty `notes` must be skipped, not serialized as `[]`: {json}" @@ -1460,10 +1460,10 @@ mod tests { !json.contains("decision"), "empty `decision` must be skipped, not serialized as `null`: {json}" ); - let graph_json = serde_json::to_string(&graph).unwrap(); + let dag_json = serde_json::to_string(&dag).unwrap(); assert!( - !graph_json.contains("edge_annotations"), - "empty `edge_annotations` must be skipped, not serialized as `[]`: {graph_json}" + !dag_json.contains("edge_annotations"), + "empty `edge_annotations` must be skipped, not serialized as `[]`: {dag_json}" ); } @@ -1487,14 +1487,14 @@ mod tests { children: vec![left_branch, right_branch], discriminator_unique_key: None, }; - let graph = export_post_asap(&root, &mut |_| None); + let dag = export_post_asap(&root, &mut |_| None); assert_eq!( - graph.nodes.iter().filter(|n| n.kind == "Scan").count(), + dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1, "the shared Scan must be merged onto one node, not duplicated" ); - assert!(graph.edge_annotations.is_empty()); + assert!(dag.edge_annotations.is_empty()); } /// Regression test: a single parent referencing the same shared child @@ -1511,26 +1511,26 @@ mod tests { left: Rc::clone(&shared_scan), right: Rc::clone(&shared_scan), }; - let graph = export_post_asap(&root, &mut |_| None); + let dag = export_post_asap(&root, &mut |_| None); assert_eq!( - graph.nodes.iter().filter(|n| n.kind == "Scan").count(), + dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1, "the shared Scan must be merged onto one node, not duplicated" ); assert!( - graph.edge_annotations.is_empty(), + dag.edge_annotations.is_empty(), "a single parent referencing the same child twice is one consumer, not a genuine \ multi-consumer share — got: {:?}", - graph.edge_annotations + dag.edge_annotations ); } #[test] fn export_never_produces_edge_annotations_since_it_never_shares_nodes() { // Plain `export` (no `export_post_asap`) never deduplicates by `Rc` - // pointer identity — even a workload-level shared subtree renders as - // two independent tree nodes here, so there is nothing to annotate. + // pointer identity — even a workload-level shared sub-DAG renders as + // two independent DAG nodes here, so there is nothing to annotate. let shared_scan = Rc::new(scan("metrics", value_col())); let root = QueryExpr::Join { kind: crate::pre_asap::query_expr::JoinKind::Inner, @@ -1538,9 +1538,9 @@ mod tests { left: Rc::clone(&shared_scan), right: Rc::clone(&shared_scan), }; - let graph = export(&root); - assert_eq!(graph.nodes.iter().filter(|n| n.kind == "Scan").count(), 2); - assert!(graph.edge_annotations.is_empty()); + let dag = export(&root); + assert_eq!(dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 2); + assert!(dag.edge_annotations.is_empty()); } #[test] @@ -1558,18 +1558,18 @@ mod tests { child: Rc::new(scan("metrics", value_col())), }), }; - let graph = export(&expr); - assert_eq!(graph.nodes.len(), 3, "Filter -> Aggregate -> Scan"); + let dag = export(&expr); + assert_eq!(dag.nodes.len(), 3, "Filter -> Aggregate -> Scan"); - let filter = &graph.nodes[graph.root as usize]; + let filter = &dag.nodes[dag.root as usize]; assert_eq!(filter.kind, "Filter"); assert_eq!(filter.children.len(), 1); - let agg = &graph.nodes[filter.children[0] as usize]; + let agg = &dag.nodes[filter.children[0] as usize]; assert_eq!(agg.kind, "Aggregate"); assert_eq!(agg.children.len(), 1); - let leaf = &graph.nodes[agg.children[0] as usize]; + let leaf = &dag.nodes[agg.children[0] as usize]; assert_eq!(leaf.kind, "Scan"); assert!(leaf.children.is_empty()); } @@ -1581,37 +1581,37 @@ mod tests { scan("b", value_col()), scan("c", value_col()), ]); - let graph = export(&expr); - assert_eq!(graph.nodes.len(), 4, "3 branches + the Concat node"); - let merge = &graph.nodes[graph.root as usize]; + let dag = export(&expr); + assert_eq!(dag.nodes.len(), 4, "3 branches + the Concat node"); + let merge = &dag.nodes[dag.root as usize]; assert_eq!(merge.kind, "Concat"); assert_eq!(merge.children.len(), 3); } #[test] - fn identical_subtrees_hash_equal_and_differing_ones_dont() { + fn identical_sub_dags_hash_equal_and_differing_ones_dont() { let left = scan("metrics", value_col()); let right = scan("metrics", value_col()); let different = scan("other_table", value_col()); - let left_graph = export(&left); - let right_graph = export(&right); - let different_graph = export(&different); + let left_dag = export(&left); + let right_dag = export(&right); + let different_dag = export(&different); assert_eq!( - left_graph.nodes[left_graph.root as usize].hash, - right_graph.nodes[right_graph.root as usize].hash, + left_dag.nodes[left_dag.root as usize].hash, + right_dag.nodes[right_dag.root as usize].hash, "structurally identical Scans must hash equal" ); assert_ne!( - left_graph.nodes[left_graph.root as usize].hash, - different_graph.nodes[different_graph.root as usize].hash, + left_dag.nodes[left_dag.root as usize].hash, + different_dag.nodes[different_dag.root as usize].hash, "a different table_ref must not collide" ); } #[test] - fn shared_subtree_hash_matches_across_a_larger_tree() { + fn shared_sub_dag_hash_matches_across_a_larger_dag() { // Two roots that each wrap the *same* Scan shape in a different outer // node — the exported hash should still flag the shared Scan even // though it's embedded at different depths / under different parents. @@ -1651,19 +1651,19 @@ mod tests { // this exact node, because it's the same function call, not a // parallel reimplementation that happens to agree. let leaf = scan("metrics", value_col()); - let graph = export(&leaf); + let dag = export(&leaf); assert_eq!( - graph.nodes[graph.root as usize].hash, + dag.nodes[dag.root as usize].hash, Some(structural_hash(&leaf, &mut HashCache::new())), "dag_export's root hash must equal cse::structural_hash(&leaf, &mut HashCache::new()) directly" ); } #[test] - fn every_node_hash_matches_cse_structural_hash_on_its_own_subtree() { - // A multi-level tree: check the parity holds at every depth, not + fn every_node_hash_matches_cse_structural_hash_on_its_own_sub_dag() { + // A multi-level DAG: check the parity holds at every depth, not // just the root — each `DagNode::hash` must equal - // `structural_hash` applied to the actual `QueryExpr` subtree that + // `structural_hash` applied to the actual `QueryExpr` sub-DAG that // node represents. let agg = QueryExpr::Aggregate { reduction: Reduction::Reduce(GroupKeys::none()), @@ -1680,20 +1680,20 @@ mod tests { child: Rc::new(agg.clone()), }; - let graph = export(&root); + let dag = export(&root); assert_eq!( - graph.nodes[graph.root as usize].hash, + dag.nodes[dag.root as usize].hash, Some(structural_hash(&root, &mut HashCache::new())), "Filter root hash must match cse::structural_hash(&root, &mut HashCache::new())" ); - let filter = &graph.nodes[graph.root as usize]; - let agg_node = &graph.nodes[filter.children[0] as usize]; + let filter = &dag.nodes[dag.root as usize]; + let agg_node = &dag.nodes[filter.children[0] as usize]; assert_eq!( agg_node.hash, Some(structural_hash(&agg, &mut HashCache::new())), "the exported Aggregate node's hash must match cse::structural_hash \ - on the Aggregate subtree it represents, not just the root" + on the Aggregate sub-DAG it represents, not just the root" ); } @@ -1772,9 +1772,9 @@ mod tests { }, guarantee: Some(guarantee), }; - let graph = export_summary(&root); - let json = serde_json::to_value(&graph).unwrap(); - let root_json = &json["nodes"][graph.root as usize]; + let dag = export_summary(&root); + let json = serde_json::to_value(&dag).unwrap(); + let root_json = &json["nodes"][dag.root as usize]; assert_eq!(root_json["guarantee"]["metric"], "rank"); assert_eq!(root_json["guarantee"]["bound"]["op"], "sum"); assert_eq!( @@ -1792,12 +1792,12 @@ mod tests { assert!(state.get("guarantee").is_none()); assert_eq!(json["nodes"][0]["guarantee"]["bound"]["op"], "zero"); - let named = NamedGraph { + let named = NamedDAG { name: "q".into(), source: None, - graph: export(&leaf), + dag: export(&leaf), replacements: vec![], - post_graph: None, + post_dag: None, workload_cost: None, rejections: vec![TargetRejection { target_pre_id: 0, @@ -1817,8 +1817,8 @@ mod tests { "unsupported_composition" ); assert_eq!(json["rejections"][0]["error"]["input_metrics"][0], "rank"); - // Additive: a graph with no rejections omits the key entirely. - let plain = NamedGraph { + // Additive: a DAG with no rejections omits the key entirely. + let plain = NamedDAG { rejections: vec![], ..named }; diff --git a/crates/types/src/post_asap/cse.rs b/crates/types/src/post_asap/cse.rs index 1872243f8..745758674 100644 --- a/crates/types/src/post_asap/cse.rs +++ b/crates/types/src/post_asap/cse.rs @@ -171,7 +171,7 @@ fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { expression_equal && left.schema == right.schema && same_value(&left.guarantee, &right.guarantee) } -/// Intern equal selected subtrees across roots while preserving every root ID. +/// Intern equal selected sub-DAGs across roots while preserving every root ID. /// /// Only structural equality is used: no grouping, parameter, accuracy or source /// coercions are performed. All roots must belong to the same data snapshot or @@ -184,7 +184,7 @@ fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { /// distinct count, entropy and L2. Candidate generation sizes a variant for /// the strictest sibling consumer so differing accuracy targets can reach /// identical states here. -pub fn share_common_summary_subtrees( +pub fn share_common_summary_sub_dags( roots: Vec<(Id, Rc)>, ) -> Vec<(Id, Rc)> { fn visit( @@ -269,7 +269,7 @@ mod tests { // Equal separately constructed roots preserve both IDs but share identity. #[test] fn shares_equal_roots_and_preserves_ids() { - let roots = share_common_summary_subtrees(vec![("a", leaf(1.0)), ("b", leaf(1.0))]); + let roots = share_common_summary_sub_dags(vec![("a", leaf(1.0)), ("b", leaf(1.0))]); assert_eq!(roots[0].0, "a"); assert_eq!(roots[1].0, "b"); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); @@ -289,7 +289,7 @@ mod tests { }, guarantee: None, }); - let roots = share_common_summary_subtrees(vec![(0, leaf(1.0)), (1, merge)]); + let roots = share_common_summary_sub_dags(vec![(0, leaf(1.0)), (1, merge)]); let SummaryExpr::SummaryMerge { children, .. } = &roots[1].1.expr else { panic!() }; @@ -302,7 +302,7 @@ mod tests { fn distinct_guarantees_and_values_are_not_shared() { let mut unknown = leaf(1.0).as_ref().clone(); unknown.guarantee = None; - let roots = share_common_summary_subtrees(vec![ + let roots = share_common_summary_sub_dags(vec![ (0, leaf(1.0)), (1, Rc::new(unknown)), (2, leaf(2.0)), @@ -317,7 +317,7 @@ mod tests { fn signed_zero_is_not_coalesced() { for values in [[0.0, -0.0], [-0.0, 0.0]] { let roots = - share_common_summary_subtrees(vec![(0, leaf(values[0])), (1, leaf(values[1]))]); + share_common_summary_sub_dags(vec![(0, leaf(values[0])), (1, leaf(values[1]))]); assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); for ((_, root), expected) in roots.iter().zip(values) { let SummaryExpr::KeepPreAsap(expr) = &root.expr else { @@ -347,10 +347,10 @@ mod tests { (f64::INFINITY, f64::NEG_INFINITY), (f64::NAN, f64::NAN), ] { - let roots = share_common_summary_subtrees(vec![(0, wrapped(a)), (1, wrapped(b))]); + let roots = share_common_summary_sub_dags(vec![(0, wrapped(a)), (1, wrapped(b))]); assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); } - let roots = share_common_summary_subtrees(vec![ + let roots = share_common_summary_sub_dags(vec![ (0, wrapped(f64::INFINITY)), (1, wrapped(f64::INFINITY)), ]); @@ -399,7 +399,7 @@ mod tests { guarantee: None, }) } - let roots = share_common_summary_subtrees(vec![ + let roots = share_common_summary_sub_dags(vec![ ("p95", readout(0.95, 0.01)), ("p99", readout(0.99, 0.01)), ("strict", readout(0.95, 0.001)), @@ -413,7 +413,7 @@ mod tests { assert!(!Rc::ptr_eq(&producer(&roots[0].1), &producer(&roots[2].1))); } - // Fifty unique input nodes must not require walking an expanded 2^24 tree. + // Fifty unique input nodes must not require walking an expanded 2^24 DAG. // The timeout is a coarse runaway guard, not a performance SLA. #[test] fn shared_diamond_does_not_expand_during_comparison() { @@ -445,7 +445,7 @@ mod tests { } current } - let roots = share_common_summary_subtrees(vec![(0, diamond()), (1, diamond())]); + let roots = share_common_summary_sub_dags(vec![(0, diamond()), (1, diamond())]); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); done.send(()).unwrap(); }); diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 3c1c3e0c1..97b9eeb42 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -40,7 +40,7 @@ //! not do is stay ambiguous inside one mixed plan: the same `Rc` //! reached once as update input and once as query-time fallback is //! [`ExecutionDataStateError::AmbiguousKeepPreAsap`], because no single execution of that -//! subtree can serve both roles. +//! sub-DAG can serve both roles. use std::collections::HashMap; use std::rc::Rc; @@ -187,7 +187,7 @@ pub enum ExecutionDataStateError { /// One shared `KeepPreAsap` node reached both as update-path raw input /// and as a query-time fallback — see the module docs. #[error( - "KeepPreAsap subtree is data_state-ambiguous: reached as {first} and as {second} in the same \ + "KeepPreAsap sub-DAG is data_state-ambiguous: reached as {first} and as {second} in the same \ plan" )] AmbiguousKeepPreAsap { @@ -635,7 +635,7 @@ fn child_domain( Ok(avail) } None => { - // A raw pre-ASAP subtree executes at whichever data_state its consumer + // A raw pre-ASAP sub-DAG executes at whichever data_state its consumer // needs: update-path input for maintenance-time operation edges, // query-time fallback for a read-time edge. State-only edges // can't consume plain rows at all. @@ -1254,7 +1254,7 @@ mod tests { #[test] fn a_shared_keep_pre_asap_reached_in_two_domains_is_ambiguous() { - // One raw subtree used both as update input (under a SummaryAgg) and + // One raw sub-DAG used both as update input (under a SummaryAgg) and // as a query-time fallback (under an ExactRead) — no single // execution can serve both, so the plan is rejected. let shared = keep(); diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs index 6c48114a2..8373f1843 100644 --- a/crates/types/src/post_asap/expr.rs +++ b/crates/types/src/post_asap/expr.rs @@ -104,7 +104,7 @@ pub struct SummaryNode { /// The machine-readable accuracy guarantee of the *value* this node /// produces (issue #172) — `Some` on every finalized, caller-visible /// value: a `SummaryEstimate` readout, an `ExactAggregate`-family - /// `SummaryAgg` (its state *is* the value), or a `KeepPreAsap` subtree + /// `SummaryAgg` (its state *is* the value), or a `KeepPreAsap` sub-DAG /// (executed exactly). `None` on raw summary state — a sketch-family /// `SummaryAgg`, `SummaryMerge`, `SummarySubtract`, `SummaryDelete`, /// `SummaryJoin` — whose guarantee only exists once something reads it @@ -121,13 +121,13 @@ pub struct SummaryNode { /// rules selectively replace logical aggregates and joins in the pre-ASAP /// `QueryExpr` with summary-bound counterparts. Final selection can retain /// supported read-time value operations around independently planned children; -/// other unsupported subtrees pass through as `KeepPreAsap(Rc)`. +/// other unsupported sub-DAGs pass through as `KeepPreAsap(Rc)`. /// /// Traversing from the root node yields a DAG; shared sub-expressions appear /// as multiple `Rc` references to the same `SummaryNode`. #[derive(Debug, Clone, PartialEq)] pub enum SummaryExpr { - /// A pre-ASAP subtree kept as-is because it has no selected implementation + /// A pre-ASAP sub-DAG kept as-is because it has no selected implementation /// or supported residual decomposition. Output schema is the inner node's /// schema, lifted to `SummarySchema` with all fields as /// `SummaryFamilyType::Plain`. diff --git a/crates/types/src/post_asap/guarantee.rs b/crates/types/src/post_asap/guarantee.rs index 88e070ce9..cfc6f3f01 100644 --- a/crates/types/src/post_asap/guarantee.rs +++ b/crates/types/src/post_asap/guarantee.rs @@ -17,7 +17,7 @@ //! //! [`ResultGuarantee`] is attached to a finalized, caller-visible value — //! [`super::SummaryNode::guarantee`] on a `SummaryEstimate` readout, an -//! exact accumulator, or a kept pre-ASAP subtree — never to raw summary +//! exact accumulator, or a kept pre-ASAP sub-DAG — never to raw summary //! state (a `SummaryAgg` sketch node carries `None`; its readout carries the //! guarantee). Its statement is: //! @@ -33,7 +33,7 @@ //! //! ## Why expressions, not numbers //! -//! [`BoundExpr`]/[`ProbabilityExpr`] are tiny serializable expression trees +//! [`BoundExpr`]/[`ProbabilityExpr`] are tiny serializable expression DAGs //! rather than bare `f64`s so a planning-time guarantee can reference a //! statistic it does not have (a group count, a stream's L1 norm) and stay //! honestly *unknown* until something instantiates it — a deployment's own diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index 6bf317420..e7041bea7 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -40,7 +40,7 @@ pub mod summary_maintenance; pub mod summary_maintenance_lifecycle; pub mod summary_window; -pub use cse::share_common_summary_subtrees; +pub use cse::share_common_summary_sub_dags; pub use execution_data_state::{ assigned_child_data_state, exact_operation_output_schema, produced_data_state, validate_execution_data_states, validate_execution_data_states_at, DataPrimitive, diff --git a/crates/types/src/post_asap/post_asap_dag.rs b/crates/types/src/post_asap/post_asap_dag.rs index adaabb1b3..4f4ba2e73 100644 --- a/crates/types/src/post_asap/post_asap_dag.rs +++ b/crates/types/src/post_asap/post_asap_dag.rs @@ -533,7 +533,7 @@ pub fn compile_post_asap_dag_with_node_ids( consumer: id, role, intermediate_schema: child.schema.clone(), - // The whole-graph validator owns contextual state assignment, + // The whole-DAG validator owns contextual state assignment, // especially for shared KeepPreAsap leaves. Export that // authoritative result instead of independently deriving the // edge state a second time. diff --git a/crates/types/src/post_asap/sketch.rs b/crates/types/src/post_asap/sketch.rs index 391d12f0f..416341e73 100644 --- a/crates/types/src/post_asap/sketch.rs +++ b/crates/types/src/post_asap/sketch.rs @@ -606,7 +606,7 @@ pub enum SketchQuery { /// `count(cms_metric{item="checkout"})` — `key` is `item`, `value` is /// `"checkout"`). `value` is carried here rather than resolved by the /// `SummaryExecutor` from a `Filter` predicate because `readout`'s - /// trait signature has no tree access — see `CostModel::readout_extension`. + /// trait signature has no DAG access — see `CostModel::readout_extension`. PointCount { key: ColumnRef, value: Option, diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/pre_asap/agg_intent.rs index 2203f4ec7..5e4079b2e 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/pre_asap/agg_intent.rs @@ -380,7 +380,7 @@ pub enum MathFunc { // `requires` / `is_per_series` / `output_column` never read `col`'s value — // only its presence via a `{ .. }` pattern — so, unlike // `QueryExpr::output_schema` (which genuinely cannot compile for an -// unresolved tree — see its own doc), nothing stops these from being generic +// unresolved DAG — see its own doc), nothing stops these from being generic // over every `C`. And a front end constructing `AggIntent` // directly (issue #179) does need `is_per_series` pre-binding — it decides // the `PerEntity`/`Reduce` reduction shape right at construction time (see @@ -930,7 +930,7 @@ mod tests { } } - /// `col` is `#[serde(default)]`, so a tree serialized before issue #115 — + /// `col` is `#[serde(default)]`, so a DAG serialized before issue #115 — /// with no `col` key — still deserializes, as the sample-value convention `None`. #[test] fn agg_intent_serde_reads_pre_115_payloads() { diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs index 3a66c5c4d..4f15dcebb 100644 --- a/crates/types/src/pre_asap/canonicalize.rs +++ b/crates/types/src/pre_asap/canonicalize.rs @@ -1,7 +1,7 @@ //! Shared post-lowering canonicalization of the resolved [`QueryExpr`]. //! //! Both language front ends funnel through [`resolve_root`](super::resolve::resolve_root), -//! which runs this pass over the resolved tree. Its job is to erase +//! which runs this pass over the resolved DAG. Its job is to erase //! *structural* differences between semantically identical queries so a //! post-ASAP binding rule matching on the intent algebra sees one canonical //! spelling regardless of source language (issue #34). @@ -29,7 +29,7 @@ use super::expr_ir::{CompareOpKind, ScalarValue}; use super::query_expr::{Predicate, QueryExpr, Reduction, SortKey, WindowFuncKind}; use crate::types::AccuracyTarget; -/// Rewrite `expr` into its canonical form (bottom-up). Idempotent: a tree that +/// Rewrite `expr` into its canonical form (bottom-up). Idempotent: a DAG that /// is already canonical is returned unchanged. pub fn canonicalize(mut expr: QueryExpr) -> QueryExpr { canon(&mut expr); @@ -48,7 +48,7 @@ fn canon(expr: &mut QueryExpr) { // pointing at the wrong column, or out of bounds, of the // post-canonicalize schema. Snapshot the schema the discriminator key // was actually resolved against, right here, before recursing into the - // children — this is the exact tree state `resolve.rs` saw. + // children — this is the exact DAG state `resolve.rs` saw. let discriminator_branch_schema_before = match expr { QueryExpr::Concat { children, @@ -96,7 +96,7 @@ fn canon(expr: &mut QueryExpr) { /// A `&mut QueryExpr` out of a child `Rc` — clone-on-write via /// [`Rc::make_mut`]: free (no clone) while `r` is uniquely owned, which is -/// the overwhelmingly common case (a tree `canonicalize` was just handed by +/// the overwhelmingly common case (a DAG `canonicalize` was just handed by /// value); falls back to cloning just *this* node (its own fields — the /// grandchildren stay shared `Rc`s, not deep-copied) only when some other /// owner still holds the same `Rc`, e.g. a caller that kept its own clone @@ -106,7 +106,7 @@ fn canon(expr: &mut QueryExpr) { /// panic on exactly that case; `make_mut` degrades to a shallow copy instead /// of requiring sole ownership as a precondition. Once a workload-level CSE /// pass runs (issue #212, #222) and canonicalize sees an already-shared -/// subtree from a *different* query, this is also the mechanism that keeps +/// sub-DAG from a *different* query, this is also the mechanism that keeps /// canonicalizing one query from silently corrupting another's view of the /// same shared node. fn rc_mut(r: &mut Rc) -> &mut QueryExpr { @@ -117,7 +117,7 @@ fn rc_mut(r: &mut Rc) -> &mut QueryExpr { /// node — `canon`'s own top-down/bottom-up walk only ever visits the /// relational skeleton, never descending into a scalar position (`Filter.pred`, /// `ProjectItem.expr`, …): none of the three rewrite rules rewrite anything -/// inside a scalar subtree, so there's nothing to gain by recursing into one, +/// inside a scalar sub-DAG, so there's nothing to gain by recursing into one, /// and every scalar variant (issue #205) hits the catch-all below. fn children_mut(expr: &mut QueryExpr) -> Vec<&mut QueryExpr> { use QueryExpr::*; diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index 6f974922e..3176730a9 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -1,7 +1,7 @@ //! Schema-driven column resolution. //! //! Front ends (issue #179) emit `ColumnRef` (name-based, optionally -//! table-qualified); the canonical tree uses positional [`ColumnId`] resolved +//! table-qualified); the canonical DAG uses positional [`ColumnId`] resolved //! against a per-node [`Schema`]. These helpers bridge the two — the //! [`SchemaResolver`](super::schema_resolver) builds the schema, and [`resolve_column_refs`] //! turns name-based refs (group keys, dedup columns) into positional ids, diff --git a/crates/types/src/pre_asap/cse.rs b/crates/types/src/pre_asap/cse.rs index 1a2758737..45842bf1d 100644 --- a/crates/types/src/pre_asap/cse.rs +++ b/crates/types/src/pre_asap/cse.rs @@ -1,14 +1,14 @@ //! Pre-ASAP structural common-subexpression elimination: bottom-up -//! hash-consing over an already-`resolve_root`'d [`QueryExpr`] tree (issue +//! hash-consing over an already-`resolve_root`'d [`QueryExpr`] DAG (issue //! #212, #222, #223). //! -//! CSE only runs on an already-bound, already-canonicalized tree — +//! CSE only runs on an already-bound, already-canonicalized DAG — //! structural matching is meaningless before canonicalization has converged //! semantically-equivalent queries onto one shape (`docs/develop_docs/pre-asap-ir.md` //! design principle 3; `median(latency)` and `approx_percentile_cont(latency, //! 0.5)` already lower to an identical `AggIntent::Quantile` today, per //! `sql_lowering.rs`'s `median_is_the_same_intent_as_an_explicit_half_percentile` -//! test). [`share_common_subtrees`] is the single entry point, run once per +//! test). [`share_common_sub_dags`] is the single entry point, run once per //! workload batch (or once per query — see "Single-query CSE" below) *after* //! `resolve_root`, *before* the pre-ASAP → post-ASAP replacement/search pass //! (`asap_aware_mapping::replacement`). @@ -19,7 +19,7 @@ //! candidacy for sharing naturally incorporates whether its own children were //! themselves shared — two parents whose children were independently //! deduplicated down to the same `Rc`s are structurally identical iff their -//! own fields also match, without re-walking the subtrees. +//! own fields also match, without re-walking the sub-DAGs. //! //! Only the **relational skeleton** participates — the same set of "operator" //! children [`canonicalize`](super::canonicalize)'s `children_mut` walks @@ -30,13 +30,13 @@ //! `QueryExpr`'s derived `PartialEq` along with the rest of that node's //! fields, rather than separately hash-consed — the same scope //! `canonicalize.rs` settled on ("none of the rewrite rules touch a scalar -//! subtree, so there's nothing to gain by recursing into one"). Widening this +//! sub-DAG, so there's nothing to gain by recursing into one"). Widening this //! to scalar positions is future work, not attempted here. //! //! ## Correctness: hash is a filter, `PartialEq` is the decision //! //! This is the one non-negotiable rule. A **false positive** here — two -//! subtrees wrongly judged shareable — is a wrong query answer, not a missed +//! sub-DAGs wrongly judged shareable — is a wrong query answer, not a missed //! optimization: two different queries would read each other's data. //! [`structural_hash`] (`DefaultHasher`/SipHash over a canonical //! serialization, no collision-freedom guarantee) may only narrow the @@ -75,23 +75,23 @@ //! A repeated sub-expression within *one* query (e.g. the same grouped //! `Aggregate` referenced twice on two `BinaryOp` branches) is deduplicated //! by the exact same bottom-up interning — a workload of size one still -//! interns bottom-up within that one tree. No separate mechanism is needed; -//! see the `single_query_shares_its_own_repeated_subtree` test below. +//! interns bottom-up within that one DAG. No separate mechanism is needed; +//! see the `single_query_shares_its_own_repeated_sub_dag` test below. //! //! ## Landing plan (issue #223) //! //! This module is stage 1 of a 4-stage plan. Stage 2 //! (`asap_aware_mapping::replacement::search_workload_with`, which runs -//! [`share_common_subtrees`] itself before searching) is a real caller, +//! [`share_common_sub_dags`] itself before searching) is a real caller, //! wired at the same time so this never becomes unwired dead code again //! (the original `asap-plan::cse::dedupe_subtrees` was deleted in #192 for //! exactly that). Stage 3 — [`dag_export`](crate::dag_export) computing its //! per-node `hash` by calling this module's [`structural_hash`] directly, //! instead of a parallel reimplementation — is also done, so -//! `tools/dag-viewer`'s "shared subtree" highlighting now flags exactly the +//! `tools/dag-viewer`'s "shared sub-DAG" highlighting now flags exactly the //! candidate pairs this module's own `InternTable` would bucket together //! (still only a hash match, not a guarantee of -//! `share_common_subtrees`-actual sharing — see `dag_export`'s module doc). +//! `share_common_sub_dags`-actual sharing — see `dag_export`'s module doc). //! Stage 4 (issue #237) is implemented in //! `asap_aware_mapping::cost_model::CostModel::cse_share_decision`, called //! from `asap_aware_mapping::replacement::CandidateLogicalASAPDAGs::cost_sorted` (via that @@ -187,27 +187,27 @@ pub type HashCache = HashMap<*const QueryExpr, u64>; /// up in `cache` if already computed there (memoized by `Rc` pointer /// identity) rather than recursed into again. /// -/// This is the DAG-aware fix a naive "just serialize the whole subtree" -/// hash would get wrong: after [`share_common_subtrees`] (or even before +/// This is the DAG-aware fix a naive "just serialize the whole sub-DAG" +/// hash would get wrong: after [`share_common_sub_dags`] (or even before /// it — a front end can emit internal `Rc` sharing directly, e.g. a -/// repeated subexpression within one query), `node` is generally a DAG, -/// not a tree. A full-subtree serialization re-serializes — re-walks — +/// repeated subexpression within one query), `node` generally has internal +/// sharing. A full-sub-DAG serialization re-serializes — re-walks — /// any descendant `node` already shares internally once per parent that /// references it; called once per node in a bottom-up pass (as /// [`InternTable::intern`] and [`dag_export`](crate::dag_export) both do), -/// that costs `O(subtree size)` *per node* instead of `O(1)` amortized — +/// that costs `O(sub-DAG size)` *per node* instead of `O(1)` amortized — /// quadratic-or-worse for a deep chain, compounding further with any real /// internal sharing. Memoizing each child's hash by pointer identity in /// `cache` (persisted across the whole pass by the caller, not reset per /// node) makes each node's own contribution `O(1)` beyond its children's /// already-known hashes, giving `O(N)` total for `N` nodes — matching -/// [`dag_node_count`]'s own DAG-vs-tree fix (issue #212/#223/#237's stage +/// [`dag_node_count`]'s own shared-node counting fix (issue #212/#223/#237's stage /// 4) in spirit, applied to hashing instead of counting. /// /// `pub` (not private) so [`dag_export`](crate::dag_export) can call /// this exact function for its exported nodes' `hash` field instead of /// maintaining its own parallel reimplementation — issue #223 stage 3. That -/// makes `tools/dag-viewer`'s "shared subtree" highlighting reflect this +/// makes `tools/dag-viewer`'s "shared sub-DAG" highlighting reflect this /// module's real hashing, not a lookalike computed a different way; see the /// module doc's "Landing plan" section. A NaN/infinite `f64` makes JSON /// serialization fail; falling back to a fixed hash just puts every such @@ -236,9 +236,9 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// Hash `own_fields` (this node's own tag and non-child scalar /// fields — anything JSON-serializable and small, i.e. never a - /// `QueryExpr` subtree) via the same canonical-JSON-string trick the - /// whole-subtree version used, just applied to `O(1)` fields instead - /// of `O(subtree size)`. + /// `QueryExpr` sub-DAG) via the same canonical-JSON-string trick the + /// whole-sub-DAG version used, just applied to `O(1)` fields instead + /// of `O(sub-DAG size)`. fn hash_own_fields(hasher: &mut impl Hasher, own_fields: &impl serde::Serialize) { serde_json::to_string(own_fields) .unwrap_or_default() @@ -408,7 +408,7 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { // (issue #205) are all leaves for this traversal's purposes — none // has an operator child to look up in `cache` — so hashing the // whole node via `serde_json` in one shot is already `O(node - // size)`, not `O(subtree size)`: exactly the same cost the + // size)`, not `O(sub-DAG size)`: exactly the same cost the // per-variant `hash_own_fields` calls above pay, just without // needing to spell out each field individually. Matches // `rebuild_children`'s and `dag_node_count`'s identical scope @@ -435,13 +435,13 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// Count of *unique* nodes reachable from `root`, deduplicated by `Rc` /// pointer identity (`Rc::as_ptr`) — the real size of the DAG rooted at -/// `root`, not a tree-walk count. +/// `root`, not a per-path walk count. /// -/// After [`share_common_subtrees`] runs (or even before it, for a tree a +/// After [`share_common_sub_dags`] runs (or even before it, for a DAG a /// front end already built with internal `Rc` sharing — e.g. re-running /// CSE, or a single-query repeated subexpression), `root` is generally a -/// **DAG**, not a tree — that is this whole module's premise. Anything that -/// walks `root` as if every reference were a fresh subtree (a naive +/// **DAG** with internal sharing — that is this whole module's premise. Anything that +/// walks `root` as if every reference were a fresh sub-DAG (a naive /// recursive walk with no identity tracking, or a naive full /// `serde_json` serialization — `Rc`'s `Serialize` impl serializes the /// pointee's *value* at every occurrence, it does not dedupe by identity) @@ -454,9 +454,9 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// `pub` so cost-aware callers outside this crate (e.g. /// `asap_aware_mapping::CostModel::cse_recompute_cost`'s default) have a /// DAG-correct structural-size proxy available, instead of reaching for -/// something tree-shaped like a raw serialization length. +/// something per-path like a raw serialization length. /// -/// Same operator-child traversal scope as [`share_common_subtrees`] itself +/// Same operator-child traversal scope as [`share_common_sub_dags`] itself /// (see the module doc's "Algorithm" section, and this module's private /// `rebuild_children`) — a scalar subexpression embedded in a wrapper /// position (`Predicate`, `ProjectItem.expr`, `Aggregate.having`, …) is not @@ -542,11 +542,11 @@ fn count_unique(node: &QueryExpr, seen: &mut std::collections::HashSet<*const Qu /// Recurse into `child`, then intern the result. `Rc::try_unwrap` recovers /// the owned node without cloning in the overwhelmingly common case — a -/// tree freshly built by a front end / `resolve_root`, not yet shared by any +/// DAG freshly built by a front end / `resolve_root`, not yet shared by any /// prior CSE pass, where every `Rc` is uniquely owned. Falls back to cloning /// this node's own fields (its children stay `Rc`s, not deep-copied) only -/// when `child` is already shared — e.g. re-running CSE over a tree that -/// went through a previous `share_common_subtrees` pass; a structural +/// when `child` is already shared — e.g. re-running CSE over a DAG that +/// went through a previous `share_common_sub_dags` pass; a structural /// duplicate collapses right back onto `child` itself via `PartialEq`, an /// already-optimal no-op. fn intern_child(table: &mut InternTable, child: Rc) -> Rc { @@ -750,17 +750,17 @@ fn rebuild_children(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { } } -/// Share structurally-identical, sharing-legal subtrees across a workload's +/// Share structurally-identical, sharing-legal sub-DAGs across a workload's /// query roots (or within one query, for `roots.len() == 1` — see the /// module doc's "Single-query CSE" section). Every root's *value* is /// unchanged (`PartialEq`-equal to its input) — only its internal `Rc` -/// structure may now alias another root's, or another part of its own tree. +/// structure may now alias another root's, or another part of its own DAG. /// /// `roots` must already be bound + canonicalized (post-`resolve_root`). /// `Id` is caller-chosen — a `QueryWorkload` entry's own key, an index, a /// query name, whatever identifies one root through the pipeline; this /// module has no opinion on its shape. -pub fn share_common_subtrees(roots: Vec<(Id, QueryExpr)>) -> Vec<(Id, Rc)> { +pub fn share_common_sub_dags(roots: Vec<(Id, QueryExpr)>) -> Vec<(Id, Rc)> { let mut table = InternTable::new(); roots .into_iter() @@ -816,7 +816,7 @@ mod tests { // blocking the merge — only the differing `col` is. let a = quantile_agg(vec![1], Some(2), 0.5); let b = quantile_agg(vec![1], Some(3), 0.5); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -841,7 +841,7 @@ mod tests { op: CompareOpKind::Gt, right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), })))]; - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -856,12 +856,12 @@ mod tests { // hoistable even though `a` and `b` are structurally identical. let a = quantile_agg(vec![], Some(2), 0.9); let b = quantile_agg(vec![], Some(2), 0.9); - assert_eq!(a, b, "fixture sanity: the two trees are structurally equal"); + assert_eq!(a, b, "fixture sanity: the two DAGs are structurally equal"); assert!( !a.output_schema().unwrap().has_unique_key(), "fixture sanity: an ungrouped aggregate has no provable unique key" ); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -878,11 +878,11 @@ mod tests { // 0.5, .. }` today (see `sql_lowering.rs`'s // `median_is_the_same_intent_as_an_explicit_half_percentile`) — here // built directly (grouped, so a unique key is provable) as two - // independently-constructed but structurally identical trees, the + // independently-constructed but structurally identical DAGs, the // way two different call sites in a workload would produce them. let median = quantile_agg(vec![1], Some(2), 0.5); let approx_percentile_cont_half = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_subtrees(vec![ + let shared = share_common_sub_dags(vec![ ("median", median), ("percentile", approx_percentile_cont_half), ]); @@ -896,13 +896,13 @@ mod tests { } #[test] - fn single_query_shares_its_own_repeated_subtree() { + fn single_query_shares_its_own_repeated_sub_dag() { // One query root referencing the same grouped Aggregate on both // BinaryOp branches — built as two separately-allocated but - // structurally identical subtrees (`.clone()` into two distinct + // structurally identical sub-DAGs (`.clone()` into two distinct // `Rc::new` calls), the shape a front end emitting a repeated // sub-expression would actually produce (no sharing yet). A - // workload of size 1 still interns bottom-up within this one tree — + // workload of size 1 still interns bottom-up within this one DAG — // no separate single-query mechanism needed. let agg = quantile_agg(vec![1], Some(2), 0.5); let root = QueryExpr::BinaryOp { @@ -911,7 +911,7 @@ mod tests { rhs: Rc::new(agg), vector_match: None, }; - let shared = share_common_subtrees(vec![("q", root)]); + let shared = share_common_sub_dags(vec![("q", root)]); let [(_, root)] = shared.as_slice() else { panic!("expected 1 root"); }; @@ -945,12 +945,12 @@ mod tests { } #[test] - fn structural_hash_of_an_internally_shared_tree_matches_the_unshared_equivalent() { + fn structural_hash_of_an_internally_shared_dag_matches_the_unshared_equivalent() { // The same BinaryOp-with-shared-branches shape as - // `dag_node_count_deduplicates_an_internally_shared_subtree` below: + // `dag_node_count_deduplicates_an_internally_shared_sub_dag` below: // hashing it (however the memoization internally short-circuits the // second branch) must produce the exact same value as hashing a - // structurally-identical tree built with *no* sharing at all — the + // structurally-identical DAG built with *no* sharing at all — the // whole point of memoization is not changing the answer, only the // work needed to reach it. let agg = quantile_agg(vec![1], Some(2), 0.5); @@ -1012,12 +1012,12 @@ mod tests { } #[test] - fn dag_node_count_deduplicates_an_internally_shared_subtree() { - // Same shape as `single_query_shares_its_own_repeated_subtree`: a + fn dag_node_count_deduplicates_an_internally_shared_sub_dag() { + // Same shape as `single_query_shares_its_own_repeated_sub_dag`: a // BinaryOp whose two branches are the *same* Rc after - // `share_common_subtrees` (2 nodes: Scan + Aggregate) — the root + // `share_common_sub_dags` (2 nodes: Scan + Aggregate) — the root // itself makes 3 unique nodes total (BinaryOp, Aggregate, Scan), - // not 5 (which a tree-walk / naive serialization, counting the + // not 5 (which a per-path walk / naive serialization, counting the // shared branch's 2 nodes twice, would report). let agg = quantile_agg(vec![1], Some(2), 0.5); let root = QueryExpr::BinaryOp { @@ -1026,7 +1026,7 @@ mod tests { rhs: Rc::new(agg), vector_match: None, }; - let shared = share_common_subtrees(vec![("q", root)]); + let shared = share_common_sub_dags(vec![("q", root)]); let [(_, root)] = shared.as_slice() else { panic!("expected 1 root"); }; @@ -1042,18 +1042,18 @@ mod tests { #[test] fn dag_node_count_deduplicates_across_two_workload_roots() { // Two workload roots sharing one Aggregate after - // `share_common_subtrees` (the `duplicate_workload_queries_...` + // `share_common_sub_dags` (the `duplicate_workload_queries_...` // shape from `crates/integration-tests/tests/cse.rs`, built // directly here): each root's own `dag_node_count` must report the - // shared subtree's real size once, not double-count anything — + // shared sub-DAG's real size once, not double-count anything — // there's nothing *to* double-count from a single root's own count // in this case (no root references the shared node twice), so this // pins the simpler, more common case that a per-candidate cost - // proxy (`CseCandidate::subtree` in `asap-aware-mapping`) actually + // proxy (`CseCandidate::sub-DAG` in `asap-aware-mapping`) actually // exercises: counting one occurrence's own reachable DAG size. let a = quantile_agg(vec![1], Some(2), 0.5); let b = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -1065,7 +1065,7 @@ mod tests { #[test] fn dedup_gates_sharing_the_same_as_aggregate() { // `Dedup { cols }` adds `cols` as a unique key — so two identical - // `Dedup` subtrees over a keyed column *do* merge, exercising the + // `Dedup` sub-DAGs over a keyed column *do* merge, exercising the // legality gate on a non-`Aggregate` node. let dedup = |cols: Vec| QueryExpr::Dedup { cols, @@ -1073,7 +1073,7 @@ mod tests { }; let a = dedup(vec![1]); let b = dedup(vec![1]); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -1102,7 +1102,7 @@ mod tests { let a = without_agg(); let b = without_agg(); assert!(!a.output_schema().unwrap().has_unique_key()); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index 2aed64aa2..cc201a617 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -1,11 +1,11 @@ //! Column-reference and scalar-operator vocabulary shared by the whole -//! canonical [`QueryExpr`](super::query_expr::QueryExpr) tree. +//! canonical [`QueryExpr`](super::query_expr::QueryExpr) DAG. //! //! Issue #205: the scalar expression shapes (`Column`/`Literal`/`Compare`/…) -//! used to live in a separate, self-recursive `Expr` tree here, reachable +//! used to live in a separate, self-recursive `Expr` DAG here, reachable //! from `QueryExpr` only through wrapper fields (`Predicate`, `ProjectItem`, //! `SortKey`). They're variants of `QueryExpr` itself now — one recursive -//! tree, not two type families joined by wrappers — generic over the same +//! DAG, not two type families joined by wrappers — generic over the same //! column-reference state `C` the rest of `QueryExpr` already carries //! (issue #179): [`ColumnRef`] (name-based, front-end-emitted) or //! [`ColumnId`](super::schema::ColumnId) (positional, once bound). diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index f8f7e3519..bb9b8e309 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -1,7 +1,7 @@ //! The canonical pre-ASAP intent algebra IR. //! //! - [`query_expr`] — the canonical, language- and deployment-independent -//! intent algebra: one recursive [`QueryExpr`] tree (relational operators +//! intent algebra: one recursive [`QueryExpr`] DAG (relational operators //! *and* scalar expression shapes both, since issue #205) + [`AggIntent`], //! generic over the column-reference state (positional [`ColumnId`] once //! bound, name-based [`ColumnRef`] before). @@ -12,16 +12,16 @@ //! - [`schema`] — the per-edge [`Schema`] every node carries. //! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` //! into a positional `ColumnId` against an in-scope [`Schema`]. -//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] tree to +//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] DAG to //! canonical [`ResolvedQueryExpr`] (issue #179): both front ends //! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` //! directly during their own `interpret` step and call //! [`resolve_root`] on the result — there is no separate per-language -//! relational tree or converter anymore. +//! relational DAG or converter anymore. //! - [`canonicalize`] — post-lowering structural normalization of [`QueryExpr`] //! (issue #34), run by [`resolve_root`]. //! - [`cse`] — workload-level structural common-subexpression elimination -//! over an already-`resolve_root`'d tree (issue #212, #222, #223), run +//! over an already-`resolve_root`'d DAG (issue #212, #222, #223), run //! *after* `resolve_root` / `canonicalize` and *before* implementation //! (`asap_aware_mapping::replacement`). //! @@ -50,7 +50,7 @@ pub use column_resolution::{ output_schema_for_aggregate, resolve_column_ref, resolve_column_refs, resolve_expr, ResolveError, }; -pub use cse::share_common_subtrees; +pub use cse::share_common_sub_dags; pub use expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; pub use query_expr::{ aggregate_output_schema, any_measure_filtered, AtModifier, BinaryOpKind, ColState, DataModel, @@ -59,6 +59,6 @@ pub use query_expr::{ SortKey, Source, TimeShift, UnresolvedQueryExpr, VectorGrouping, VectorMatch, VectorMatchKind, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; -pub use resolve::{resolve_root, ResolveTreeError}; +pub use resolve::{resolve_root, ResolveDAGError}; pub use schema::{Column, ColumnId, DataType, Schema}; pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index 9f0e800c1..60f0e4ec0 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -1,13 +1,13 @@ //! The canonical pre-ASAP intent algebra IR. //! -//! Language- and deployment-independent. `Rc`-owned tree — a child field is +//! Language- and deployment-independent. `Rc`-owned DAG — a child field is //! `Rc>` rather than `Box>` so a structurally //! identical sub-expression can be shared (the same `Rc`) across more than //! one parent, within one query or across a `QueryWorkload` batch, instead of //! being duplicated. Nothing in this module produces that sharing on its //! own — construction still allocates a fresh `Rc` per node, the same shape -//! as the old `Box` tree — a separate CSE pass is what turns two -//! independently constructed, structurally-equal subtrees into two +//! as the old `Box` DAG — a separate CSE pass is what turns two +//! independently constructed, structurally-equal sub-DAGs into two //! references to one `Rc` (issue #212, #222). Column identity is //! **positional** (`Aggregate.reduction: Reduction`, wrapping `GroupKeys` //! for the grouped case), resolved by the [`SchemaResolver`](super::schema_resolver) against @@ -23,13 +23,13 @@ use super::agg_intent::AggIntent; use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; use super::schema::{Column, ColumnId, DataType, Schema}; -/// The column-reference resolution state a [`QueryExpr`] tree carries — +/// The column-reference resolution state a [`QueryExpr`] DAG carries — /// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always /// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every /// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] /// before binding. The only place the two states differ in *shape* rather /// than just in which type fills `C` is [`QueryExpr::Scan`]'s `schema` field: -/// a bound tree's binding schema is always known (the SchemaResolver is total, so +/// a bound DAG's binding schema is always known (the SchemaResolver is total, so /// [`ScanSchema`](Self::ScanSchema) `= Schema`); an unresolved front-end /// `Scan` knows its schema only when the front end already has it without /// binding — a SQL leaf, catalog-backed (`Some`) — `None` (PromQL) defers to @@ -37,7 +37,7 @@ use super::schema::{Column, ColumnId, DataType, Schema}; pub trait ColState: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de> { - /// What [`QueryExpr::Scan`]'s `schema` field holds for a tree in this state. + /// What [`QueryExpr::Scan`]'s `schema` field holds for a DAG in this state. type ScanSchema: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de>; } @@ -49,7 +49,7 @@ impl ColState for ColumnRef { type ScanSchema = Option; } -/// Errors from schema derivation over a canonical tree. +/// Errors from schema derivation over a canonical DAG. #[derive(Debug, Error)] pub enum QueryExprError { #[error("invalid scalar function signature: {0}")] @@ -652,7 +652,7 @@ impl ConcatDiscriminatorKey { #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] pub enum QueryExpr { /// Outermost leaf. `schema` is the **binding schema** — the resolved column - /// set every positional `ColumnId` in the tree indexes into, *not* a full + /// set every positional `ColumnId` in the DAG indexes into, *not* a full /// description of the runtime row — once bound (`schema: Schema`, always /// present: the [`SchemaResolver`](super::schema_resolver) is total). Before binding, a /// front-end-emitted `Scan` (`C = ColumnRef`) knows it only when the front @@ -671,7 +671,7 @@ pub enum QueryExpr { predicates: Vec>, schema: C::ScanSchema, }, - /// A scalar sub-expression sitting in an **operator-tree position** — a + /// A scalar sub-expression sitting in an **operator-DAG position** — a /// [`BinaryOp`](Self::BinaryOp) operand for ` op ` /// thresholds / unit conversions (#35), a /// [`PromqlVectorFromScalar`](Self::PromqlVectorFromScalar) child, or a @@ -680,7 +680,7 @@ pub enum QueryExpr { /// Formerly its own leaf variant, `PromqlScalar(f64)`. Issue #220: that /// variant held exactly the same value [`Literal`](Self::Literal) does /// (every PromQL scalar is `f64`), duplicating it for no reason but - /// *which tree position* it was allowed to appear in. This wrapper + /// *which DAG position* it was allowed to appear in. This wrapper /// carries that position instead of the value — the inner node is an /// ordinary scalar sub-language expression (in practice always /// `Literal(ScalarValue::Float64(_))`, since a front end only ever @@ -941,9 +941,9 @@ pub enum QueryExpr { // ── Scalar expression shapes (issue #205) ─────────────────────────── // - // Formerly a separate, self-recursive `Expr` tree, reachable from the + // Formerly a separate, self-recursive `Expr` DAG, reachable from the // operator variants above only through wrapper fields (`Predicate`, - // `ProjectItem`, `SortKey`). They're variants of this same tree now — a + // `ProjectItem`, `SortKey`). They're variants of this same DAG now — a // scalar sub-expression is only ever reachable through one of those same // wrapper positions (`Filter.pred`, `ProjectItem.expr`, `Aggregate.having`, // `PromqlRelabel.value`, `SQLWindowFunc.args`, …), which is a *convention* this @@ -957,7 +957,7 @@ pub enum QueryExpr { // restricting which variants are constructible in a scalar position) adds // real type-level machinery for a distinction every constructor already // has to get right structurally anyway (a `Filter` is never built with an - // operator subtree as its `pred`). + // operator sub-DAG as its `pred`). /// A column reference — unresolved [`ColumnRef`] (front-end-emitted, `C = /// ColumnRef`) or positional [`ColumnId`] (once bound, `C = ColumnId`). Column(C), @@ -1014,7 +1014,7 @@ pub enum QueryExpr { impl QueryExpr { /// Construct the [`PromqlScalarBridge`](Self::PromqlScalarBridge) leaf /// for a bare PromQL numeric literal / folded constant scalar (issue - /// #220) — `Literal(ScalarValue::Float64(v))` at an operator-tree + /// #220) — `Literal(ScalarValue::Float64(v))` at an operator-DAG /// position. The one constructor every front end / test that used to /// write `QueryExpr::PromqlScalar(v)` should use instead. pub fn promql_scalar(v: f64) -> Self { @@ -1086,7 +1086,7 @@ impl QueryExpr { } } - /// Recursively collect every column reference in a **scalar** subtree — + /// Recursively collect every column reference in a **scalar** sub-DAG — /// used by the [`SchemaResolver`](super::schema_resolver::SchemaResolver) to seed usage-derived /// leaf schemas, and available to post-ASAP binding for column-lineage / /// selectivity. @@ -1146,20 +1146,20 @@ impl QueryExpr { } } -/// The canonical, positional, resolved tree — what the bare `QueryExpr` name +/// The canonical, positional, resolved DAG — what the bare `QueryExpr` name /// has always meant (the default `C = ColumnId`). Every existing consumer /// keeps using `QueryExpr` unparameterized; this alias exists only to name /// the resolved state explicitly at a use site that also wants to name /// [`UnresolvedQueryExpr`] nearby. pub type ResolvedQueryExpr = QueryExpr; -/// The front-end-emitted, name-based, unresolved tree — +/// The front-end-emitted, name-based, unresolved DAG — /// `QueryExpr`: front ends construct this directly during their /// own `interpret` step (issue #179), and the [`SchemaResolver`](super::schema_resolver) /// resolves it into [`ResolvedQueryExpr`]. pub type UnresolvedQueryExpr = QueryExpr; -// `output_schema` needs a fully bound tree — it reads `Scan.schema` as a plain +// `output_schema` needs a fully bound DAG — it reads `Scan.schema` as a plain // `Schema` and resolves every scalar `Expr::Column` positionally — so it lives // only on the resolved instantiation, not `impl QueryExpr`. // Same reasoning as `AggIntent`'s `output_column`/`requires`/`is_per_series` @@ -1171,7 +1171,7 @@ impl QueryExpr { infer_expr_type(self, input) } - /// Output schema of the root of a canonical tree. + /// Output schema of the root of a canonical DAG. pub fn output_schema(&self) -> Result { match self { QueryExpr::Scan { schema, .. } => Ok(schema.clone()), @@ -2629,7 +2629,7 @@ mod tests { /// place of the old `PromqlScalar(v)` leaf — wraps exactly /// `Literal(ScalarValue::Float64(v))`: the same value a SQL-emitted typed /// float literal in a scalar-sub-language position would carry, just at a - /// different tree position. `as_promql_scalar` is the round-trip inverse. + /// different DAG position. `as_promql_scalar` is the round-trip inverse. #[test] fn promql_scalar_bridges_a_literal_float_at_an_operator_position() { let bridge = QueryExpr::::promql_scalar(2.5); @@ -2641,7 +2641,7 @@ mod tests { // The same value a SQL `Compare`/`Arithmetic` operand would carry, in // its native (unwrapped, no row schema) scalar-sub-language position — - // no longer a different variant, just not bridged to this tree + // no longer a different variant, just not bridged to this DAG // position. let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); assert_eq!(bridge.as_promql_scalar(), Some(2.5)); @@ -2655,9 +2655,9 @@ mod tests { assert_eq!(scan(vec![], None, vec![]).as_promql_scalar(), None); } - /// Pins the tree-position distinction issue #220 asks for: the very same + /// Pins the DAG-position distinction issue #220 asks for: the very same /// `Literal(ScalarValue::Float64(_))` value has a row schema when it sits - /// at the operator-tree position (wrapped in `PromqlScalarBridge` — a + /// at the operator-DAG position (wrapped in `PromqlScalarBridge` — a /// `BinaryOp` operand, `PromqlVectorFromScalar` child, or a query root), /// and has none when it sits bare, in a scalar-sub-language position /// (`Compare`/`Arithmetic`/… operand) — no longer decided by which of two diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs index 95b482e54..104c51a68 100644 --- a/crates/types/src/pre_asap/resolve.rs +++ b/crates/types/src/pre_asap/resolve.rs @@ -20,7 +20,7 @@ //! through logical optimization, only going positional once they lower to a //! physical plan. Resolving once, immediately after each front end's own //! `interpret` step, is the better trade for *this* codebase's shape — one -//! front-end-facing tree feeding several independent downstream passes +//! front-end-facing DAG feeding several independent downstream passes //! (`canonicalize`, the cost model, `dag_export`, schema/type inference, //! `asap-aware-mapping`'s summary binding) — for three concrete reasons: //! @@ -30,7 +30,7 @@ //! schemas are concatenated. A bare name is ambiguous the moment two sources //! share one; `ColumnId` is what makes "the second `service`, position 4, not //! the first" a fact recorded once, instead of a lookup redone at every use site. -//! 2. **A name's meaning changes going up the tree.** `Project` renames/aliases, +//! 2. **A name's meaning changes going up the DAG.** `Project` renames/aliases, //! `Aggregate` collapses columns and introduces synthetic ones, `Join` //! concatenates two schemas — a name valid at a `Scan` leaf isn't //! automatically the right binding three nodes up; it has to be reinterpreted @@ -64,9 +64,9 @@ use super::query_expr::{ use super::schema::{ColumnId, Schema}; use super::schema_resolver::SchemaResolver; -/// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] tree. +/// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] DAG. #[derive(Debug, Error)] -pub enum ResolveTreeError { +pub enum ResolveDAGError { /// A column reference did not resolve against its in-scope schema. #[error("column resolution failed: {0}")] Resolve(#[from] ResolveError), @@ -76,22 +76,22 @@ pub enum ResolveTreeError { Schema(#[from] QueryExprError), } -/// Resolve a whole [`UnresolvedQueryExpr`] tree rooted at `tree` into canonical +/// Resolve a whole [`UnresolvedQueryExpr`] DAG rooted at `dag` into canonical /// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the /// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the /// result. -pub fn resolve_root(tree: &UnresolvedQueryExpr) -> Result { - resolve_root_with_inherited(tree, &[]) +pub fn resolve_root(dag: &UnresolvedQueryExpr) -> Result { + resolve_root_with_inherited(dag, &[]) } /// [`resolve_root`] with label names inherited from an enclosing scope seeded /// into the leaf schema, used when re-binding a `BinaryOp` side (issue #52). fn resolve_root_with_inherited( - tree: &UnresolvedQueryExpr, + dag: &UnresolvedQueryExpr, inherited: &[String], -) -> Result { - let fallback = SchemaResolver::new().resolve_schema_with_inherited(tree, inherited); - let l3 = resolve(tree, &fallback)?; +) -> Result { + let fallback = SchemaResolver::new().resolve_schema_with_inherited(dag, inherited); + let l3 = resolve(dag, &fallback)?; Ok(super::canonicalize::canonicalize(l3)) } @@ -100,11 +100,11 @@ fn resolve_root_with_inherited( /// derived output schema — so a `JOIN`'s concatenated schema and a cross- /// series aggregate's frozen-closed output bind to the right positions. fn resolve( - tree: &UnresolvedQueryExpr, + dag: &UnresolvedQueryExpr, fallback: &Schema, -) -> Result { +) -> Result { use super::query_expr::QueryExpr as QE; - Ok(match tree { + Ok(match dag { QE::Scan { source, predicates, @@ -123,7 +123,7 @@ fn resolve( } // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220) sitting at this operator-tree position — resolved through + // #220) sitting at this operator-DAG position — resolved through // `resolve_expr`, same as every other scalar position (`Predicate`, // `ProjectItem.expr`, …), not the operator walk. In practice it's // always a `Literal`, which has no `ColumnRef` to resolve, so @@ -232,7 +232,7 @@ fn resolve( }; let having = having .as_ref() - .map(|Predicate(h)| -> Result { + .map(|Predicate(h)| -> Result { let out_schema = aggregate_output_schema( &child_schema, &reduction, @@ -279,7 +279,7 @@ fn resolve( // other side. let discriminator_unique_key = discriminator_unique_key .as_ref() - .map(|key| -> Result<_, ResolveTreeError> { + .map(|key| -> Result<_, ResolveDAGError> { let schema = children .first() .ok_or(QueryExprError::EmptyConcat)? @@ -427,7 +427,7 @@ fn resolve( // different label sets, so each branch resolves against its OWN // bound schema; but an independently-bound side still has to see // label names an *enclosing* node references (issue #52). - let own = super::schema_resolver::collect_referenced_columns(tree); + let own = super::schema_resolver::collect_referenced_columns(dag); let inherited: Vec = inherited_names(fallback) .into_iter() .filter(|n| !own.contains(n)) @@ -501,7 +501,7 @@ fn resolve_group_keys( /// uniformly to every `Aggregate`, not just PromQL's: SQL's `GROUP BY` keys /// are always genuinely present (DataFusion validates the plan), so the /// "drop instead of reject" branch is simply never exercised there — the -/// lenient resolver is a no-op difference for a SQL tree, not a behavior +/// lenient resolver is a no-op difference for a SQL DAG, not a behavior /// change. fn resolve_reduction( reduction: &Reduction, @@ -815,7 +815,7 @@ mod tests { } /// Issue #228 review, end-to-end: `resolve_root` over a `Concat` whose - /// discriminator column is referenced *nowhere else* in the tree, with a + /// discriminator column is referenced *nowhere else* in the DAG, with a /// schema-less (usage-derived) leaf `Scan` in the first branch — exactly /// the scenario the review flagged. Before the `schema_resolver.rs` fix, the /// SchemaResolver's fallback schema wouldn't contain `phi` at all, and this diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index b8d23b6a3..d6bbd1610 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -1,7 +1,7 @@ //! The **SchemaResolver** — name resolution as an explicit pass. //! //! [`SchemaResolver::resolve_schema`] produces the complete, self-contained [`Schema`] every -//! `ColumnId` in the canonical tree indexes into. [`resolve`](super::resolve) +//! `ColumnId` in the canonical DAG indexes into. [`resolve`](super::resolve) //! then becomes purely structural: it threads the SchemaResolver's schema and //! positional resolution downstream is **total**. //! @@ -66,27 +66,27 @@ impl SchemaResolver { Self { catalog } } - /// Resolve the complete [`Schema`] in scope for a query rooted at `tree`. + /// Resolve the complete [`Schema`] in scope for a query rooted at `dag`. /// /// Contains the time axis, the synthetic `value` column, and one column - /// per distinct name referenced anywhere in the tree — so positional + /// per distinct name referenced anywhere in the DAG — so positional /// `ColumnId` resolution downstream is total. - pub fn resolve_schema(&self, tree: &UnresolvedQueryExpr) -> Schema { - self.resolve_schema_with_inherited(tree, &[]) + pub fn resolve_schema(&self, dag: &UnresolvedQueryExpr) -> Schema { + self.resolve_schema_with_inherited(dag, &[]) } /// Like [`resolve_schema`](Self::resolve_schema), but also seeds `inherited` label names that are - /// referenced by an **enclosing** scope rather than by `tree` itself. This is + /// referenced by an **enclosing** scope rather than by `dag` itself. This is /// how an independently-bound `BinaryOp` side (each side re-binds against its - /// own sub-tree) still sees an outer aggregate's group keys — e.g. the + /// own sub-DAG) still sees an outer aggregate's group keys — e.g. the /// `__name__` / `job` in `sum by (__name__)(a or b)`, which appear in neither /// side's own matchers (issue #52). pub fn resolve_schema_with_inherited( &self, - tree: &UnresolvedQueryExpr, + dag: &UnresolvedQueryExpr, inherited: &[String], ) -> Schema { - let mut columns: Vec = leftmost_scan_name(tree) + let mut columns: Vec = leftmost_scan_name(dag) .and_then(|name| self.catalog.columns_for(name)) .unwrap_or_else(default_leaf_columns); @@ -99,7 +99,7 @@ impl SchemaResolver { // Append one column per referenced-but-unknown name (group keys etc.), // plus any inherited-from-enclosing-scope names. - let referenced = collect_referenced_columns(tree); + let referenced = collect_referenced_columns(dag); for name in referenced.iter().chain(inherited) { if !columns.iter().any(|c| c.name == *name) { columns.push(Column::new(name.clone(), DataType::Utf8, true)); @@ -136,13 +136,13 @@ fn push_ref_name(c: &ColumnRef, out: &mut Vec) { } } -/// The leftmost `Scan`'s source name in a canonical (`UnresolvedQueryExpr`) tree — +/// The leftmost `Scan`'s source name in a canonical (`UnresolvedQueryExpr`) DAG — /// the [`collect_referenced_columns`] counterpart to what a dedicated -/// `Source` leaf type would carry as a method; the canonical tree's `Scan` +/// `Source` leaf type would carry as a method; the canonical DAG's `Scan` /// leaf needs this walk written out instead. -fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { +fn leftmost_scan_name(dag: &UnresolvedQueryExpr) -> Option<&str> { use UnresolvedQueryExpr as QE; - match tree { + match dag { QE::Scan { source, .. } => Some(match source { super::query_expr::Source::TimeSeries { metric } => metric.as_str(), super::query_expr::Source::Table { table_ref } => table_ref.as_str(), @@ -191,7 +191,7 @@ fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { } } -/// Collect every distinct column name referenced anywhere in `tree` that +/// Collect every distinct column name referenced anywhere in `dag` that /// resolves positionally — every place a front end constructing /// [`QueryExpr`](super::query_expr::QueryExpr) directly (issue /// #179) puts a name-based reference: `Scan.predicates`, `Aggregate`'s @@ -200,7 +200,7 @@ fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { /// `SQLWindowFunc.args`/`partition_by`/`order_by`, `Join.pred`, `PromqlRelabel.value`. /// The SchemaResolver seeds these into the usage-derived leaf so positional /// resolution downstream is total. -pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec { +pub(crate) fn collect_referenced_columns(dag: &UnresolvedQueryExpr) -> Vec { use UnresolvedQueryExpr as QE; fn named(expr: &UnresolvedQueryExpr, out: &mut Vec) { for c in expr.columns_referenced() { @@ -322,7 +322,7 @@ pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec Vec = Vec::new(); - walk(tree, &mut out); + walk(dag, &mut out); out.sort(); out.dedup(); out @@ -390,7 +390,7 @@ mod tests { #[test] fn pearson_corr_inputs_seed_usage_derived_schema() { use crate::pre_asap::{AggIntent, Reduction}; - let tree = UnresolvedQueryExpr::Aggregate { + let dag = UnresolvedQueryExpr::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::PearsonCorr { left: ColumnRef::Named("x".into()), @@ -401,8 +401,8 @@ mod tests { having: None, child: Rc::new(src("m")), }; - assert_eq!(collect_referenced_columns(&tree), vec!["x", "y"]); - let schema = SchemaResolver::new().resolve_schema(&tree); + assert_eq!(collect_referenced_columns(&dag), vec!["x", "y"]); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!(schema.column_id("x").is_some()); assert!(schema.column_id("y").is_some()); } @@ -420,7 +420,7 @@ mod tests { fn sort_partition_keys_land_in_schema() { // Per-group ranking keys (`topk by (host)` → `Sort.partition_by`) must be // seeded into the usage-derived leaf so they resolve positionally. - let tree = UnresolvedQueryExpr::Sort { + let dag = UnresolvedQueryExpr::Sort { keys: vec![super::super::query_expr::SortKey { expr: UnresolvedQueryExpr::Column(ColumnRef::SampleValue), ascending: false, @@ -429,23 +429,23 @@ mod tests { partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), child: Rc::new(src("hits")), }; - let schema = SchemaResolver::new().resolve_schema(&tree); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!(schema.column_id("host").is_some()); } /// Issue #228 review: a `Concat`'s `discriminator_unique_key` columns — - /// even one referenced nowhere else in the tree — must be seeded into + /// even one referenced nowhere else in the DAG — must be seeded into /// the usage-derived fallback schema, exactly like `Dedup.cols`, or /// `resolve.rs`'s later `resolve_column_ref` fails `NotFound` for a /// column the caller correctly named. #[test] fn concat_discriminator_key_is_seeded_into_the_resolver_schema() { - let tree = UnresolvedQueryExpr::concat_with_discriminator( + let dag = UnresolvedQueryExpr::concat_with_discriminator( vec![src("m")], ColumnRef::Named("phi".into()), vec![ColumnRef::Named("host".into())], ); - let schema = SchemaResolver::new().resolve_schema(&tree); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!( schema.column_id("phi").is_some(), "discriminator column must be seeded" @@ -458,7 +458,7 @@ mod tests { #[test] fn inherited_names_are_seeded_alongside_referenced() { - // A `BinaryOp` side re-binds against its own sub-tree, but must still see + // A `BinaryOp` side re-binds against its own sub-DAG, but must still see // an enclosing aggregate's group key (`__name__` / `job`) that appears in // neither side's own matchers (issue #52). `resolve_schema_with_inherited` seeds it. let schema = diff --git a/crates/types/tests/planner_vocabulary.rs b/crates/types/tests/planner_vocabulary.rs index 567e2c284..f14a56ee8 100644 --- a/crates/types/tests/planner_vocabulary.rs +++ b/crates/types/tests/planner_vocabulary.rs @@ -33,14 +33,14 @@ fn window_edge_names_preserve_wire_values() { // External consumers can use the new resolver and resource names without changing behavior. #[test] fn renamed_schema_and_handoff_apis_are_public() { - let tree = UnresolvedQueryExpr::Scan { + let dag = UnresolvedQueryExpr::Scan { source: Source::TimeSeries { metric: "requests".into(), }, predicates: vec![], schema: None, }; - let schema = SchemaResolver::new().resolve_schema(&tree); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!(schema.column_id("value").is_some()); let bytes = PhysicalHandoffBytes { network_bytes: 12, diff --git a/docs/design_docs/architecture/parse-and-canonicalize.md b/docs/design_docs/architecture/parse-and-canonicalize.md index b111f9730..61149efbb 100644 --- a/docs/design_docs/architecture/parse-and-canonicalize.md +++ b/docs/design_docs/architecture/parse-and-canonicalize.md @@ -18,7 +18,7 @@ SQL ORDER BY COUNT(*) DESC LIMIT 10 ## Canonicalize -`canonicalize` normalizes semantically equivalent intent trees so that equivalent queries +`canonicalize` normalizes semantically equivalent intent DAGs so that equivalent queries from different languages, or differently phrased queries within one language, converge on the same canonical shape. diff --git a/docs/design_docs/architecture/physical-plan-integration.md b/docs/design_docs/architecture/physical-plan-integration.md index 47fc3a1b6..01e38fc46 100644 --- a/docs/design_docs/architecture/physical-plan-integration.md +++ b/docs/design_docs/architecture/physical-plan-integration.md @@ -425,7 +425,7 @@ requires a normal result; setting both guards or attaching a guard to a non-divi operator is invalid. Compilers must preserve this typed condition rather than recovering average semantics from query text. -### Candidate pruning is a subgraph +### Candidate pruning is a sub-DAG Candidate-based TopK uses a summary key readout, a general semi-join over explicit matching key columns, grouped Sort by the authoritative score, and diff --git a/docs/design_docs/concepts/post-asap-ir.md b/docs/design_docs/concepts/post-asap-ir.md index 67b5535a1..e844687a3 100644 --- a/docs/design_docs/concepts/post-asap-ir.md +++ b/docs/design_docs/concepts/post-asap-ir.md @@ -2,7 +2,7 @@ The goal of the post-ASAP IR is to represent operations using ASAP primitives such as sketches, exact summaries, samples and wavelets. Post-ASAP IR also -retains exact Pre-ASAP subtrees and supports operations over summary readouts, +retains exact Pre-ASAP sub-DAGs and supports operations over summary readouts, since only some query operations can be satisfied using summaries. The lists below cover every current variant of @@ -36,7 +36,7 @@ summary family supports incremental maintenance. ## Exact work and composition nodes -- `KeepPreAsap`: retain an exact Pre-ASAP subtree when it is not rewritten. +- `KeepPreAsap`: retain an exact Pre-ASAP sub-DAG when it is not rewritten. - `BinaryOp`: combine independently planned operands with the specified binary semantics and execution timing. - `ValueOperation`: apply aggregate, exact-function, population, projection, @@ -55,12 +55,12 @@ approximate readouts still require composed accuracy guarantees. See the and [physical-plan integration](../architecture/physical-plan-integration.md) for the corresponding correctness and realization requirements. -## Tree and exported DAG forms +## In-memory and exported DAG forms The Pre-ASAP DAG and the Post-ASAP DAG are both logical: they describe what is computed, not which physical operators execute it. The Post-ASAP DAG has two -forms of the same content. Planning builds and shares `SummaryNode` trees. -`compile_post_asap_dag` converts a selected tree into a +forms of the same content. Planning builds and shares `SummaryNode` DAGs. +`compile_post_asap_dag` converts a selected DAG into a [`PostAsapDag`](../../../crates/types/src/post_asap/post_asap_dag.rs) with stable node IDs and typed edges; `PostAsapDagDocument` is its versioned wire envelope. Physical compilation consumes `PostAsapDag` and produces a separate diff --git a/docs/design_docs/concepts/pre-asap-ir.md b/docs/design_docs/concepts/pre-asap-ir.md index 227e61e39..735f522f3 100644 --- a/docs/design_docs/concepts/pre-asap-ir.md +++ b/docs/design_docs/concepts/pre-asap-ir.md @@ -31,7 +31,7 @@ Only semantics that affect correctness, summary applicability, or cost become fi ### PromQL-specific -- PromqlScalarBridge — holds a scalar sub-expression at an operator-tree position. +- PromqlScalarBridge — holds a scalar sub-expression at an operator-DAG position. - EvalTimestamp — provides the evaluation timestamp as a scalar. - PromqlVectorFromScalar — promotes a scalar to a label-less instant vector. - PromqlScalarFromVector — collapses a single-series vector to a scalar. diff --git a/docs/design_docs/decisions/concat-unique-keys.md b/docs/design_docs/decisions/concat-unique-keys.md index 778f6061e..9c0148abf 100644 --- a/docs/design_docs/decisions/concat-unique-keys.md +++ b/docs/design_docs/decisions/concat-unique-keys.md @@ -36,7 +36,7 @@ that a discriminator-based unique key would let it drop? ## What was checked Both current `Concat`-constructing call sites, and every consumer of -`Schema::unique_keys` in the tree: +`Schema::unique_keys` in the DAG: - **PromQL `histogram_quantiles`** — [`walk_histogram_quantiles`](../../../crates/frontend-promql/src/promql.rs). @@ -66,8 +66,8 @@ Both current `Concat`-constructing call sites, and every consumer of `sql_lowering.rs`, `dag_export.rs`, `variant_coverage.rs`, netflow/synthetic test fixtures) — none of them builds a fresh `Concat` with a `Dedup` on top that this feature could remove. -- **Every consumer of `Schema::unique_keys`** in the tree, to check for a - cost beyond "a literal `Dedup` node": `pre_asap::cse::share_common_subtrees` +- **Every consumer of `Schema::unique_keys`** in the DAG, to check for a + cost beyond "a literal `Dedup` node": `pre_asap::cse::share_common_sub_dags` (gates CSE producer-sharing on `Schema::has_unique_key()`) and `asap_aware_mapping::rollup::is_legal_rollup_source` (gates rollup-source legality the same way, on an *`Aggregate`'s* own output schema). Neither @@ -88,7 +88,7 @@ Both current `Concat`-constructing call sites, and every consumer of The investigation's conclusion stands: neither `histogram_quantiles` nor `ROLLUP`/`CUBE`/`GROUPING SETS` lowering emits a `Dedup` (or anything playing that role) after its `Concat` today, so there is nothing redundant in the -tree for a discriminator-based override to remove *right now*. On review, +DAG for a discriminator-based override to remove *right now*. On review, the decision was made to build the extension point anyway, ahead of a proven call-site win, rather than wait for one. That is a legitimate call to make differently from the investigation's own recommendation — "no current @@ -114,7 +114,7 @@ for why it's fine to ship unused. overclaimed. - `QueryExpr::concat(children)` — the ordinary constructor (`discriminator_unique_key: None`), meant to replace the bare `QueryExpr::Concat { children }` struct literal - everywhere in the tree so a future field addition doesn't force every call + everywhere in the DAG so a future field addition doesn't force every call site to re-litigate this choice. - `QueryExpr::concat_with_discriminator(children, discriminator, inner_key)` — the override constructor. @@ -127,7 +127,7 @@ for why it's fine to ship unused. branch's own output schema — the same schema `output_schema()` derives the merged shape from — so the feature works correctly end-to-end for a future caller upstream of `resolve_root`, even though no such caller exists yet. -- Every other match/construction site touching `Concat` across the tree +- Every other match/construction site touching `Concat` across the DAG (`canonicalize.rs`, `cse.rs`, `schema_resolver.rs`, `dag_export.rs`, `asap-aware-mapping`'s `replacement.rs`/`explanation.rs`, and every test/tooling AST walker) was mechanically updated to bind or ignore the new @@ -156,9 +156,9 @@ Three independent things hold `discriminator_unique_key: None` as the observable behavior for every caller that doesn't ask for the override: 1. **Every real construction path defaults to `None`.** `QueryExpr::concat` - hardcodes it; every call site in the tree (including both real lowering + hardcodes it; every call site in the DAG (including both real lowering call sites) uses `concat`, not `concat_with_discriminator`, so nothing in - the current tree can produce `Some` at all. + the current DAG can produce `Some` at all. 2. **`output_schema()`'s branch on the field is additive.** The `None` arm is textually the same clear-and-return the code already did — `s.unique_keys.clear(); ... Ok(s)` — with the `Some` branch reached only when the field is populated. This is @@ -236,7 +236,7 @@ accuracy issue. All three are fixed on the same PR: (which *is* walked, `push_ref_name`-style). Concretely: a future `concat_with_discriminator(branches, discriminator_col, inner_key)` call over an open query, where the discriminator column isn't otherwise - referenced anywhere else in the tree, with a schema-less leaf `Scan` in + referenced anywhere else in the DAG, with a schema-less leaf `Scan` in the first branch — the SchemaResolver's fallback schema wouldn't contain the discriminator name, and `resolve.rs`'s later `resolve_column_ref` call would fail `NotFound` for a column the caller correctly named. Fixed: @@ -294,4 +294,4 @@ accuracy issue. All three are fixed on the same PR: - Reopening SQL's rejection of `GROUPING()` so `lower_grouping_sets` has a real discriminator (`__grouping_id`) to assert — a separate design decision. - Any canonicalization rule or CSE/rollup scenario that would actually *read* - a `Concat`'s asserted `unique_keys` for the first time in the current tree. + a `Concat`'s asserted `unique_keys` for the first time in the current DAG. diff --git a/docs/design_docs/decisions/cse-cost-model.md b/docs/design_docs/decisions/cse-cost-model.md index 5689390aa..ed7e54e08 100644 --- a/docs/design_docs/decisions/cse-cost-model.md +++ b/docs/design_docs/decisions/cse-cost-model.md @@ -4,12 +4,12 @@ ## Context -[`asap_types::pre_asap::cse::share_common_subtrees`](../../../crates/types/src/pre_asap/cse.rs) +[`asap_types::pre_asap::cse::share_common_sub_dags`](../../../crates/types/src/pre_asap/cse.rs) (issue #223 stages 1-2, PR #235) already *detects* every structurally-identical, -legally-shareable (`Schema::unique_keys`-gated) subtree and shares it +legally-shareable (`Schema::unique_keys`-gated) sub-DAG and shares it **unconditionally** — there is no cost gate on top of legality. This document decides the framework for stage 4, "wire workload-level CSE credit into -`CostModel`" — turning "these two subtrees are the same computation" into +`CostModel`" — turning "these two sub-DAGs are the same computation" into "and it's actually worth maintaining one shared summary for them." ## The two textbook framings (as posed in #237) @@ -27,7 +27,7 @@ compares two real, overridable cost estimates for every CSE candidate with two or more consumers: - `cse_recompute_cost(candidate) * candidate.consumer_count` — the total cost - of recomputing the subtree independently at every use site. + of recomputing the sub-DAG independently at every use site. - `cse_shared_maintenance_cost(candidate)` — the cost of keeping one shared summary alive and continuously updated for the workload's lifetime. @@ -40,7 +40,7 @@ the way sharing a relational scan is in a textbook OLTP optimizer — it is a sketch/accumulator that (per this crate's stated purpose: *workload*-level planning, not single-query) is typically kept **continuously updated** as new data arrives, for as long as the workload runs, regardless of how often it's -actually read. A structurally-shareable subtree that is cheap to recompute on +actually read. A structurally-shareable sub-DAG that is cheap to recompute on demand, or rarely queried, can cost more to keep alive as a standing shared summary than to just recompute independently at each of its (few, or cheap) use sites. A blanket "always share" rule cannot express that trade-off; a @@ -63,7 +63,7 @@ same way a real cost-based optimizer would. ## Layering constraint -`share_common_subtrees` lives in `asap-types::pre_asap` — a lower layer that +`share_common_sub_dags` lives in `asap-types::pre_asap` — a lower layer that `asap-aware-mapping` (which owns `CostModel`) depends on, never the reverse. Detection therefore cannot consult cost even if it wanted to. This is why stage 1/2's detection stays unconditional (correctly, as a legality-only @@ -74,13 +74,13 @@ gate) and the cost-aware decision is applied downstream, in [`CandidateLogicalASAPDAGs::cost_sorted`](../../../crates/asap-aware-mapping/src/replacement.rs) is where this hooks in today. `search_workload_with` computes each shared -subtree's true `consumer_count` across the whole workload up front (the same +sub-DAG's true `consumer_count` across the whole workload up front (the same role `implement_workload_with`'s pre-pass used to play, before that function was retired along with `bind.rs` — this crate no longer commits to one physically-materialized answer at all; picking and building one final -`SummaryNode` per shared subtree is a downstream deployment's job, not this +`SummaryNode` per shared sub-DAG is a downstream deployment's job, not this crate's). For a `TargetSubDAGCandidates` whose candidates are a -[`SharedSubtreeStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) +[`SharedSubDAGStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) share-vs-recompute pair, `cost_sorted`'s ranking step (`rank_group`/ `cse_preference`) asks `CostModel::cse_share_decision` once per group — using one representative bound `SummaryNode` built just for that comparison, not @@ -92,18 +92,18 @@ does not prune them. ## Defaults `cse_recompute_cost`'s default is a structural-size proxy: `cse::dag_node_count`, -the number of *unique* nodes in the subtree's DAG (deduplicated by `Rc` +the number of *unique* nodes in the sub-DAG's DAG (deduplicated by `Rc` pointer identity), not a raw serialization length. This distinction matters -here specifically — a `CseCandidate`'s subtree is, by definition, something -CSE already found sharing in, so it's generally a DAG, not a tree; a naive -tree-shaped size measure (a full `serde_json` serialization, or a recursive -walk with no identity tracking) would re-count any descendant the subtree +here specifically — a `CseCandidate`'s sub-DAG is, by definition, something +CSE already found sharing in, so it generally has internal sharing; a naive +per-path size measure (a full `serde_json` serialization, or a recursive +walk with no identity tracking) would re-count any descendant the sub-DAG already shares internally once per parent that reaches it, over-stating the real cost of holding or recomputing it once. `cse_shared_maintenance_cost`'s default is a small per-`SummaryFamilyType` weight table (exact accumulators cheapest, sketches/samples/wavelets/stat-models progressively more expensive to keep continuously updated) scaled to the same order of magnitude as typical -subtree sizes. Both are documented as coarse heuristic proxies — a real +sub-DAG sizes. Both are documented as coarse heuristic proxies — a real deployment with actual memory/update-cost/query-frequency knowledge overrides either or both, same as `size_params` already lets a deployment override `asap-plan`'s built-in sizing formulas without forking anything else. diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index c13c080af..a4eb0c9d3 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -30,8 +30,8 @@ associated with the logical DAG, not a separate computation IR. The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the language-independent query semantics before summary selection. Both are -logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` -exports the selected tree as a `PostAsapDag`, which is the Physical Plan +logical. Planning builds Post-ASAP `SummaryNode` DAGs; `compile_post_asap_dag` +exports the selected DAG as a `PostAsapDag`, which is the Physical Plan Compiler's input. Its per-node execution phase (ingestion or query time) is decided by the selected summary maintenance lifecycle, as the layer contract below states. diff --git a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md index 04c7015f5..bcaa56f7a 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md +++ b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md @@ -579,7 +579,7 @@ The parent/child compatibility rules are: | `PerEvaluation` | `PerEvaluation` | valid | | `PerEvaluation` | `Once` | valid only when the child exposes retained state | -For a tree-shaped pipeline, peak memory is normally the maximum live pipeline +For a linear pipeline, peak memory is normally the maximum live pipeline state, not the sum of every node's memory. At a fan-out, join, merge, or nested summary boundary, multiple child states may coexist and must be combined. diff --git a/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md b/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md index 3a391a53b..d705f6b7a 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md +++ b/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md @@ -489,7 +489,7 @@ a candidate was rejected. provenance, allocations, and rejection reasons. The observable proxy is that a rejected candidate can be diagnosed from exported data without replaying cost ranking. -- **Performance and scalability:** expressions are small trees evaluated during +- **Performance and scalability:** expressions are small DAGs evaluated during candidate construction. No numerical performance claim is made; candidate count and planning latency should be measured before adding richer allocation enumeration. diff --git a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md index 065559650..005670db8 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md +++ b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md @@ -27,7 +27,7 @@ value is replaced. This distinction requires an explicit rule and state contract an append-only quantile sketch cannot by itself implement current-series updates. The current rule retains an exact population. It does not prescribe a particular -tree, heap or sketch implementation, and it does not imply a deletable DDSketch. +DAG, heap or sketch implementation, and it does not imply a deletable DDSketch. ## Membership semantics diff --git a/docs/design_docs/proposals/asap-aware-mapping/optimizations.md b/docs/design_docs/proposals/asap-aware-mapping/optimizations.md index c5b6dd1a9..ace3838af 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/optimizations.md +++ b/docs/design_docs/proposals/asap-aware-mapping/optimizations.md @@ -232,7 +232,7 @@ The two operators are not identical, but their grouping keys are related. If the aggregation is mergeable, Query B may be derived by rolling up Query A. -This creates reuse opportunities across **hierarchically related groupings**, not just identical subtrees. +This creates reuse opportunities across **hierarchically related groupings**, not just identical sub-DAGs. The legality and cost of this transformation depend on: diff --git a/docs/design_docs/proposals/asapquery-rule-coverage.md b/docs/design_docs/proposals/asapquery-rule-coverage.md index dd6b9a924..e496c9fe9 100644 --- a/docs/design_docs/proposals/asapquery-rule-coverage.md +++ b/docs/design_docs/proposals/asapquery-rule-coverage.md @@ -23,7 +23,7 @@ cost, and selection rules under `optimizer/`. The reviewed source is | Collapsible temporal + spatial aggregates | Semantic-equivalent rewriting | The existing rewrite strategy uses accumulator algebra: sum∘sum, sum∘count, min∘min, and max∘max. It rejects all other pairs and requires identical output schemas. | | Sketch alternatives and exact fallback | Covered more generally | `SketchAlgorithmStrategy` enumerates legal summary realizations. The enclosing memo group always retains the original raw expression as the exact fallback; the strategy does not falsely label an approximate sketch as exact. | | Subpopulation label placement | Covered more generally | `HydraGroupingStrategy` and `GroupingStrategy` express per-subpopulation and shared multi-subpopulation realizations. | -| Shared computation | Covered more generally | workload-wide CSE and `SharedSubtreeStrategy` operate on physical DAG identity rather than AQE names. | +| Shared computation | Covered more generally | workload-wide CSE and `SharedSubDAGStrategy` operate on physical DAG identity rather than AQE names. | | Average decomposition | Semantic-equivalent rewriting | The same rewrite strategy exposes independently optimizable sum/count accumulators when null semantics and schema permit it. | | Merge/delete legality | Covered | Summary-family capabilities and lifecycle validation determine which maintenance operations are legal. | | Window-framework selection | Separate physical-planning work | Window selection must compare an extensible set of implementations, including tumbling, sliding, PromSketch-style exponential-histogram windows, and other window frameworks. This audit does not introduce a closed window enum or choose among them. | @@ -41,7 +41,7 @@ does not create a new strategy category. | Which summary algorithm can implement one aggregate intent | `SketchAlgorithmStrategy` | | How grouping/subpopulation state is laid out | `HydraGroupingStrategy` | | Whether an equivalent logical expression exposes better accumulators | `SemanticEquivalentRewriteStrategy` (the broadened existing avg rewrite; `AvgToSumOverCountStrategy` remains a compatibility name) | -| Whether identical physical work is shared | `SharedSubtreeStrategy` | +| Whether identical physical work is shared | `SharedSubDAGStrategy` | | Whether a finer grouping can answer a coarser grouping | `RollupStrategy` | | Whether tighter accuracy can answer a looser request | `AccuracyReconciliationStrategy` | | Whether a larger Top-K result can answer a smaller limit | `TopKLimitReuseStrategy` | diff --git a/docs/design_docs/proposals/decoupling_op_and_expr.md b/docs/design_docs/proposals/decoupling_op_and_expr.md index 58859db41..fe458bb80 100644 --- a/docs/design_docs/proposals/decoupling_op_and_expr.md +++ b/docs/design_docs/proposals/decoupling_op_and_expr.md @@ -55,7 +55,7 @@ where needed, rather than copying every current `QueryExpr` variant unchanged: Use the canonical [`NonASAPOp` and `OperatorNode` definitions](operator-sharing.md#11-unified-operator-type) from the sharing proposal. Both documents describe the same resolved model: operator inputs and scalar query-result references use `Rc`. -`NonASAPOp` is the payload of an ordinary operator, not a second graph-node type. +`NonASAPOp` is the payload of an ordinary operator, not a second DAG-node type. `BinaryOp` likewise uses the single `BinaryOperator` payload specified there. Names are resolved to `ColumnId` before constructing these nodes. Parsing and @@ -193,7 +193,7 @@ For an open label schema, lowering must retain the complete series identity, including unreferenced labels. If the input provides neither a complete label schema nor a full identity value, this lowering is not valid. `vector(s)` remains a real conversion to a one-element, label-free vector. -Scalar expression trees are owned, while their operator references preserve graph +Scalar expression DAGs are owned, while their operator references preserve DAG identity. The companion's [complete DAG example](operator-sharing.md#13-example-composing-a-logical-dag) shows these expressions inside ordinary operators before and after an ASAP rewrite. @@ -241,7 +241,7 @@ older repository dependency. This documentation change upgrades neither dependen | `DISTINCT`, `UNION`, `INTERSECT`, `EXCEPT`, their supported `ALL` forms | `Dedup`, `Concat` / `SetOp` | **Direct.** Preserve bag multiplicity and SQL duplicate/NULL equality rules. | | `ORDER BY`, `LIMIT`, offset-only queries | `Sort`, `Limit` | **Direct.** Preserve direction and NULL placement; `n = None` means no fetch limit. | | Uncorrelated scalar subquery, `EXISTS`, `IN` / `NOT IN (SELECT ...)` | Explicit scalar plan-reading variants | **Direct.** Preserve the cardinality and NULL contracts in §2.2, including when used in a SELECT list. | -| Derived tables and nonrecursive CTEs | Existing operator subgraphs; aliases resolved to output columns | **Lowered.** Naming alone needs no computation node. Reuse must not alter volatile evaluation. | +| Derived tables and nonrecursive CTEs | Existing operator sub-DAGs; aliases resolved to output columns | **Lowered.** Naming alone needs no computation node. Reuse must not alter volatile evaluation. | ### 3.3 PromQL semantic mapping @@ -318,13 +318,13 @@ of complete SQL/PromQL support. Acceptance requires: - The scalar queries and mixed scalar/vector examples in §2.3 need no `PromqlScalarBridge` or equivalent constant-wrapper node. - Operator dependencies inside scalar conversions/subqueries remain visible and - shared; scalar trees remain owned. Invalid result-kind combinations are rejected. + shared; scalar DAGs remain owned. Invalid result-kind combinations are rejected. - SQL NULL/cardinality rules, PromQL labels and evaluation times survive conversion. Implementation will require frontend, validation and plan-format migration for these explicit structural changes. DDL/DML, session commands, physical execution, new optimization algorithms, accuracy and execution-timing policy are outside this -proposal. The companion document defines the common pre-/post-ASAP operator graph. +proposal. The companion document defines the common pre-/post-ASAP operator DAG. [df-release]: https://github.com/apache/datafusion/releases/tag/55.1.0 [prom-release]: https://github.com/prometheus/prometheus/releases/tag/v3.15.0 diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 30bbe3f79..884226238 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -7,7 +7,7 @@ ## Goal and problem Use one operator model before and after ASAP optimization, so ordinary query -operations and summary operations can form one visible computation graph. +operations and summary operations can form one visible computation DAG. Today, the post-ASAP representation wraps relational subplans and duplicates some relational operators outside those wrappers. This causes three problems: @@ -19,7 +19,7 @@ relational operators outside those wrappers. This causes three problems: children. For example, consider a p99 latency query that projects its input columns, builds a -KLL summary, and projects the estimated result. The trees below read from the result +KLL summary, and projects the estimated result. The DAGs below read from the result at the top to the data source at the bottom: ```text @@ -33,7 +33,7 @@ Post-ASAP projection Project ``` Today the two projections need separate representations, and the scan is hidden -inside the wrapped subplan. In the proposed graph, both projections use the same +inside the wrapped subplan. In the proposed DAG, both projections use the same operator definition and the scan is directly visible. A union can likewise consume summary estimates without needing a separate post-ASAP union definition. @@ -56,7 +56,7 @@ operation variants and schema internals are expanded afterward. These declaratio are shared by the detailed sections, not separate abbreviated types. ```rust -// A graph node combines its operation with common planning properties (§2). +// A DAG node combines its operation with common planning properties (§2). struct OperatorNode { operator: Operator, result_kind: OperatorResultKind, @@ -72,14 +72,14 @@ enum Operator { } // NonASAPOp / ASAPOp: detailed below; their inputs are Rc. -// ScalarExpr: an owned value-expression tree, defined in the companion proposal. +// ScalarExpr: an owned, unshared value expression, defined in the companion proposal. // Schema / OperatorResultKind: defined in §2.1. ``` An operator owns its scalar expressions and references input nodes through `Rc`. Either operation category can consume the other's outputs when the input contract permits it. `NonASAP` describes one operation, not its entire -subgraph. Frontend graphs contain only NonASAP operations; ASAP optimization may +sub-DAG. Frontend DAGs contain only NonASAP operations; ASAP optimization may introduce state construction and readout. | Category | Meaning | All operations | @@ -199,13 +199,13 @@ A filter is an operator because it transforms a table. Its predicate, such as Scalar expressions belong to an operator field or a scalar query. Predicates, projection expressions and sort keys describe value computation in that context. Explicit scalar conversions and subqueries may reference operators; those are -visible graph dependencies with defined cardinality rules. This prevents an +visible DAG dependencies with defined cardinality rules. This prevents an arbitrary expression from being mistaken for a table-producing plan. The [companion proposal](decoupling_op_and_expr.md) defines this distinction. The companion's `ScalarExpr` uses `Rc` for `PromqlScalarFromVector`, `ScalarSubquery`, `Exists` and `InSubquery`, so those expressions already reference -this common graph before and after optimization. +this common DAG before and after optimization. In `scalar(sum(up))`, `scalar()` is Prometheus PromQL's built-in vector-to-scalar function, explicitly written by the query author. This proposal does not insert @@ -214,9 +214,9 @@ The scalar expression `PromqlScalarFromVector` represents that function and refe result to obtain one number. A valid ASAP rewrite may replace that producer with a summary readout, preserving the required vector and accuracy semantics; it cannot substitute raw summary state. Ordinary expressions such as `price * 2` reference -columns and literals, not a query subgraph. +columns and literals, not a query sub-DAG. -These are **query subgraphs referenced by scalar expressions**, with the same +These are **query sub-DAGs referenced by scalar expressions**, with the same producer identity as any other operator dependency. ### 1.3 Example: composing a logical DAG @@ -292,7 +292,7 @@ existing capability and rewrite checks permit that exact implementation. | Part of the design | Role in this example | |---|---| -| `OperatorNode` | Every graph node, holding its operation and common result/schema, guarantee and timing properties. | +| `OperatorNode` | Every DAG node, holding its operation and common result/schema, guarantee and timing properties. | | `Operator` | Selects the `NonASAP` or `ASAP` operation category in each node. | | `NonASAPOp` | Scan, filter, aggregate and projection before optimization; scan, filter and projection still use these definitions afterward. | | `ASAPOp` | Builds accumulator state and finalizes it after the rewrite. | @@ -306,7 +306,7 @@ For this example, assume `bytes` is nullable `Int64`. The output metadata is: | Aggregate before optimization | `Relation` | `sum_bytes: Plain(Int64)`, nullable | | Summary build after optimization | `State` | `sum_state: ExactAggregate(Sum, Sum)`, non-null accumulator state | | Finalize after optimization | `Relation` | `sum_bytes: Plain(Int64)`, nullable | -| Project in either graph | `Relation` | `total_bytes: Plain(Int64)`, nullable | +| Project in either DAG | `Relation` | `total_bytes: Plain(Int64)`, nullable | The empty accumulator finalizes to SQL NULL; the accumulator itself is state, not a nullable numeric value. The projection consumes the finalized column. Guarantees @@ -315,7 +315,7 @@ physical planning. The topmost Project node produces the query result. This illustrates the connection between the two proposals: scalar separation makes predicates and value expressions explicit; operator unification lets those -same ordinary operations consume ASAP results through normal graph edges. +same ordinary operations consume ASAP results through normal DAG edges. ### 1.4 Scope of operator sharing @@ -452,11 +452,11 @@ contract of `PromqlScalarFromVector` and other scalar plan reads. | Validation entry | Scope and stage | |---|---| -| `OperatorNode::validate_structure()` | Walks the reachable operator graph, including scalar plan references; checks input contracts, scalar typing and agreement between retained and derived output metadata. Valid for logical and physical plans; permits `timing = None`. | +| `OperatorNode::validate_structure()` | Walks the reachable operator DAG, including scalar plan references; checks input contracts, scalar typing and agreement between retained and derived output metadata. Valid for logical and physical plans; permits `timing = None`. | | `OperatorNode::validate_execution_timing()` | Includes structural validation, then requires assigned timing on every executable operator and checks phase dependencies. Used for executable physical candidates. | | Existing planner assessment and selection (#509) | Establishes guarantees using the existing accuracy models and checks them against request requirements and deployment capabilities. Neither node method re-proves a guarantee or decides request feasibility. | -The two node methods need only the graph and its annotations. Request requirements +The two node methods need only the DAG and its annotations. Request requirements and deployment models remain inputs to the existing planning/selection workflow, not implicit globals of `validate_structure`. Passing the timing check alone does not establish that a physical candidate satisfies the query's accuracy requirement. @@ -521,7 +521,7 @@ The representation must preserve the resulting execution constraints: ingestion- work cannot depend on query-time results, and consumers must receive values or state that are available when needed. Materialization choices, retention and plan selection remain governed by #509; this document does not define another lifecycle policy. -These constraints also apply to query subgraphs referenced by scalar expressions. +These constraints also apply to query sub-DAGs referenced by scalar expressions. PromQL evaluation timestamps and SQL statement time are separate from these execution phases. `TimeShift`, subquery grids and `EvalTimestamp` retain their @@ -534,11 +534,11 @@ The design is successful when: - A projection uses the same semantics above and below summary computations. - A union or another ordinary operator can consume summary estimates on its inputs. -- Unifying the representation preserves existing graph dependencies, including +- Unifying the representation preserves existing DAG dependencies, including any shared inputs; it does not introduce new sharing rules. - Existing value/state, accuracy and execution constraints remain enforceable on the unified representation. - Scalar expressions and conversions use the same representation before and after optimization, with no bridge nodes or hidden subplans. -- Structural and timing validation include query subgraphs referenced by scalar +- Structural and timing validation include query sub-DAGs referenced by scalar expressions; planner assessment includes their accuracy dependencies. diff --git a/docs/design_docs/proposals/univmon-frequency-summary.md b/docs/design_docs/proposals/univmon-frequency-summary.md index 21a4cb3d8..75cfd080b 100644 --- a/docs/design_docs/proposals/univmon-frequency-summary.md +++ b/docs/design_docs/proposals/univmon-frequency-summary.md @@ -30,11 +30,11 @@ The parameter contract records heap size, sketch rows, sketch columns and number of layers. Default dimensions define a candidate configuration, not an epsilon guarantee. A deployment accuracy model must supply calibrated evidence before an approximate readout can satisfy an accuracy target. Without -that evidence, the Planner keeps the exact subtree. HLL/Theta/KMV remain +that evidence, the Planner keeps the exact sub-DAG. HLL/Theta/KMV remain cardinality alternatives, and exact count remains the cheaper first count candidate. -All four readouts have the same unit-weight update, input subtree, grouping, +All four readouts have the same unit-weight update, input sub-DAG, grouping, window, parameter identity and state schema. Existing post-ASAP structural sharing can therefore intern their state producer while preserving distinct readout nodes. Sharing is only legal within the same execution/data scope. diff --git a/docs/develop_docs/asap-aware-mapping-architecture.md b/docs/develop_docs/asap-aware-mapping-architecture.md index f7e2a9a7b..5170f7d91 100644 --- a/docs/develop_docs/asap-aware-mapping-architecture.md +++ b/docs/develop_docs/asap-aware-mapping-architecture.md @@ -59,9 +59,9 @@ Terminology used in the diagram: those queries. **Pre-ASAP** means this logical input form, before the planner realizes an operation as a concrete ASAP realization; **post-ASAP** means the resulting realization form. -- A **DAG** (directed acyclic graph) represents query operators whose subtrees +- A **DAG** (directed acyclic graph) represents query operators whose sub-DAGs may be shared. **CSE** (common subexpression elimination) finds equivalent - subtrees and represents legal reuse by making them the same shared node. + sub-DAGs and represents legal reuse by making them the same shared node. Rust's `Rc` (reference-counted pointer) records that shared node identity. - A **target** is one replaceable site. A **candidate** is one valid alternative for it. `Replacement::Summary` is a constructed post-ASAP summary—maintained state @@ -120,7 +120,7 @@ flowchart TB ``` The generic `ReplacementStrategy` box is the extension point. The default -registry supplies summary realization, Hydra grouping, shared-subtree, +registry supplies summary realization, Hydra grouping, shared-sub-DAG, average-rewrite and exact-composition strategies. Section 3.3 describes the registries and the workload-derived roll-up rule. @@ -138,7 +138,7 @@ complete alternative set has been built. Use `search_workload` or `search_workload_with` for normal planner search. The search performs these steps: -1. Run CSE once to merge structurally identical subtrees that may legally be +1. Run CSE once to merge structurally identical sub-DAGs that may legally be shared. 2. Walk the complete DAG beneath every query root, including nodes below unshared parents. @@ -158,9 +158,9 @@ flowchart LR classDef common fill:#fff6dd,stroke:#b78922,color:#513d0c ROOTS["Input
one or more named QueryExpr roots"]:::workload - ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable subtrees"]:::workload + ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable sub-DAGs"]:::workload CSE --> WALK["Discover sites
walk the complete DAG, including nodes below unshared parents"]:::workload - WALK --> T["Build TargetSubDAG
retain the subtree's Rc identity and measured consumer_count"]:::workload + WALK --> T["Build TargetSubDAG
retain the sub-DAG's Rc identity and measured consumer_count"]:::workload T --> MATCH MATCH["matches(target)
cheaply decide whether this strategy has alternatives"]:::common MATCH -->|"true"| REPLACE["propose(target)
construct supported legal alternatives;
retain structured accuracy rejections"]:::common @@ -169,7 +169,7 @@ flowchart LR ``` `consumer_count` is workload information, not an estimate of runtime -executions. It matters to strategies such as `SharedSubtreeStrategy`, which +executions. It matters to strategies such as `SharedSubDAGStrategy`, which only has a share-versus-recompute choice when a target has multiple consumers. ### 3.2 Generate candidates through `ReplacementStrategy` @@ -205,7 +205,7 @@ The default context-free registry contains five `ReplacementStrategy` implementa realizations. Candidates are sized and ordered for the target's accuracy requirement; candidates without a sufficient guarantee are rejected before costing. -- `SharedSubtreeStrategy` uses `consumer_count` to identify shared targets. It +- `SharedSubDAGStrategy` uses `consumer_count` to identify shared targets. It emits both build-once-and-share and recompute-independently rewrites when a target has multiple consumers. - `HydraGroupingStrategy` proposes eligible shared multi-subpopulation layouts. diff --git a/docs/develop_docs/asap-aware-mapping-contracts.md b/docs/develop_docs/asap-aware-mapping-contracts.md index d1245366f..d7019db0e 100644 --- a/docs/develop_docs/asap-aware-mapping-contracts.md +++ b/docs/develop_docs/asap-aware-mapping-contracts.md @@ -28,7 +28,7 @@ For example, consider two top-level queries: - `sum by (service) (rate(m[5m]))` - `avg by (service) (rate(m[5m]))` -After `share_common_subtrees` merges their identical `rate(m[5m])` subtrees, both query trees point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. +After `share_common_sub_dags` merges their identical `rate(m[5m])` sub-DAGs, both query DAGs point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. Use: @@ -85,7 +85,7 @@ is a `Summary`; KLL (Karnin–Lang–Liberty) is a quantile-sketch algorithm. ```text compute independently vs. -reuse an already shared logical subtree +reuse an already shared logical sub-DAG ``` is represented as a `Rewrite`. @@ -253,7 +253,7 @@ bounds, but does not execute workloads or own deployment measurements. Most hook fn readout_extension(&self, ext_kind: &str, payload: &serde_json::Value, col: &ColumnRef) -> SketchQuery; ``` -- **`cse_recompute_cost`** — estimate the one-time cost of recomputing a CSE candidate's subtree independently at a single consumer. Default: `default_cse_recompute_cost`, a structural-size proxy. +- **`cse_recompute_cost`** — estimate the one-time cost of recomputing a CSE candidate's sub-DAG independently at a single consumer. Default: `default_cse_recompute_cost`, a structural-size proxy. ```rust fn cse_recompute_cost(&self, candidate: &CseCandidate) -> Cost; @@ -311,9 +311,9 @@ pub struct RankedTargetSubDAGCandidates<'a> { } ``` -`search_workload(roots)` runs the shared-subtree pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubtreeStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. +`search_workload(roots)` runs the shared-sub-DAG pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubDAGStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. -`CandidateLogicalASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubtreeStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks +`CandidateLogicalASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubDAGStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks may already have removed proposals before this boundary. In particular, `search_workload_with_targets` checks explicit per-root targets, while retaining direct DDSketch ratios with missing domain evidence and no root guarantee for @@ -379,7 +379,7 @@ The crate provides no default `Matcher` implementation because the answer depend Concretely, `explanation.rs` reports three candidate kinds from each `TargetSubDAGCandidates`: - `ExplanationKind::SketchApproximation` — the set contains a `Replacement::Summary` that realizes `SummaryFamilyType::Sketch(..)`, not just an exact/pass-through candidate. -- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubtreeStrategy`'s "build once and share" candidate (the `Replacement::Rewrite` whose `Rc` is the set's `target`). +- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubDAGStrategy`'s "build once and share" candidate (the `Replacement::Rewrite` whose `Rc` is the set's `target`). - `ExplanationKind::ExactComposition` — the candidate set contains an exact operation composed with a child target whose realization remains a coordinated choice. diff --git a/docs/develop_docs/extend-asap-aware-mapping.md b/docs/develop_docs/extend-asap-aware-mapping.md index 235da5460..c8c1b4973 100644 --- a/docs/develop_docs/extend-asap-aware-mapping.md +++ b/docs/develop_docs/extend-asap-aware-mapping.md @@ -65,7 +65,7 @@ fn matches(&self, target: &TargetSubDAG<'_>) -> bool { } ``` -is enough for the current shared-subtree strategy. +is enough for the current shared-sub-DAG strategy. #### Guideline @@ -293,9 +293,9 @@ aggregate choices remain independent. --- -### Example: current `SharedSubtreeStrategy` +### Example: current `SharedSubDAGStrategy` -`SharedSubtreeStrategy` is the reference implementation for a logical rewrite strategy. +`SharedSubDAGStrategy` is the reference implementation for a logical rewrite strategy. It applies when: @@ -331,9 +331,9 @@ That preference belongs to the cost model. share-versus-recompute candidate pair. The strategy still returns both alternatives because enumeration and ranking are separate steps: -- `consumer_count >= 2` means `share_common_subtrees` has already merged the expression into one shared `Rc`. The shared alternative is therefore an `Rc::clone`; the independent alternative requires a deep clone. +- `consumer_count >= 2` means `share_common_sub_dags` has already merged the expression into one shared `Rc`. The shared alternative is therefore an `Rc::clone`; the independent alternative requires a deep clone. - `cse_share_decision` is used by the ranking path, not by - `SharedSubtreeStrategy`. + `SharedSubDAGStrategy`. - The strategy must return both valid alternatives even if the current cost model strongly prefers one. A future whole-plan search may choose differently from today's local comparison. This example is useful when implementing transformations such as: @@ -462,7 +462,7 @@ assert!( For a strategy whose explanation includes important context, also test that context. -For example, the shared-subtree tests verify that the consumer count appears in the rationale. +For example, the shared-sub-DAG tests verify that the consumer count appears in the rationale. --- @@ -470,7 +470,7 @@ For example, the shared-subtree tests verify that the consumer count appears in For logical rewrites, test the structural property that distinguishes the alternatives. -For example, the current shared-subtree tests verify: +For example, the current shared-sub-DAG tests verify: ```rust Rc::ptr_eq(shared, &q) @@ -660,7 +660,7 @@ This complements `realize_extension`: realization defines what gets maintained; #### `cse_recompute_cost` -Use to estimate the cost of computing a common subtree independently at each consumer. +Use to estimate the cost of computing a common sub-DAG independently at each consumer. ```rust fn cse_recompute_cost( @@ -673,7 +673,7 @@ fn cse_recompute_cost( #### `cse_shared_maintenance_cost` -Use to estimate the cost of computing and maintaining a shared subtree. +Use to estimate the cost of computing and maintaining a shared sub-DAG. ```rust fn cse_shared_maintenance_cost( diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index b1d1226b0..7bfc33631 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -302,7 +302,7 @@ pass. An omitted strategy contributes no proposals of its own. | --- | --- | --- | | `SketchAlgorithmStrategy::new(&model)` | Enumerates supported exact/sketch implementations and parameter choices for aggregate targets | Yes | | `HydraGroupingStrategy::new(&model)` | Considers a shared multi-subpopulation structure for supported grouped sketch families, subject to accuracy evidence | Yes | -| `SharedSubtreeStrategy` | Proposes sharing versus independent recomputation at reused subtrees | Yes | +| `SharedSubDAGStrategy` | Proposes sharing versus independent recomputation at reused sub-DAGs | Yes | | `SemanticEquivalentRewriteStrategy` | Proposes supported equivalent aggregate rewrites, including decomposing average into sum/count | Yes | | `ExactCompositionStrategy::new(&model)` | Proposes supported exact operations around summary readouts or in maintenance | Yes | | Your `ReplacementStrategy` implementation | Adds domain-specific legal replacement proposals | No | @@ -353,7 +353,7 @@ use asap_types::workload::{ }; use asap_aware_mapping::{ search_workload_with_targets, DefaultAccuracyModel, DefaultCostModel, - ReplacementStrategy, SketchAlgorithmStrategy, SharedSubtreeStrategy, + ReplacementStrategy, SketchAlgorithmStrategy, SharedSubDAGStrategy, }; use asap_types::types::AccuracyTarget; @@ -387,7 +387,7 @@ fn main() -> Result<(), Box> { let model = DefaultCostModel; let strategies: Vec> = vec![ Box::new(SketchAlgorithmStrategy::new(&model)), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), ]; let space = search_workload_with_targets( vec![("q1", root, Some(accuracy))], &strategies, &DefaultAccuracyModel, @@ -701,7 +701,7 @@ in the workload**. Here, “global” describes that cross-target scope. It does mean a proven globally optimal solution over every possible physical plan, nor selection across every machine in a deployment. -Consider this conceptual dependency graph: +Consider this conceptual dependency DAG: ```text Q1 --+ @@ -752,7 +752,7 @@ GlobalSelection::assemble_selected_dag(&self, target: &Rc) ``` For structural inspection only, this complete example selects a semantic root -and exports its inspection graph. It performs no lifecycle or deployment planning. +and exports its inspection DAG. It performs no lifecycle or deployment planning. Use lifecycle-aware selection above when the comparison needs those decisions. ```rust @@ -795,8 +795,8 @@ fn main() -> Result<(), Box> { let selection = space.global_selection(&DefaultCostModel); // Search may canonicalize roots; use the root returned by CandidateLogicalASAPDAGs. if let Some(summary) = selection.assemble_selected_dag(&space.roots[0].1)? { - let graph = asap_types::dag_export::export_summary(&summary); - println!("{graph:#?}"); + let dag = asap_types::dag_export::export_summary(&summary); + println!("{dag:#?}"); } Ok(()) } @@ -806,14 +806,14 @@ fn main() -> Result<(), Box> { | Function/type | Purpose | | --- | --- | -| `asap_types::dag_export::export(&query)` | Pre-ASAP inspection graph | -| `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection graph | +| `asap_types::dag_export::export(&query)` | Pre-ASAP inspection DAG | +| `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection DAG | | `asap_types::post_asap::compile_post_asap_dag(&root)` | Compile a semantic DAG with execution-data-state validation; not a physical plan | | `PostAsapDagDocument::new(dag)` and `.validate()` | Versioned semantic envelope and explicit validation; constructing it alone does not validate | -| `asap_aware_mapping::export_summary_maintenance_plan(&plan)` | Graph plus lifecycle deployments, alternatives and available cost/guarantee information | +| `asap_aware_mapping::export_summary_maintenance_plan(&plan)` | DAG plus lifecycle deployments, alternatives and available cost/guarantee information | | `explain_replacements` / `explain_replacements_with` | Findings from default/custom-strategy search; not a complete physical feasibility report | -Choose the export matching your intended handoff: an inspection graph is not +Choose the export matching your intended handoff: an inspection DAG is not interchangeable with a versioned execution contract. Preserve lifecycle and cost/guarantee evidence needed downstream instead of exporting only a bare DAG. For public symbol details, build local API documentation with: diff --git a/docs/develop_docs/metrics-observability-corpora.md b/docs/develop_docs/metrics-observability-corpora.md index 3dd1d2b8f..d127e92ce 100644 --- a/docs/develop_docs/metrics-observability-corpora.md +++ b/docs/develop_docs/metrics-observability-corpora.md @@ -62,7 +62,7 @@ query root. It does not measure workload-wide search or the other default strategies. The default workload search currently registers `SketchAlgorithmStrategy`, -`HydraGroupingStrategy`, `SharedSubtreeStrategy`, and +`HydraGroupingStrategy`, `SharedSubDAGStrategy`, and `AvgToSumOverCountStrategy`. Workload context can additionally contribute `RollupStrategy` and `AccuracyReconciliationStrategy`. This baseline is therefore a sketch-only comparison point. diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md index aa1a52b63..d263b7825 100644 --- a/docs/develop_docs/native-promql-inputs.md +++ b/docs/develop_docs/native-promql-inputs.md @@ -48,7 +48,7 @@ tests, not proof of Backend candidate selection or durable deployment execution. Spatial heap candidates use the same complete series identity. Planner's `current_series_topk_candidates` explores a CountSketch-with-heap realization -of canonical Sort/Limit under an explicit accuracy target. The physical graph +of canonical Sort/Limit under an explicit accuracy target. The physical DAG selects the latest eligible samples before building a fresh heap. A maintained population boundary can supply that snapshot directly. Arbitrary signed metric values do not authorize CMS; counter Rate's non-negative proof is separate. diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 3fa54fb4e..6d96181c1 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -29,7 +29,7 @@ Status values: | # | Backend site | Computation | Planner node | Status at #475 | Notes | |---|---|---|---|---|---| -| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { QueryExpr }` subtrees plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | +| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` DAG for a native query | `Fallback { QueryExpr }` sub-DAGs plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | | 2 | `QueryTimeOperator::Aggregate` (sum/min/max/avg/count) | Grouped value aggregation | `Value::Exact(Aggregate)`; `SummaryAgg{ExactAggregate, Reduce}` over finalized values | Supported | Also `promql_values::compile_aggregate`. | | 3 | `QueryTimeOperator::Sort`, `Limit` (topk, sort, sort_desc) | Ordering and per-group limits | `Value::Sort`, `Value::Limit` | Supported | | | 4 | `QueryTimeOperator::Binary`, `QueryPlanNode::Binary` (vector ⊗ scalar) | Arithmetic with a scalar operand | `Binary` whose operand is `Fallback{PromqlScalarBridge(Literal)}` | Missing | Query-time `Binary` accepts only label-map vector schemas. The literal node has no native binding. | @@ -188,7 +188,7 @@ Totals after this change: 20 Supported, 5 Partial, 4 Missing, 2 Backend. An argument whose output provably lacks `le`, such as `sum by (job) (rate(x_bucket[5m]))`, is rejected at lowering. Prometheus returns an empty vector for it. Candidate search keeps the classic form as one -exact `KeepPreAsap` subtree for every accuracy target; it has no sketch +exact `KeepPreAsap` sub-DAG for every accuracy target; it has no sketch candidate. `histogram_quantiles` lowers each branch the same way; the Fallback compiler accepts its `Concat` of relabeled branches and rejects duplicate output label sets. Nested aggregation, such as diff --git a/docs/develop_docs/planner-vocabulary-migration.md b/docs/develop_docs/planner-vocabulary-migration.md index f5f7f70ae..ea46e992c 100644 --- a/docs/develop_docs/planner-vocabulary-migration.md +++ b/docs/develop_docs/planner-vocabulary-migration.md @@ -36,8 +36,8 @@ names. | `SketchAlgorithmStrategy::with_models_and_evidence` | `SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence` | | `HydraGroupingStrategy::with_models_and_evidence` | `HydraGroupingStrategy::new_with_planning_inputs_and_evidence` | -For example, `Binder::new().bind(&tree)` becomes -`SchemaResolver::new().resolve_schema(&tree)`. Cost-model implementations that +For example, `Binder::new().bind(&dag)` becomes +`SchemaResolver::new().resolve_schema(&dag)`. Cost-model implementations that accept or return `Implementation` now use `Realization`; variants and ranking contracts remain the same. diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index 749f97cf9..dcb3597e2 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -41,7 +41,7 @@ to one source language. - [`Concat`](#concat) — exact, untyped `UNION ALL` of union-compatible branches. **[PromQL-specific nodes](#promql-specific-nodes)** -- [`PromqlScalarBridge`](#promqlscalarbridge) — a scalar sub-expression at an operator-tree position. +- [`PromqlScalarBridge`](#promqlscalarbridge) — a scalar sub-expression at an operator-DAG position. - [`EvalTimestamp`](#evaltimestamp) — the query evaluation time as a scalar (PromQL `time()`). - [`PromqlVectorFromScalar`](#promqlvectorfromscalar) — promotes a scalar to a label-less instant vector. - [`PromqlScalarFromVector`](#promqlscalarfromvector) — collapses a single-series vector to a scalar. @@ -174,7 +174,7 @@ meaningful summary implementation. `measures[i]`; groups are still formed from every row. It is positional against `child`'s output (like `Filter.pred`), not against the aggregate's output like `having`. Empty means no measure is filtered; that is the only spelling of "unfiltered" a resolved - tree carries, so `[None, None]` is normalized to `[]`. + DAG carries, so `[None, None]` is normalized to `[]`. - `having` — an optional post-aggregation filter predicate (SQL `HAVING`). - `child` — the input being aggregated. @@ -241,7 +241,7 @@ Example for `having`: ) ``` - but the tree can still carry the same condition as a wrapping `Filter` near the root: + but the DAG can still carry the same condition as a wrapping `Filter` near the root: ```text Filter( @@ -329,7 +329,7 @@ the same logical data domain. - `predicates` — row-level filters pushed all the way down to this scan (Rules/Invariants rule 1); enforced structurally at lowering time — a `Filter` directly over a `Scan` never survives. -- `schema` — the binding schema every positional column reference in the tree resolves against. +- `schema` — the binding schema every positional column reference in the DAG resolves against. ### Filter @@ -487,7 +487,7 @@ histogram_quantiles(rate(http_request_duration_seconds_bucket[5m]), "le", 0.5, 0 ### PromqlScalarBridge A scalar sub-expression (issue #220: in practice always `Literal(ScalarValue::Float64(_))` — -a PromQL number literal, or a folded constant scalar expression) sitting at an **operator-tree +a PromQL number literal, or a folded constant scalar expression) sitting at an **operator-DAG position** — a `BinaryOp` operand for ` op ` thresholds and unit conversions, a `PromqlVectorFromScalar` child, or a whole query's root. This wrapper is what marks the position; it no longer duplicates `Literal`'s value the way the old `PromqlScalar(f64)` variant diff --git a/docs/user_guide_docs/run-a-query.md b/docs/user_guide_docs/run-a-query.md index af2cfc07e..3f5825c29 100644 --- a/docs/user_guide_docs/run-a-query.md +++ b/docs/user_guide_docs/run-a-query.md @@ -1,6 +1,6 @@ # ASAPPlanner CLI user guide -Use the `asap-devtools` commands to inspect query IR, export graphs, and inspect +Use the `asap-devtools` commands to inspect query IR, export DAGs, and inspect corpus coverage. These commands do not deploy or execute a physical plan. To develop an application using the Rust library, start with @@ -17,8 +17,8 @@ tool on first use. | --- | --- | --- | | `show_pre_asap_ir --data-ingestion-interval-ms 1000 queries.txt` | File path, or stdin when omitted | Prints canonical Pre-ASAP IR | | `show_post_asap_ir --data-ingestion-interval-ms 1000 queries.txt` | Same query file format | Prints all sketch-strategy Post-ASAP candidates using a fixed approximate target, in cost-model order | -| `dag_export --data-ingestion-interval-ms 1000 --promql ""` | One PromQL expression | Exports a query graph for inspection | -| `dag_export --sql ""` | One SQL expression using the tool's catalog | Exports a query graph for inspection | +| `dag_export --data-ingestion-interval-ms 1000 --promql ""` | One PromQL expression | Exports a query DAG for inspection | +| `dag_export --sql ""` | One SQL expression using the tool's catalog | Exports a query DAG for inspection | | `analyze_corpora --corpora --data-ingestion-interval-ms 1000 --out-dir ` | Repository PromQL corpora, output directory | Writes successful/error IR dumps and summary reports | | `analyze_corpora --sql-corpora --out-dir ` | Repository SQL corpora, output directory | Writes SQL corpus reports | | `variant_coverage --data-ingestion-interval-ms 1000` | Repository corpora | Reports Pre-ASAP IR variant coverage | diff --git a/old_docs/README.md b/old_docs/README.md index d29667e26..0989e65b4 100644 --- a/old_docs/README.md +++ b/old_docs/README.md @@ -18,7 +18,7 @@ are still planned. | Layer | What | Where it lives | |---|---|---| | 1 | Query-language parsing (PromQL, SQL; DataFusion, ElasticDSL planned) | `asap-frontend-promql`, `asap-frontend-sql` | -| 2 | Per-language relational algebra tree + the shared L2→L3 converter (incl. the post-lowering canonicalization pass both languages run through) | `asap-l2` (emitted by the front ends) | +| 2 | Per-language relational algebra DAG + the shared L2→L3 converter (incl. the post-lowering canonicalization pass both languages run through) | `asap-l2` (emitted by the front ends) | | 3 | Intent algebra defined by ASAPPlanner itself — query language and runtime independent IR (intent only — no summary type, no summary params). | `asap-ir::intent_algebra` | | 4 | Cost-aware optimizer — CSE + pluggable cost model + the summary-vs-exact accuracy decision (`AggIntent → SummaryKind`, landed). Produces the **summary-bound** IR (kind + params committed), plus the serving-time `SummaryExecutor` interface that answers a query against it. | binding/optimizer passes in `asap-plan`; summary-bound IR + serving-time executor in `asap-sketch` | | 5 | Physical runtime / Data plane — The planner emits configurations to physical runtime, with the execution environment consider the parallelism, hardware types, lifecycle stages, distributed workers | The implementation of this layer should be in different downstream application repos. | @@ -43,7 +43,7 @@ runtime coupling. Two isolation wins fall out of this: - **The front ends quarantine their parsers** — a caller that needs only PromQL depends on `asap-frontend-promql` and never compiles DataFusion, and vice-versa (verified with `cargo tree`). `asap-lower` is the facade for callers that want both. -- **L3-only consumers skip the L2 machinery** — `asap-sketch` (L4 types) depends on `asap-ir` alone, and `asap-plan` (L3→L4 binding + optimizer) adds only `asap-sketch` on top of that — neither pulls the L2 relational tree, the converter, or the binder (`asap-l2`). Only the front ends, which actually *lower* queries, need `asap-l2`. +- **L3-only consumers skip the L2 machinery** — `asap-sketch` (L4 types) depends on `asap-ir` alone, and `asap-plan` (L3→L4 binding + optimizer) adds only `asap-sketch` on top of that — neither pulls the L2 relational DAG, the converter, or the binder (`asap-l2`). Only the front ends, which actually *lower* queries, need `asap-l2`. ### Directory structure @@ -62,7 +62,7 @@ crates/ │ └── names.rs # BindingName / QueryId ├── l2/ # asap-l2 — L2 relational algebra + L2→L3 converter │ └── src/ -│ ├── relational.rs # L2: per-language relational tree the front ends emit +│ ├── relational.rs # L2: per-language relational DAG the front ends emit │ ├── lower.rs # L2→L3 converter (convert_root) │ ├── canonicalize.rs # shared post-lowering normalization (heavy-hitter TopK, #34) │ ├── binder.rs # positional name-resolution seed @@ -119,7 +119,7 @@ cargo test --workspace one or more `--sql`/`--promql` queries through L1→L2→L3 and flattens each resulting `QueryExpr` — the L3 canonical IR (see [`docs/l2-intent-algebra.md`](docs/l2-intent-algebra.md)) — into a generic -node/edge graph, printed as JSON: +node/edge DAG, printed as JSON: ```bash cargo run -p asap-lower --example dag_export -- \ @@ -130,7 +130,7 @@ cargo run -p asap-lower --example dag_export -- \ `--name` is optional (defaults to `q`); repeat `--sql`/`--promql` to pack several queries into one file. Each node carries a `kind`, a short `label`, its own fields under `detail`, `children` ids, and a bottom-up structural -`hash` so identical subtrees (e.g. a `Scan` shared across two queries) can be +`hash` so identical sub-DAGs (e.g. a `Scan` shared across two queries) can be spotted by comparing hashes. The query above exports as: ```json @@ -138,7 +138,7 @@ spotted by comparing hashes. The query above exports as: "queries": [ { "name": "q1", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(http_requests_total)", "children": [], "detail": { "source": { "TimeSeries": { "metric": "http_requests_total" } }, "predicates": [], "schema": { "...": "..." } } }, { "id": 1, "kind": "TimeRange", "label": "TimeRange(300s)", "children": [0], "detail": { "range": { "secs": 300, "nanos": 0 } } }, @@ -161,4 +161,4 @@ loads) — to browse it as an interactive DAG: click a node for its full `detail` in a side panel, switch between queries via tabs, and toggle highlighting of structurally-identical nodes shared across queries. See [`tools/dag-viewer/README.md`](tools/dag-viewer/README.md) for the shared- -subtree-highlighting caveat (it's a client-side hash proxy, not real CSE). +sub-DAG-highlighting caveat (it's a client-side hash proxy, not real CSE). diff --git a/old_docs/docs/design.md b/old_docs/docs/design.md index 04a78f6d5..0e5ab4ad0 100644 --- a/old_docs/docs/design.md +++ b/old_docs/docs/design.md @@ -27,7 +27,7 @@ query string ▼ per-language AST/DAG, unresolved column references ← L1 │ pass': resolve (bind names to schema positions, substitute them - │ throughout the already-canonical-shaped tree) + │ throughout the already-canonical-shaped DAG) ▼ │ pass'': canonicalize (cross-language / cross-phrasing pattern │ normalization — e.g. promoting a generic @@ -75,7 +75,7 @@ language, or differently-phrased within the same language — converge on the identical canonical shape. `canonicalize` catches the cases where a language has no dedicated syntax for an intent — e.g. SQL's `ORDER BY count DESC LIMIT k` has no `topk()`-shaped AST node; a -pattern-detection pass over the already-assembled tree recognizes it as +pattern-detection pass over the already-assembled DAG recognizes it as the same `Aggregate{aggs:[TopK]}` shape PromQL's dedicated `topk()` produces directly in pass 1. @@ -93,14 +93,14 @@ flowchart LR ## L2 — intent algebra -**Job: define the canonical intent tree's vocabulary — the shape L1's +**Job: define the canonical intent DAG's vocabulary — the shape L1's passes produce — expressed declaratively (e.g. "a quantile to this accuracy," "the top-k by this ranking"), with implementation strategy left to L3.** L2 is the vocabulary/rule set L1's output conforms to, enforced by construction: every front end's output runs through the same `resolve`+`canonicalize`. - The result is a language- and deployment-independent canonical - intent tree: what to compute, without committing to how. Deployment + intent DAG: what to compute, without committing to how. Deployment here refers to a physical execution context — e.g. parallelism and the lifecycle stage a computation runs at — a different sense of "deployment" than the Glossary's "Deployment model" entry below; @@ -111,12 +111,12 @@ output runs through the same `resolve`+`canonicalize`. sub-computations are properties of this canonical form. Both depend on L1's `canonicalize` pass having already converged semantically-equivalent queries onto the same shape: only - structurally-identical sub-trees can be recognized as the same + structurally-identical sub-DAGs can be recognized as the same reusable computation. ```mermaid flowchart LR - L1T["canonical intent tree\n(from L1)"] --> V["intent vocabulary\n+ design rules"] + L1T["canonical intent DAG\n(from L1)"] --> V["intent vocabulary\n+ design rules"] V --> L2G["governs what a valid\nL1 output looks like"] ``` diff --git a/old_docs/docs/l1-query-language.md b/old_docs/docs/l1-query-language.md index 9d2530505..7b3ccd7a1 100644 --- a/old_docs/docs/l1-query-language.md +++ b/old_docs/docs/l1-query-language.md @@ -1,7 +1,7 @@ # L1 — query string → canonical intent algebra Per-language front ends, one per supported query language. Each turns a -raw query string all the way into the canonical intent tree that L2 +raw query string all the way into the canonical intent DAG that L2 (see [`l2-intent-algebra.md`](./l2-intent-algebra.md)) is the vocabulary for. That journey is **two passes**, and this doc is organized around that split (see `design.md`'s representation/pass table for the @@ -36,7 +36,7 @@ language A's native AST language B's native AST │ interpret directly into │ interpret directly into │ the canonical shape │ the canonical shape ▼ ▼ - canonical-shaped tree, named column references + canonical-shaped DAG, named column references ═══════════════ pass 1 done; pass 2 starts ═══════════════ │ one shared pass, for every language: │ 1. resolve — bind names to schema @@ -44,7 +44,7 @@ language A's native AST language B's native AST │ 2. canonicalize — cross-language / │ cross-phrasing normalization ▼ - canonical intent tree ← L1's output, L2's vocabulary + canonical intent DAG ← L1's output, L2's vocabulary columns are positions + a self-contained schema on every scan ``` @@ -52,7 +52,7 @@ language A's native AST language B's native AST ## Pass 1 — interpret Different front ends can look nothing alike where they start — one may -begin from a bare AST, another from an already-planned tree — but every +begin from a bare AST, another from an already-planned DAG — but every front end produces the same canonical shape as its output, through the same node vocabulary, regardless of source language. @@ -66,10 +66,10 @@ to positions is pass 2's job, once a schema exists to resolve against. ### The nesting contract -Nesting — an operator tree inside a source clause or a function +Nesting — an operator DAG inside a source clause or a function argument, an aggregate over an aggregate, a binary operation over two -subtrees, a range function over a sub-query, and so on — is required to -lower **structurally**: the canonical intent tree is a recursive, +sub-DAGs, a range function over a sub-query, and so on — is required to +lower **structurally**: the canonical intent DAG is a recursive, arbitrarily-nestable structure, so "an operation over a sub-query" needs no special-cased IR shape. Any language whose grammar allows nesting must be able to express it this way, using the same recursive @@ -88,7 +88,7 @@ language — does two things in order: 1. **Resolve.** Bind names to schema positions (see "Name resolution: binding" below), then substitute every column reference throughout - the tree — a generic walk over the already-canonical-shaped tree + the DAG — a generic walk over the already-canonical-shaped DAG pass 1 produced. 2. **Canonicalize.** Run a cross-language normalization pass so that semantically equivalent queries, from any supported language or @@ -102,12 +102,12 @@ language — does two things in order: dedicated `topk(k, count_over_time(…))` produces directly in pass 1. This ordering matters: canonicalization operates on an already-resolved -tree, so its pattern-matching rules work over stable schema positions, +DAG, so its pattern-matching rules work over stable schema positions, not per-language surface syntax. ### Name resolution: binding -Binding walks the canonical-shaped, named-reference tree pass 1 +Binding walks the canonical-shaped, named-reference DAG pass 1 produced and derives one self-contained schema that every column reference indexes into: it seeds columns from whatever catalog is available (or a minimal always-present floor if the catalog knows @@ -119,7 +119,7 @@ inline while interpreting, in pass 1: - **Everything downstream becomes purely structural and total.** Take those two words literally: *structural* means later steps match on - tree shape and column position, never on a name string again — the + DAG shape and column position, never on a name string again — the string comparisons all happened once, here. *Total* is the computer-science sense — a function defined for every input, with no case left unhandled — applied to column resolution: once binding has @@ -136,7 +136,7 @@ inline while interpreting, in pass 1: complement is represented and deferred to serving time, rather than resolved eagerly. -Each independent sub-tree (e.g. either side of a binary operation) +Each independent sub-DAG (e.g. either side of a binary operation) binds against its own schema, since the two sides may reference entirely different sources — but a side must still see names referenced by an *enclosing* operation (a grouping key mentioned above, @@ -209,7 +209,7 @@ pub trait SchemaCatalog { pub struct Binder { .. } impl Binder { - pub fn bind(&self, tree: &QueryExpr) -> Schema; + pub fn bind(&self, dag: &QueryExpr) -> Schema; } ``` @@ -225,7 +225,7 @@ with the default: ```rust // PromQL: `sum by (job) (http_requests_total)`, no catalog available. -let schema = Binder::default().bind(&tree); +let schema = Binder::default().bind(&dag); // -> Schema { columns: [ts, value, job], time_index: Some(0), closed: false } // ("job" was seeded because the query references it; anything the // query never mentions is simply absent from this schema) @@ -243,7 +243,7 @@ impl SchemaCatalog for SqlCatalog { .. } } -let schema = Binder::with_catalog(SqlCatalog { .. }).bind(&tree); +let schema = Binder::with_catalog(SqlCatalog { .. }).bind(&dag); // -> Schema { columns: [host, bytes], time_index: None, closed: true } // (the catalog's declared columns are used verbatim, regardless of // which ones the query actually references) @@ -254,10 +254,10 @@ migration tracked in #179, since it already operates on the canonical type either way: ```rust -pub fn canonicalize(tree: QueryExpr) -> QueryExpr; +pub fn canonicalize(dag: QueryExpr) -> QueryExpr; ``` -Idempotent, bottom-up: rewrites a tree already in canonical shape into +Idempotent, bottom-up: rewrites a DAG already in canonical shape into its normal form (e.g. promoting a generic `Limit{Sort{Aggregate([Count])}}` shape to the explicit heavy-hitter `Aggregate{aggs:[TopK]}` form) — same type in, same type out. diff --git a/old_docs/docs/l2-intent-algebra.md b/old_docs/docs/l2-intent-algebra.md index 1b09a7a10..ad280534b 100644 --- a/old_docs/docs/l2-intent-algebra.md +++ b/old_docs/docs/l2-intent-algebra.md @@ -5,9 +5,9 @@ this layer — expressing what to compute; summary type, summary parameters, and physical operator choice are committed later, entirely at L3's discretion (see [`l3-summary-bound-ir.md`](./l3-summary-bound-ir.md)). Data-model -agnostic by design: the same tree shape covers both time-series-style +agnostic by design: the same DAG shape covers both time-series-style sources and tabular sources, through a source's data model living on -its scan node rather than forking the tree type. +its scan node rather than forking the DAG type. ## Design rules @@ -31,13 +31,13 @@ its scan node rather than forking the tree type. 3. **Intent at L2, summary at L3.** An intent like "a quantile to this accuracy" says what to compute; it never says which summary computes it. -4. **A DAG, not a tree.** A producer can have more than one consumer +4. **A DAG with sharing.** A producer can have more than one consumer sharing its computed result — from explicit naming in the source query (e.g. a named sub-query), from a later layer's decision to reuse a shared computation across two different queries, or from physical structure (a pre-aggregate feeding several downstream consumers). The canonical form supports this by construction, via a - binding/reference pair of node kinds — a tree-only representation + binding/reference pair of node kinds — a representation without sharing would have to duplicate the producer per consumer and lose the sharing. @@ -60,10 +60,10 @@ aggregate" as syntactic shapes — see design rule 1; both fold into composition of the generic operators instead, except for the one strategy-driven exception noted there. -**Why a scan's source is a variant, not a separate tree type per data +**Why a scan's source is a variant, not a separate DAG type per data model.** Most operators (filter, aggregate, …) have identical semantics regardless of whether the input is a time-series window or a table scan -— only the leaf differs. One tree with a polymorphic leaf lets +— only the leaf differs. One DAG with a polymorphic leaf lets data-model-agnostic rules apply uniformly across every data model; rules that do care about data-model specifics gate on that leaf's declared kind. Summaries themselves are meant to be data-model-agnostic @@ -104,8 +104,8 @@ unique key, a designated time column — it is a position into a schema, not a string. A pre-bind string name means nothing without knowing which schema it resolves against and at what offset; a position is settled once, at bind time, and by construction can't fail to resolve for any -later pass. This also makes the canonical tree self-describing: every -scan carries its own schema, so any sub-tree's output schema is +later pass. This also makes the canonical DAG self-describing: every +scan carries its own schema, so any sub-DAG's output schema is computable purely from its inputs, without external context. A named alias for the column-identity type (rather than a bare integer) is still kept, so code that touches it can express "this is a column @@ -113,15 +113,15 @@ position" as a distinct kind of value, not just any number. ## Schema flow -Every edge in the canonical tree carries a schema: its columns, which +Every edge in the canonical DAG carries a schema: its columns, which column (if any) is the designated time index, its unique keys, and whether it's closed (a complete enumeration) or open (a runtime row may -carry more). The tree is locally type-checked: a node's output schema +carry more). The DAG is locally type-checked: a node's output schema is a pure function of its inputs' schemas and its own parameters, -verifiable without consulting the rest of the tree. +verifiable without consulting the rest of the DAG. Three distinct kinds of schema-shaped information are easy to conflate -and shouldn't be: the schema flowing along the canonical tree's own +and shouldn't be: the schema flowing along the canonical DAG's own edges; the schema of the underlying data source itself (what tables or metrics actually exist, consulted only during L1's internal name resolution); and a catalog of what summaries exist and what they can @@ -218,7 +218,7 @@ pub enum QueryExpr { ``` `reduction` is a field *on* the `Aggregate` variant itself — not a -separate node in the tree, and not something any other variant carries. +separate node in the DAG, and not something any other variant carries. It answers a question only `Aggregate` ever needs to ask: is this node collapsing rows at all, and if so, by which (possibly empty) key set — or does it have no grouping concept to begin with. Making that an @@ -230,7 +230,7 @@ handling downstream — see [`l3-summary-bound-ir.md`](./l3-summary-bound-ir.md# actually load-bearing. The two small types that field's shape turns on, in full — neither is a -tree node either; both are plain data reachable only through +DAG node either; both are plain data reachable only through `Aggregate.reduction`, and `GroupKeys` only exists at all when `reduction` is `Reduce` (it's meaningless for `PerEntity`, which is exactly why it isn't a sibling field instead): @@ -378,7 +378,7 @@ for its particular statistic, which is exactly the design rule 1 exception (a genuinely different computational access pattern, not an ordinary composition of existing operators). -And the schema every edge in the tree carries: +And the schema every edge in the DAG carries: ```rust pub struct Schema { diff --git a/old_docs/docs/l3-summary-bound-ir.md b/old_docs/docs/l3-summary-bound-ir.md index a86d9b9cd..ae59815e8 100644 --- a/old_docs/docs/l3-summary-bound-ir.md +++ b/old_docs/docs/l3-summary-bound-ir.md @@ -23,7 +23,7 @@ answer it — with no reference to what's actually stored anywhere yet. ### The shape of a summary-bound plan -A summary-bound plan mirrors the canonical intent tree but replaces +A summary-bound plan mirrors the canonical intent DAG but replaces each summarizable node with a decision: - **Unbound / logical** — nothing rewrote this node (e.g. a filter or a @@ -43,11 +43,11 @@ each summarizable node with a decision: - **Summary merge** — combining multiple built summaries into one; only valid when every input agrees on family and parameters. -A summary-bound plan is a DAG, not just a tree — a shared +A summary-bound plan is a DAG with sharing — a shared sub-computation can appear as more than one reference to the same bound node, mirroring the sharing already present in the canonical form (see [`l2-intent-algebra.md`](./l2-intent-algebra.md)). Binding only fires -where an intent is recognizable in the tree; anything underneath an +where an intent is recognizable in the DAG; anything underneath an unrewritten (logical) parent stays logical too — extending `implement` to rewrite through a logical parent is a known open design question, left to a deployment that has a real need for it. @@ -79,17 +79,17 @@ a deployment supplies: how to find candidate materialized instances for a bound node, how to fetch one instance's state, how to merge several instances' states, how to read a value out of a state, and how to evaluate an unrewritten logical node directly. A shared execution -routine walks the bound tree and calls into these deployment-supplied +routine walks the bound DAG and calls into these deployment-supplied operations, so every deployment gets the same walk and merge logic for free. Finding candidates for a summary-aggregation node requires seeing that -node's entire input sub-tree, not just a bare name — a summary +node's entire input sub-DAG, not just a bare name — a summary aggregation node carries no source identity of its own; that identity lives further down, inside a scan. Walking down to find it is deployment-specific knowledge (how a real store names and indexes summarized data); the shared execution routine doesn't need to -interpret it, only pass the sub-tree along. +interpret it, only pass the sub-DAG along. ### Nested composition @@ -186,7 +186,7 @@ reaches serving-time execution: | Node | `summary` field | Constraint | |---|---|---| -| `Logical` | — | none — wraps an arbitrary unrewritten `QueryExpr` subtree | +| `Logical` | — | none — wraps an arbitrary unrewritten `QueryExpr` sub-DAG | | `SummaryAgg` | own | none beyond `col`'s intent already requiring an accuracy target compatible with `summary`; this is the leaf every other row's constraints are checked *against* | | `SummaryJoin` | own | only emitted by a join-specific `Bind*OnJoin` rule — none exist yet in the base design, so this variant has no live producer | | `SummarySubtract` | read from `left`/`right` | `left`/`right` agree on `(kind, params)`; that kind's catalog entry sets `subtractable` | @@ -218,11 +218,11 @@ flowchart LR LG -. "✗ already a value" .-> SM ``` -`implement` — turning a canonical `QueryExpr` into an `L3Node` tree: +`implement` — turning a canonical `QueryExpr` into an `L3Node` DAG: ```rust -pub fn implement_tree(expr: &QueryExpr) -> Result, ImplementError>; -pub fn implement_tree_with(expr: &QueryExpr, cost_model: &dyn CostModel) -> Result, ImplementError>; +pub fn implement_dag(expr: &QueryExpr) -> Result, ImplementError>; +pub fn implement_dag_with(expr: &QueryExpr, cost_model: &dyn CostModel) -> Result, ImplementError>; ``` **"Implementation" is the answer to one question: how is this one @@ -277,7 +277,7 @@ pub trait CostModel { ``` Serving-time — the interface a deployment implements to actually answer -a query against an already-bound `L3Node` tree: +a query against an already-bound `L3Node` DAG: ```rust pub trait SummaryExecutor { diff --git a/old_docs/docs/l4-physical-plan.md b/old_docs/docs/l4-physical-plan.md index 8538ed964..ba8af34d8 100644 --- a/old_docs/docs/l4-physical-plan.md +++ b/old_docs/docs/l4-physical-plan.md @@ -94,7 +94,7 @@ pub trait TopologyDescriptor { // A `StageEdge` names a pair of stages data is allowed to flow // between (e.g. "edge → backend" if that deployment's edge tier // ships summaries up to a backend tier) — the topology's connectivity - // graph, distinct from which stages merely *exist* (`stages()` above). + // DAG, distinct from which stages merely *exist* (`stages()` above). // `StageAllocator::allocate` only assigns a piece of the plan to move // from one stage to another along an edge this list actually // contains; a `TopologyDescriptor` with no edge between two stages is @@ -119,7 +119,7 @@ pub struct Executor { pub address: ExecutorAddr, // OpAMP agent / HTTP endpoint / in-process handle — deployment-defined } -// Stage-level allocation: given an L3 tree + a topology, decide which +// Stage-level allocation: given an L3 DAG + a topology, decide which // nodes land on which stage, subject to deployment constraints. // Per-executor fan-out is the deployment model's own PhysicalPlanner, // using the executor list from DeploymentConstraints::executors(). diff --git a/tools/dag-viewer/README.md b/tools/dag-viewer/README.md index d2c1a65ff..d6c082e6e 100644 --- a/tools/dag-viewer/README.md +++ b/tools/dag-viewer/README.md @@ -93,13 +93,13 @@ cargo run -p asap-devtools --bin dag_export -- \ It ranks candidates with the planner's structural `DefaultCostModel`, so the structure of the export is real — which replacements the search found, which -one won per group, and the merged post-ASAP graph — while no cost is exported +one won per group, and the merged post-ASAP DAG — while no cost is exported at all. Every `CostAnnotation` stays `Unavailable` with no `value` and renders as **Not estimated**; the structural ranking number is never serialized. Use it to see what ASAPPlanner does with a workload before there is a deployment to calibrate against, and `--planner-cost-json` once there is. -Without either flag, `--post-asap` exports the raw graph only. +Without either flag, `--post-asap` exports the raw DAG only. The viewer also accepts the JSON produced by `export_summary_maintenance_plan`. It renders the materialized summary DAG as @@ -121,7 +121,7 @@ always opens in Pre/Post-ASAP mode; there is no `--mode` option. ## JSON contract -`NamedGraph.graph` is the original pre-ASAP DAG. `NamedGraph.post_graph` +`NamedDAG.dag` is the original pre-ASAP DAG. `NamedDAG.post_dag` is the complete translated DAG. Every post-ASAP node produced or carried by a selected replacement directly contains: @@ -184,14 +184,14 @@ operation, `1e-10` per scan byte, and `1e-9` per peak-memory byte, the displayed not statistics inferred by the viewer. The same three fields also appear on `TargetReplacement` -(replacement-region baseline/selected/benefit), `NamedGraph.workload_cost` / -`WorkloadGraph.workload_cost` (whole selected-workload cost/benefit, shared -decisions counted once via `decision.id` dedup). `DagGraph.edge_annotations` +(replacement-region baseline/selected/benefit), `NamedDAG.workload_cost` / +`WorkloadDAG.workload_cost` (whole selected-workload cost/benefit, shared +decisions counted once via `decision.id` dedup). `ExportDAG.edge_annotations` is reserved for a higher layer that has physical evidence for a particular -edge; graph sharing alone never creates an edge cost. The sidebar shows the full breakdown +edge; DAG sharing alone never creates an edge cost. The sidebar shows the full breakdown (value, unit, provenance, baseline, ratio, inputs) on node/edge click and in the workload-scope summary; a post-ASAP node with a costed decision also -gets a concise on-graph `▼NN%`/`▲NN%` badge next to its label. +gets a concise on-DAG `▼NN%`/`▲NN%` badge next to its label. All of this is additive and optional: an export with none of these fields (anything produced before issue #286) renders exactly as before. diff --git a/tools/dag-viewer/dag.example.json b/tools/dag-viewer/dag.example.json index da3e88708..85c0e97b6 100644 --- a/tools/dag-viewer/dag.example.json +++ b/tools/dag-viewer/dag.example.json @@ -3,7 +3,7 @@ { "name": "q1", "source": "SELECT service, COUNT(*) FROM metrics GROUP BY service", - "graph": { + "dag": { "nodes": [ { "id": 0, @@ -361,14 +361,14 @@ }, "after": { "kind": "Summary", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": { - "pre_asap_subgraph": { + "pre_asap_sub_dag": { "nodes": [ { "children": [], @@ -564,7 +564,7 @@ } } ], - "post_graph": { + "post_dag": { "nodes": [ { "id": 0, diff --git a/tools/dag-viewer/index.html b/tools/dag-viewer/index.html index 62ffd88f2..17754d280 100644 --- a/tools/dag-viewer/index.html +++ b/tools/dag-viewer/index.html @@ -447,7 +447,7 @@

ASAP query DAG viewer

-
Drop WorkloadGraph JSON here, or click to choose file(s)
+
Drop WorkloadDAG JSON here, or click to choose file(s)
@@ -456,7 +456,7 @@

ASAP query DAG viewer

Click an edge to inspect its schema
diff --git a/tools/dag-viewer/post_asap_fixture.json b/tools/dag-viewer/post_asap_fixture.json index a680d28f4..ea2de185f 100644 --- a/tools/dag-viewer/post_asap_fixture.json +++ b/tools/dag-viewer/post_asap_fixture.json @@ -3,7 +3,7 @@ { "name": "p95_pktlen", "source": "SELECT srcip, approx_percentile_cont(pkt_len, 0.95) AS p95_pkt_len FROM netflow_table GROUP BY srcip", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {"source": {"Table": {"table_ref": "netflow_table"}}}, "children": [], "hash": 111 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(1 measures)", "detail": {"measures": [{"kind": "quantile", "q": 0.95, "col": 6, "accuracy": {"Epsilon": 0.01}}], "output_names": ["approx_percentile_cont(netflow_table.pkt_len,Float64(0.95))"], "reduction": {"Reduce": [1]}}, "children": [0], "hash": 222, @@ -12,11 +12,11 @@ ], "root": 2 }, - "post_graph": { + "post_dag": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}], "root": 0}}, "children": [] }, + { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}], "root": 0}}, "children": [] }, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 }, - { "id": 2, "kind": "KeepPreAsap", "label": "KeepPreAsap(Project)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": []}], "root": 0}}, "children": [1] } + { "id": 2, "kind": "KeepPreAsap", "label": "KeepPreAsap(Project)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": []}], "root": 0}}, "children": [1] } ], "root": 2 }, @@ -37,9 +37,9 @@ }, "after": { "kind": "Summary", - "graph": { + "dag": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": [], "hash": 111}], "root": 0}}, "children": [] }, + { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": [], "hash": 111}], "root": 0}}, "children": [] }, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 } ], "root": 1 @@ -62,7 +62,7 @@ }, "after": { "kind": "Summary", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {}, "children": [] }, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(HydraKll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "grouping": {"SharedMultiSubpopulation": {"params": {}}}}, "children": [0], "origin_pre_id": 1 } @@ -76,7 +76,7 @@ { "name": "avg_latency", "source": "SELECT service, AVG(latency) FROM metrics GROUP BY service", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {"source": {"Table": {"table_ref": "metrics"}}}, "children": [], "hash": 444 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(1 measures)", "detail": {"measures": [{"kind": "avg", "col": 3}], "output_names": ["avg(metrics.latency)"]}, "children": [0], "hash": 555 }, @@ -101,7 +101,7 @@ }, "after": { "kind": "Rewrite", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {}, "children": [], "hash": 444 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(2 measures)", "detail": {"measures": [{"kind": "sum", "col": 3}, {"kind": "count"}], "output_names": ["sum(metrics.latency)", "count(*)"]}, "children": [0], "hash": 777 }, @@ -116,7 +116,7 @@ { "name": "cse_share_demo", "source": "-- q_a -- SELECT service, COUNT(*) FROM metrics GROUP BY service ;\n-- q_b -- SELECT service, COUNT(*) FROM metrics WHERE region='us' GROUP BY service", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {}, "children": [], "hash": 111 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(1 measures)", "detail": {"measures": [{"kind": "count"}]}, "children": [0], "hash": 999 } @@ -126,7 +126,7 @@ "replacements": [ { "target_pre_id": 0, - "strategy": "SharedSubtree", + "strategy": "SharedSubDAG", "provenance": "CseShare", "rationale": "Scan(metrics) has 2 consumers across this workload — building it once and sharing the Rc beats recomputing it independently per consumer at this estimated row count.", "rank": 0, @@ -137,7 +137,7 @@ }, "after": { "kind": "Rewrite", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics) [shared, 2 consumers]", "detail": {}, "children": [], "hash": 111 } ], "root": 0 } diff --git a/tools/dag-viewer/render.py b/tools/dag-viewer/render.py index 6659561ef..b131bbc9f 100755 --- a/tools/dag-viewer/render.py +++ b/tools/dag-viewer/render.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Bake WorkloadGraph JSON into a single, portable Pre/Post-ASAP HTML page, +"""Bake WorkloadDAG JSON into a single, portable Pre/Post-ASAP HTML page, with the same query selection, workload-union, and node details as index.html, but with the vendored JS libraries and the query data all inlined into one file. @@ -19,13 +19,13 @@ comment) and only differs in packaging: one query's worth of exported `QueryExpr` detail *is* its plan (see the side panel on node click), and shared-hash highlighting *is* what this repo has for CSE today — both a -hash-based proxy, not real CSE output; see README.md's "Shared-subtree +hash-based proxy, not real CSE output; see README.md's "Shared-sub-DAG highlighting is a proxy" section. Structured cost/benefit annotations (issue #286, `CostAnnotation` in crates/types/src/cost.rs) pass through this deep-copy untouched, same as everything else `prepare_workload` below doesn't explicitly rewrite — viewer.js reads `decision.baseline_cost` / -`.selected_cost` / `.benefit`, `NamedGraph.workload_cost`, and -`DagGraph.edge_annotations` directly, with no help needed from this file. +`.selected_cost` / `.benefit`, `NamedDAG.workload_cost`, and +`ExportDAG.edge_annotations` directly, with no help needed from this file. Usage: cargo run -p asap-devtools --bin dag_export -- --sql "..." --name q1 \\ @@ -113,7 +113,7 @@ def _measure(value: object, input_schema: object = None) -> str: def _bounded_lines(lines: list[str], maximum: int = 4, width: int = 64) -> str: - """Keep graph boxes scannable; the sidebar owns the lossless detail.""" + """Keep DAG boxes scannable; the sidebar owns the lossless detail.""" shortened = [line if len(line) <= width else line[: width - 1] + "…" for line in lines] if len(shortened) > maximum: shortened = shortened[: maximum - 1] + [f"… +{len(shortened) - maximum + 1} more"] @@ -213,30 +213,30 @@ def _semantic_label(node: dict, input_schema: object = None) -> str: elif kind == "SummaryDelete": lines.append(f"key: {_compact(detail.get('key'))}") elif kind == "KeepPreAsap": - nested = detail.get("pre_asap_subgraph") + nested = detail.get("pre_asap_sub_dag") nested_nodes = nested.get("nodes", []) if isinstance(nested, dict) else [] nested_root = nested.get("root") if isinstance(nested, dict) else None root = next((item for item in nested_nodes if item.get("id") == nested_root), None) - lines.append(f"unchanged: {root.get('kind', 'pre-ASAP subtree') if root else 'pre-ASAP subtree'}") + lines.append(f"unchanged: {root.get('kind', 'pre-ASAP sub-DAG') if root else 'pre-ASAP sub-DAG'}") else: # Less common variants still show their own scalar IR fields. Avoid - # schema/subgraph blobs, which belong in the click-to-inspect panel. + # schema/sub-DAG blobs, which belong in the click-to-inspect panel. for key, value in detail.items(): - if key not in {"schema", "pre_asap_subgraph"} and value not in (None, [], {}): + if key not in {"schema", "pre_asap_sub_dag"} and value not in (None, [], {}): lines.append(f"{key}: {_compact(value)}") return _bounded_lines(lines) def prepare_workload(workload: dict) -> dict: - """Copy a workload and replace every graph label with readable IR text.""" + """Copy a workload and replace every DAG label with readable IR text.""" prepared = copy.deepcopy(workload) - def prepare_graph(graph: object) -> None: - if not isinstance(graph, dict): + def prepare_dag(dag: object) -> None: + if not isinstance(dag, dict): return - by_id = {node.get("id"): node for node in graph.get("nodes", [])} - for node in graph.get("nodes", []): + by_id = {node.get("id"): node for node in dag.get("nodes", [])} + for node in dag.get("nodes", []): child = by_id.get((node.get("children") or [None])[0]) input_schema = ( child.get("schema") or (child.get("detail") or {}).get("schema") @@ -244,17 +244,17 @@ def prepare_graph(graph: object) -> None: else None ) node["label"] = _semantic_label(node, input_schema) - nested = (node.get("detail") or {}).get("pre_asap_subgraph") - prepare_graph(nested) + nested = (node.get("detail") or {}).get("pre_asap_sub_dag") + prepare_dag(nested) for query in prepared.get("queries", []): - prepare_graph(query.get("graph")) - prepare_graph(query.get("post_graph")) + prepare_dag(query.get("dag")) + prepare_dag(query.get("post_dag")) return prepared def load_workload(paths: list[Path]) -> dict: - """Merge one or more WorkloadGraph JSON files into one, matching + """Merge one or more WorkloadDAG JSON files into one, matching viewer.js's loadFiles(): a query name colliding with an earlier one is disambiguated by suffixing the source filename.""" queries = [] @@ -262,11 +262,11 @@ def load_workload(paths: list[Path]) -> dict: for path in paths: data = json.loads(path.read_text()) incoming = data.get("queries", []) - if not incoming and isinstance(data.get("graph"), dict) and isinstance(data.get("deployments"), list): + if not incoming and isinstance(data.get("dag"), dict) and isinstance(data.get("deployments"), list): incoming = [{ "name": path.stem or "Summary maintenance plan", - "graph": data["graph"], - "post_graph": data["graph"], + "dag": data["dag"], + "post_dag": data["dag"], "lifecycle_plan": True, "lifecycle_summary": { "selected_raw_recompute": data.get("selected_raw_recompute", False), @@ -291,7 +291,7 @@ def load_workload(paths: list[Path]) -> dict: def _json_script(obj: object) -> str: """JSON-serialize `obj` for embedding inside an HTML ') def test_embedded_workload_round_trips(self): - workload = {"queries": [named_graph("q1"), named_graph("q2")]} + workload = {"queries": [named_dag("q1"), named_dag("q2")]} html = render(workload) m = re.search( @@ -177,7 +177,7 @@ def test_embedded_workload_round_trips(self): def test_standalone_export_carries_the_bulk_selection_control(self): """A generated page gets Select-all for free: render.py inlines the markup and viewer.js verbatim, so neither fix needs its own step.""" - html = render({"queries": [named_graph("q1"), named_graph("q2")]}) + html = render({"queries": [named_dag("q1"), named_dag("q2")]}) self.assertIn('id="selectAllToggle"', html) self.assertIn("function bulkSelectionState(", html) # And it must stay a selection control, not a second Clear all: the @@ -190,14 +190,14 @@ def test_standalone_export_carries_the_bulk_selection_control(self): def test_standalone_export_lanes_are_pannable(self): """Dragging the lane background pans the viewport in the generated page too -- `grabbable: false` alone made it a dead zone.""" - html = render({"queries": [named_graph("q1")]}) + html = render({"queries": [named_dag("q1")]}) lanes = re.findall(r"classes: 'laneParent'[^}]*}", html) self.assertEqual(len(lanes), 2, "expected the union and single-query lanes") for lane in lanes: self.assertIn("pannable: true", lane) def test_render_does_not_mutate_callers_workload(self): - workload = {"queries": [named_graph("q1")]} + workload = {"queries": [named_dag("q1")]} original = json.loads(json.dumps(workload)) render(workload) self.assertEqual(workload, original) @@ -207,7 +207,7 @@ def test_render_adds_no_legacy_mode_config(self): # (which documents the window.__DAG_RENDER__ config object in prose) # once that file is inlined verbatim -- match the actual assignment # statement instead. - workload = {"queries": [named_graph("q1"), named_graph("q2")]} + workload = {"queries": [named_dag("q1"), named_dag("q2")]} html = render(workload) self.assertNotRegex(html, r"window\.__DAG_RENDER__ =") @@ -216,14 +216,14 @@ def test_embedded_data_placed_before_viewer_js_body(self): # synchronously as soon as it runs, so it must appear earlier in the # document than viewer.js's own inlined # " substring must not prematurely close the embedded # '")]} + workload = {"queries": [named_dag("q1", source="SELECT ''")]} html = render(workload) m = re.search( @@ -293,7 +293,7 @@ def test_sort_names_expression_direction_and_null_order(self): } self.assertEqual(_semantic_label(node), "Sort\nsort: col[2] descending, nulls first") - def test_prepares_before_after_and_whole_post_asap_graphs(self): + def test_prepares_before_after_and_whole_post_asap_dags(self): node = { "id": 0, "kind": "Aggregate", @@ -301,22 +301,22 @@ def test_prepares_before_after_and_whole_post_asap_graphs(self): "detail": {"measures": [{"kind": "avg", "col": 3}]}, "children": [], } - def graph(): + def dag(): return {"nodes": [dict(node)], "root": 0} workload = { "queries": [ { - "graph": graph(), - "post_graph": graph(), - "replacements": [{"before": graph(), "after": {"graph": graph()}}], + "dag": dag(), + "post_dag": dag(), + "replacements": [{"before": dag(), "after": {"dag": dag()}}], } ] } prepared = prepare_workload(workload) query = prepared["queries"][0] labels = [ - query["graph"]["nodes"][0]["label"], - query["post_graph"]["nodes"][0]["label"], + query["dag"]["nodes"][0]["label"], + query["post_dag"]["nodes"][0]["label"], ] self.assertEqual(labels, ["Aggregate\nmeasure: avg(col[3])"] * 2) self.assertEqual( @@ -324,13 +324,13 @@ def graph(): "Aggregate(1 measures)", ) - def test_replacement_subgraphs_are_not_prepared_for_the_current_viewer(self): - def graph(): + def test_replacement_sub_dags_are_not_prepared_for_the_current_viewer(self): + def dag(): return {"nodes": [{"id": 0, "kind": "Scan", "label": "legacy", "detail": {}, "children": []}], "root": 0} - workload = {"queries": [{"graph": graph(), "replacements": [{"before": graph(), "after": {"graph": graph()}}]}]} + workload = {"queries": [{"dag": dag(), "replacements": [{"before": dag(), "after": {"dag": dag()}}]}]} replacement = prepare_workload(workload)["queries"][0]["replacements"][0] self.assertEqual(replacement["before"]["nodes"][0]["label"], "legacy") - self.assertEqual(replacement["after"]["graph"]["nodes"][0]["label"], "legacy") + self.assertEqual(replacement["after"]["dag"]["nodes"][0]["label"], "legacy") class MainCliTests(unittest.TestCase): @@ -381,7 +381,7 @@ def annotation(value, cache): result["cache_profile"] = cache return result - return {"sourceBatch": batch, "post_graph": {"nodes": [{"decision": { + return {"sourceBatch": batch, "post_dag": {"nodes": [{"decision": { "id": 1, "baseline_cost": annotation(10, profile), "selected_cost": annotation(4, selected_profile or profile), @@ -438,7 +438,7 @@ def test_empty_render_resets_bulk_selection(self): def test_cache_profile_and_inputs_are_rendered(self): """The sidebar exposes the assumptions behind a cache-adjusted cost.""" - annotation = self.query("warm-cache-v1")["post_graph"]["nodes"][0]["decision"]["baseline_cost"] + annotation = self.query("warm-cache-v1")["post_dag"]["nodes"][0]["decision"]["baseline_cost"] annotation["inputs"] = [{"name": "result_cache_hit_ratio", "value": 0.5, "unit": "ratio"}] html = self.js.call("renderCostAnnotation", "Baseline", annotation) self.assertIn("cache warm-cache-v1", html) diff --git a/tools/dag-viewer/viewer.js b/tools/dag-viewer/viewer.js index 62b2e84ca..96e992895 100644 --- a/tools/dag-viewer/viewer.js +++ b/tools/dag-viewer/viewer.js @@ -8,20 +8,20 @@ cytoscape.use(window.cytoscapeDagre); // ── State ──────────────────────────────────────────────────────────────── -// queries: [{ name, graph: { nodes, root }, replacements?, post_graph? }], +// queries: [{ name, dag: { nodes, root }, replacements?, post_dag? }], // flattened across every loaded file (a later file whose query name // collides with an earlier one is kept distinct by suffixing the file // index). `replacements` is the optional --post-asap array of // ReplacementSite entries for that query (each with its own `before`/`after` -// subtree) — defaulted to `[]` when the loaded JSON omits the field, so -// callers never need an extra existence check. `post_graph` is the optional -// --post-asap whole-query merged post-ASAP graph (same flattened -// `{nodes, root}` shape as `graph`, but nodes may be post-ASAP-only kinds +// sub-DAG) — defaulted to `[]` when the loaded JSON omits the field, so +// callers never need an extra existence check. `post_dag` is the optional +// --post-asap whole-query merged post-ASAP DAG (same flattened +// `{nodes, root}` shape as `dag`, but nodes may be post-ASAP-only kinds // like "SummaryAgg" mixed in, and any such node has no `hash` — there's no // corresponding QueryExpr to hash) — left `undefined` when absent (omitted // whenever --post-asap wasn't set, or this query had zero replacements), // unlike `replacements` which always defaults to an array. `workload_cost` -// is the optional per-query `NamedGraph.workload_cost` (issue #286), also +// is the optional per-query `NamedDAG.workload_cost` (issue #286), also // left `undefined` when absent. `sourceBatch` is a viewer-assigned integer // (never present in the JSON itself) shared by every query loaded from the // same document — see computeSelectionWorkloadCost's own doc for why it @@ -122,10 +122,10 @@ function loadFiles(fileList) { reader.onload = () => { try { const parsed = JSON.parse(reader.result); - const incoming = parsed.queries || (parsed.graph && parsed.deployments ? [{ + const incoming = parsed.queries || (parsed.dag && parsed.deployments ? [{ name: file.name.replace(/\.json$/i, '') || 'Summary maintenance plan', - graph: parsed.graph, - post_graph: parsed.graph, + dag: parsed.dag, + post_dag: parsed.dag, lifecycle_plan: true, lifecycle_summary: lifecyclePlanSummary(parsed), }] : []); @@ -137,7 +137,7 @@ function loadFiles(fileList) { let name = q.name; if (existingNames.has(name)) name = `${q.name} (${file.name})`; existingNames.add(name); - queries.push({ name, graph: q.graph, source: q.source, replacements: q.replacements || [], post_graph: q.post_graph, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch }); + queries.push({ name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch }); }); } catch (err) { alert(`Failed to parse ${file.name}: ${err.message}`); @@ -417,7 +417,7 @@ function buildCyStyle() { // Default layout animates dagre's computed positions in rather than snapping // to them — every mode switch, checkbox toggle, and file load rebuilds `cy` // from scratch (see the `cy.destroy()` below), so this animation is what -// makes the graph read as "assembling" instead of flickering to a new +// makes the DAG read as "assembling" instead of flickering to a new // static image each time. `elements` also start at opacity 0 and fade in // (paired with the 'node'/'edge' transition-property set in buildCyStyle) // so a freshly-added element doesn't just pop in mid-layout-animation. @@ -442,7 +442,7 @@ function buildCy(elements, layout) { } // Root borders and click/tap wiring for the Pre/Post-ASAP lanes. -function finalizeGraphInteractions() { +function finalizeDAGInteractions() { // Use borders rather than pictograms so IR text owns the whole node box. cy.nodes('[?root]').addClass('root'); @@ -487,13 +487,13 @@ function renderPrePostAsap() { showModeHint('Select one query for a complete pre/post DAG, or several queries for a batch workload view.'); return; } - const missing = selected.filter((q) => !q.post_graph); + const missing = selected.filter((q) => !q.post_dag); if (missing.length > 0) { viewTitleEl.textContent = selected.length === 1 ? `Pre/Post-ASAP: ${selected[0].name}` : `Pre/Post-ASAP workload: ${selected.length} queries`; const action = document.getElementById('plannerRun') ? 'Open Query planner and click “Plan selected workload”, or re-export with dag_export --post-asap.' : 'Re-export with dag_export --post-asap.'; - showModeHint(`No post-ASAP graph for: ${missing.map((q) => q.name).join(', ')}. ${action}`); + showModeHint(`No post-ASAP DAG for: ${missing.map((q) => q.name).join(', ')}. ${action}`); return; } hideModeHint(); @@ -502,12 +502,12 @@ function renderPrePostAsap() { const elements = laneElements( 'summary-maintenance', `${selected[0].name} · lifecycle plan`, - selected[0].post_graph, + selected[0].post_dag, selected[0], 'post', ); buildCy(elements); - finalizeGraphInteractions(); + finalizeDAGInteractions(); applyHighlighting(); const initial = cy.nodes().filter((node) => !node.data('isLane') && node.data('root')).first(); if (initial && initial.length) { @@ -526,8 +526,8 @@ function renderPrePostAsap() { const costSummary = selected.length === 1 ? selected[0].workload_cost : computeSelectionWorkloadCost(selected); const elements = selected.length === 1 ? [ - ...laneElements('pre-asap', `${selected[0].name} · pre-ASAP`, selected[0].graph, selected[0], 'pre', costSummary && costSummary.baseline_cost), - ...laneElements('post-asap', `${selected[0].name} · post-ASAP`, selected[0].post_graph, selected[0], 'post', costSummary && costSummary.selected_cost), + ...laneElements('pre-asap', `${selected[0].name} · pre-ASAP`, selected[0].dag, selected[0], 'pre', costSummary && costSummary.baseline_cost), + ...laneElements('post-asap', `${selected[0].name} · post-ASAP`, selected[0].post_dag, selected[0], 'post', costSummary && costSummary.selected_cost), ] : [ ...unionStageLaneElements('pre', chosen, costSummary && costSummary.baseline_cost), @@ -535,7 +535,7 @@ function renderPrePostAsap() { ]; buildCy(elements); - finalizeGraphInteractions(); + finalizeDAGInteractions(); applyHighlighting(); const initial = cy.nodes().filter((node) => !node.data('isLane') && node.data('stage') === 'pre' && node.data('root') @@ -553,13 +553,13 @@ function unionStageLaneElements(stage, chosen, laneCost) { // Workload merging is an exporter decision. `workload_node_id` is the // explicit JSON mapping; do not reconstruct identity from node content. const laneId = `${stage}-asap-union`; - const graphFor = (query) => (stage === 'pre' ? query.graph : query.post_graph); + const dagFor = (query) => (stage === 'pre' ? query.dag : query.post_dag); const owners = new Map(); chosen.forEach((qIdx) => { const query = queries[qIdx]; - const graph = graphFor(query); - graph.nodes.forEach((node) => { + const dag = dagFor(query); + dag.nodes.forEach((node) => { if (node.workload_node_id === undefined) return; if (!owners.has(node.workload_node_id)) owners.set(node.workload_node_id, new Set()); owners.get(node.workload_node_id).add(query.name); @@ -579,16 +579,16 @@ function unionStageLaneElements(stage, chosen, laneCost) { // viewport. `grabbable: false` alone made it a dead zone — cytoscape // begins a pan only when the element under the pointer is `pannable()`, // which defaults to false for nodes, and the lane covers most of the - // canvas once the graph is zoomed past the viewport. Still not grabbable: + // canvas once the DAG is zoomed past the viewport. Still not grabbable: // cytoscape overrides `grabbable` for a pannable node. { data: { id: laneId, label: laneCostLabel(`${stage === 'pre' ? 'pre-ASAP' : 'post-ASAP'} · workload union`, laneCost, stage), isLane: true }, classes: 'laneParent', selectable: false, grabbable: false, pannable: true }, ]; chosen.forEach((qIdx) => { const query = queries[qIdx]; - const graph = graphFor(query); - const byId = new Map(graph.nodes.map((node) => [node.id, node])); - graph.nodes.forEach((node) => { + const dag = dagFor(query); + const byId = new Map(dag.nodes.map((node) => [node.id, node])); + dag.nodes.forEach((node) => { const key = keyFor(qIdx, node); let entry = entries.get(key); if (!entry) { @@ -599,7 +599,7 @@ function unionStageLaneElements(stage, chosen, laneCost) { for (const decision of translationsForNode(query, node, stage)) { entry.decisions.set(decision.id, decision); } - if (node.id === graph.root) entry.rootFor.add(query.name); + if (node.id === dag.root) entry.rootFor.add(query.name); node.children.forEach((childId) => { const childKey = keyFor(qIdx, byId.get(childId)); const child = byId.get(childId); @@ -636,18 +636,18 @@ function unionStageLaneElements(stage, chosen, laneCost) { // Builds one Pre/Post-ASAP lane (a dashed compound parent plus its // nodes/edges). `nodes` is either a plain pre-ASAP -// DagNode list (the `before` subtree, or an `after.kind === "Rewrite"` -// graph) or a SummaryDagNode list (an `after.kind === "Summary"` graph) — +// DagNode list (the `before` sub-DAG, or an `after.kind === "Rewrite"` +// DAG) or a SummaryDagNode list (an `after.kind === "Summary"` DAG) — // both shapes carry id/kind/label/detail/children, which is all a lane // needs; SummaryDagNode's missing `hash`/`notes` fields are simply never // read by this function or by showPrePostDetail below. -function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { - const nodes = graph.nodes; +function laneElements(laneId, laneLabel, dag, query, stage, laneCost) { + const nodes = dag.nodes; const byId = new Map(nodes.map((node) => [node.id, node])); // Keep the delimiter escaped in source. A literal NUL is replaced by the // HTML tokenizer when viewer.js is embedded by render.py, which otherwise // makes these keys differ from the lookup below. - const edgeCostByPair = new Map((graph.edge_annotations || []).map((edge) => [`${edge.from}\u0000${edge.to}`, edge.cost])); + const edgeCostByPair = new Map((dag.edge_annotations || []).map((edge) => [`${edge.from}\u0000${edge.to}`, edge.cost])); const elements = [ // Pannable for the same reason as the union lane above. { data: { id: laneId, label: laneCostLabel(laneLabel, laneCost, stage), isLane: true }, classes: 'laneParent', selectable: false, grabbable: false, pannable: true }, @@ -657,7 +657,7 @@ function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { data: { id: `${laneId}-${node.id}`, parent: laneId, - // On-graph label carries a concise cost/benefit badge (issue #286) + // On-DAG label carries a concise cost/benefit badge (issue #286) // when this node's decision has one; `node.label` itself (nested, // used everywhere else — the sidebar, schema derivation, …) stays // exactly the plain IR label. @@ -669,7 +669,7 @@ function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { // object. kind: node.kind, category: categoryOf(node.kind), - root: node.id === graph.root, + root: node.id === dag.root, isPrePost: true, laneId, stage, @@ -687,7 +687,7 @@ function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { elements.push({ data: { // Arrow points from input to consumer (data-flow direction), the - // reverse of the tree's parent->child structure. + // reverse of the DAG's parent->child structure. id: `e-${laneId}-${node.id}-${childId}`, source: `${laneId}-${childId}`, target: `${laneId}-${node.id}`, @@ -800,8 +800,8 @@ function renderDecisionCostBlock(entry) { `; } -// Short, on-graph badge text for a post-ASAP node's own winning decision — -// "concise on-graph benefit/cost badges" per issue #286; the full +// Short, on-DAG badge text for a post-ASAP node's own winning decision — +// "concise on-DAG benefit/cost badges" per issue #286; the full // breakdown only ever appears in the sidebar (`renderDecisionCostBlock`). // Empty string whenever there's nothing to show (no decision, or its // benefit is `Unavailable`) so an un-costed node's label is untouched. @@ -842,7 +842,7 @@ function computeSelectionWorkloadCost(selected) { let cacheProfile = null; let hasCacheProfile = false; for (const query of selected) { - const nodes = (query.post_graph && query.post_graph.nodes) || []; + const nodes = (query.post_dag && query.post_dag.nodes) || []; for (const node of nodes) { const decision = node.decision; if (!decision) continue; @@ -967,12 +967,12 @@ function showEdgeDetail(edge) { function renderScopeSummary(selected) { scopePickerEl.classList.add('visible'); const strategies = new Set(); - selected.forEach((query) => (query.post_graph?.nodes || []).forEach((node) => { + selected.forEach((query) => (query.post_dag?.nodes || []).forEach((node) => { if (node.decision && node.decision.rank === 0) strategies.add(node.decision.strategy); })); const scope = selected.length === 1 ? 'Single query' : `Batch workload · ${selected.length} queries`; const strategyText = strategies.size ? `Winning strategies: ${Array.from(strategies).join(', ')}` : 'No selected replacements'; - // Single query: use the exporter's own precomputed `NamedGraph.workload_cost` + // Single query: use the exporter's own precomputed `NamedDAG.workload_cost` // directly. Multiple queries: no single precomputed field covers exactly // this subset, so aggregate the explicit per-decision annotations already // in the export (dedup by `decision.id`) — see @@ -980,7 +980,7 @@ function renderScopeSummary(selected) { // client-side cost estimation. const costSummary = selected.length === 1 ? selected[0].workload_cost : computeSelectionWorkloadCost(selected); const benefitValue = costSummary && costSummary.benefit && costSummary.benefit.value; - const hasEdgeCosts = selected.some((query) => (query.post_graph?.edge_annotations || []).length > 0); + const hasEdgeCosts = selected.some((query) => (query.post_dag?.edge_annotations || []).length > 0); const outcomeLabel = typeof benefitValue !== 'number' ? 'difference' : benefitValue > 0 ? 'saved' : benefitValue < 0 ? 'additional cost' : 'no change'; @@ -1113,7 +1113,7 @@ function renderTableSchemas(qs) { const box = document.getElementById('tableSchemaBox'); const schemas = new Map(); qs.forEach((query) => { - (query.graph?.nodes || []).filter((node) => node.kind === 'Scan').forEach((node) => { + (query.dag?.nodes || []).filter((node) => node.kind === 'Scan').forEach((node) => { const key = JSON.stringify([node.detail, node.schema]); if (!schemas.has(key)) schemas.set(key, { node, owners: [] }); schemas.get(key).owners.push(query.name); @@ -1128,12 +1128,12 @@ function renderTableSchemas(qs) { function applyHighlighting() { if (!cy) return; const selected = getParticipants().map((i) => queries[i]); - const ownersFor = (graphOf) => { + const ownersFor = (dagOf) => { const owners = new Map(); selected.forEach((query) => { const seen = new Set(); - const graph = graphOf(query); - (graph ? graph.nodes : []).forEach((node) => { + const dag = dagOf(query); + (dag ? dag.nodes : []).forEach((node) => { const id = node.workload_node_id; if (id === undefined || seen.has(id)) return; seen.add(id); @@ -1143,8 +1143,8 @@ function applyHighlighting() { }); return owners; }; - const preOwners = ownersFor((query) => query.graph); - const postOwners = ownersFor((query) => query.post_graph); + const preOwners = ownersFor((query) => query.dag); + const postOwners = ownersFor((query) => query.post_dag); cy.nodes().forEach((element) => { if (element.data('isLane')) return; const node = element.data('node'); @@ -1174,7 +1174,7 @@ function renderLegend() { const panelBg = getComputedStyle(document.documentElement).getPropertyValue('--panel2').trim() || '#f0f2f5'; const mutedColor = getComputedStyle(document.documentElement).getPropertyValue('--muted').trim() || '#6b7280'; rows.push(`
- Pass-through (KeepPreAsap)Unchanged pre-ASAP subtree carried into the Summary graph as-is
`); + Pass-through (KeepPreAsap)Unchanged pre-ASAP sub-DAG carried into the Summary DAG as-is`); legendList.innerHTML = rows.join(''); } @@ -1203,13 +1203,13 @@ function loadWorkload(parsed) { const incoming = (parsed && parsed.queries) || []; // One batch id for this whole document — see `sourceBatch`'s own doc above. const sourceBatch = nextSourceBatch++; - incoming.forEach((q) => queries.push({ name: q.name, graph: q.graph, source: q.source, replacements: q.replacements || [], post_graph: q.post_graph, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch })); + incoming.forEach((q) => queries.push({ name: q.name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch })); if (activeIndex === -1 && queries.length > 0) activeIndex = 0; if (participants.size === 0 && activeIndex >= 0) participants.add(activeIndex); } // Entry point used by planner-ui.js after the local HTTP backend returns a -// freshly generated WorkloadGraph. Replace (rather than append to) the +// freshly generated WorkloadDAG. Replace (rather than append to) the // current data and open the complete workload pre/post view. window.renderPlannerWorkload = function renderPlannerWorkload(parsed) { queries = []; @@ -1220,7 +1220,7 @@ window.renderPlannerWorkload = function renderPlannerWorkload(parsed) { render(); }; -// render.py's standalone output bakes the WorkloadGraph JSON directly into +// render.py's standalone output bakes the WorkloadDAG JSON directly into // the page as a