diff --git a/Cargo.lock b/Cargo.lock index 79da8962c..de76c1704 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -391,6 +391,7 @@ dependencies = [ "asap-frontend-promql", "asap-frontend-sql", "asap-physical-operators", + "asap-planner", "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "futures", diff --git a/crates/asap-aware-mapping/src/accuracy/composition.rs b/crates/asap-aware-mapping/src/accuracy/composition.rs index f42e8f158..daad8071c 100644 --- a/crates/asap-aware-mapping/src/accuracy/composition.rs +++ b/crates/asap-aware-mapping/src/accuracy/composition.rs @@ -448,9 +448,7 @@ fn composed_provenance( } pub(super) fn exact_operation_rule(operation: &ExactOperation) -> Option { - let ExactOperation::Aggregate { measures, .. } = operation else { - return None; - }; + let ExactOperation::Aggregate { measures, .. } = operation; match measures.as_slice() { [intent] => crate::function_rules::function_rules(intent).map(|rules| rules.accuracy), // The remaining functions are exact over exact samples, but have @@ -1035,8 +1033,8 @@ mod tests { reduction: asap_types::pre_asap::Reduction::PerEntity, measures: vec![intent], output_names: vec![], - having: None, filters: vec![], + having: None, }; assert_eq!( DefaultAccuracyModel.exact_operation_rule(&operation(AggIntent::Rate)), diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs b/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs index 46559358d..9482c539a 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs @@ -65,7 +65,7 @@ mod tests { } #[test] - fn heap_readout_retains_frequency_metric() { + fn heap_evaluation_retains_frequency_metric() { use asap_types::post_asap::{GroupingStrategy, SketchKind}; let cms_heap = SketchParams::CmsWithHeap { width: 272, diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs b/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs index 4431a4118..f4556446c 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs @@ -1,7 +1,7 @@ //! Estimator-specific confidence for classic HLL's linear-counting branch. //! //! This is conditional on independent uniform bucket hashes and an enforced -//! upper bound on distinct items in the complete readout population (including +//! upper bound on distinct items in the complete evaluation population (including //! all merged panes). It is not an RSE-to-normal conversion or an ERP fit. use super::*; @@ -56,13 +56,13 @@ impl ClassicHllConfidence { value: self.relative_error, }, failure_probability: ProbabilityExpr::Constant { value: delta }, - provenance: vec![GuaranteeSource::SketchReadout { + provenance: vec![GuaranteeSource::SketchEvaluation { algorithm: "Hll".into(), contract: "classic_hll_linear_counting_collision_bound_v1".into(), params: serde_json::json!({"precision": precision, "max_distinct": self.max_distinct, "relative_error": self.relative_error, "hash_assumption": "independent_uniform_buckets", - "population_scope": "complete_readout_including_merged_panes"}), + "population_scope": "complete_evaluation_including_merged_panes"}), query: "Cardinality".into(), }], }) @@ -187,7 +187,7 @@ mod tests { } } } - /// The model's readout formula matches the actual classic estimator after merge. + /// The model's evaluation formula matches the actual classic estimator after merge. #[test] fn native_classic_estimator_and_merged_registers_use_the_same_contract() { use asap_sketchlib::sketches::hll::{Classic, HyperLogLogP16}; diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs b/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs index 4cfe6f073..4f880894d 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs @@ -69,7 +69,7 @@ mod tests { assert_eq!(g.approximate_layer_count(), 1); assert!(g.provenance.iter().any(|source| matches!( source, - GuaranteeSource::SketchReadout { contract, .. } + GuaranteeSource::SketchEvaluation { contract, .. } if contract == "apache_datasketches_kll_empirical_99_a9b42755072b" ))); } diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs b/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs index 4eabf40be..0cd5fcb4f 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs @@ -46,7 +46,7 @@ fn bounded_guarantee( metric, bound: BoundExpr::Constant { value: bound }, failure_probability: delta, - provenance: vec![GuaranteeSource::SketchReadout { + provenance: vec![GuaranteeSource::SketchEvaluation { algorithm: format!("{algorithm:?}"), contract: contract.into(), params: serde_json::to_value(params).unwrap_or(serde_json::Value::Null), @@ -169,9 +169,9 @@ impl<'a> EstimatorAccuracy<'a> { fn hll(&self) -> Option { let EstimatorContract::ClassicHll { - max_distinct_per_readout, + max_distinct_per_evaluation, } = self.contract?; - hll::ClassicHllConfidence::new(max_distinct_per_readout, self.epsilon) + hll::ClassicHllConfidence::new(max_distinct_per_evaluation, self.epsilon) } pub(crate) fn size_params(&self, algorithm: &SketchAlgorithm) -> Option { diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs b/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs index 297cf53d3..6a89a105e 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs +++ b/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs @@ -1,4 +1,4 @@ -//! UnivMon currently certifies only its exact unit-update total readout. +//! UnivMon currently certifies only its exact unit-update total evaluation. use super::*; pub(super) fn guarantee(query: &SketchStatistic) -> Option { diff --git a/crates/asap-aware-mapping/src/accuracy/evidence.rs b/crates/asap-aware-mapping/src/accuracy/evidence.rs index 6391c746e..f779d5782 100644 --- a/crates/asap-aware-mapping/src/accuracy/evidence.rs +++ b/crates/asap-aware-mapping/src/accuracy/evidence.rs @@ -2,12 +2,12 @@ use super::*; /// A trusted source assertion scoped by `AccuracyEvidenceProvider` to one -/// complete readout. Choosing this variant asserts the estimator and hash +/// complete evaluation. Choosing this variant asserts the estimator and hash /// assumptions; it must not be inferred from sampled population statistics. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum EstimatorContract { /// Classic HLL with independent uniform bucket hashing, including merged panes. - ClassicHll { max_distinct_per_readout: u32 }, + ClassicHll { max_distinct_per_evaluation: u32 }, } /// An enforced domain for every sample of a direct quantile operand, in every @@ -18,7 +18,7 @@ pub enum EstimatorContract { pub struct QuantileInputDomain { pub lower: f64, pub upper: f64, - /// Upper bound on samples per evaluation, matching the pinned readout's + /// Upper bound on samples per evaluation, matching the pinned evaluation's /// exact Float64 rank limit. The population must also be nonempty. pub max_samples: u64, pub contract: String, @@ -87,31 +87,22 @@ pub struct PropagationStats { /// Supplies typed planning-time evidence required by propagation rules. pub trait AccuracyEvidenceProvider { /// Trusted estimator contract for this complete aggregate expression, - /// including source, filters, grouping and all panes in each readout. + /// including source, filters, grouping and all panes in each evaluation. /// An observed cardinality is not an enforced population bound. - fn estimator_contract( - &self, - _expression: &asap_types::pre_asap::QueryExpr, - ) -> Option { + fn estimator_contract(&self, _expression: &OperatorNode) -> Option { None } /// Enforced upper bound on distinct (partition, item) identities across a - /// complete TopK readout. Used to union-bound score errors for adaptively + /// complete TopK evaluation. Used to union-bound score errors for adaptively /// selected candidates. Observed cardinality is not sufficient evidence. - fn topk_max_distinct_items( - &self, - _expression: &asap_types::pre_asap::QueryExpr, - ) -> Option { + fn topk_max_distinct_items(&self, _expression: &OperatorNode) -> Option { None } /// Proof scoped to this complete quantile expression, including its source, /// filters, grouping and window. `None` means unknown, including emptiness. - fn quantile_input_domain( - &self, - _operand: &asap_types::pre_asap::query_expr::QueryExpr, - ) -> Option { + fn quantile_input_domain(&self, _operand: &OperatorNode) -> Option { None } diff --git a/crates/asap-aware-mapping/src/accuracy/mod.rs b/crates/asap-aware-mapping/src/accuracy/mod.rs index 9026de468..a2f3354c0 100644 --- a/crates/asap-aware-mapping/src/accuracy/mod.rs +++ b/crates/asap-aware-mapping/src/accuracy/mod.rs @@ -20,18 +20,20 @@ pub use evidence::{ QuantileInputDomain, WorkloadAccuracyEvidence, }; +use asap_types::ir::OperatorNode; use asap_types::post_asap::{ - AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ExactOperation, FieldDataType, - GuaranteeSource, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchParams, - SketchStatistic, + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, FieldDataType, GuaranteeSource, + ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchParams, SketchStatistic, }; use asap_types::types::AccuracyTarget; +use crate::exact_composition::ExactOperation; + /// The deployment-extensible accuracy algebra. `asap-aware-mapping` ships /// [`DefaultAccuracyModel`]; a deployment with a proof for a composition the /// default rejects (a registered cross-metric conversion, say) implements /// this trait and passes it to -/// [`crate::replacement::SketchAlgorithmStrategy::new_with_planning_inputs`]. +/// [`crate::replacement::ASAPStrategies::new_with_planning_inputs`]. pub trait AccuracyModel { /// The definition-registered rule for applying `operation` to an /// approximate input. `None` means the function is exact only over exact @@ -78,7 +80,7 @@ pub struct DefaultAccuracyModel; const SATISFACTION_TOLERANCE: f64 = 1e-9; impl DefaultAccuracyModel { - /// Derive the guarantee for the committed estimator parameters and readout. + /// Derive the guarantee for the committed estimator parameters and evaluation. pub fn sketch_guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 6014d3ba3..36a219d4e 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -28,10 +28,10 @@ //! //! ## What counts as a "near-duplicate", and why //! -//! Two [`QueryExpr::Aggregate`] nodes are accuracy-near-duplicates here iff, +//! Two `NonASAPOp::Aggregate` nodes are accuracy-near-duplicates here iff, //! **in this order**: //! -//! 1. Both are the same bindable shape [`crate::replacement::SketchAlgorithmStrategy`] +//! 1. Both are the same bindable shape [`crate::replacement::ASAPStrategies`] //! itself targets — a single measure, no `HAVING` (`bindable_intent`'s own //! scope) — **and** that one measure is one of the four accuracy-bearing //! [`AggIntent`] variants ([`crate::replacement::accuracy_target`]'s own @@ -105,7 +105,7 @@ //! //! Like every [`ReplacementStrategy`], this only ever *proposes* — the //! looser-accuracy consumer's own independently-sized candidate (from -//! [`crate::replacement::SketchAlgorithmStrategy`]) stays in its +//! [`crate::replacement::ASAPStrategies`]) stays in its //! [`crate::replacement::TargetSubDAGCandidates`] right alongside this strategy's //! "read the tighter sibling instead" [`Replacement::Rewrite`] candidate; //! [`crate::cost_model::CostModel`]-driven ranking picks between them; @@ -149,11 +149,13 @@ //! same structural child, so adding the edge preserves the reference DAG's //! parent-before-child topological ordering. +use asap_types::ir::non_asap::any_measure_filtered; use std::cmp::Ordering; use std::rc::Rc; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, QueryExpr, Reduction}; use asap_types::types::AccuracyTarget; use crate::replacement::{ @@ -169,27 +171,27 @@ type BindableAccuracyAggregate<'a> = ( &'a AggIntent, &'a AccuracyTarget, &'a [String], - &'a Rc, + &'a Rc, ); /// The `(reduction, intent, accuracy, output_names, child)` shape this /// module operates on: the same single-measure, no-`HAVING` bindable shape -/// [`crate::replacement::SketchAlgorithmStrategy`] targets (see that +/// [`crate::replacement::ASAPStrategies`] targets (see that /// module's private `bindable_intent`), further narrowed to a measure whose /// intent actually carries an [`AccuracyTarget`] /// ([`crate::replacement::accuracy_target`]'s own scope: `Count` / /// `Quantile` / `Cardinality` / `TopK`). `None` for anything else, including /// a multi-measure or `HAVING` aggregate, a non-`Aggregate` node, or an /// accuracy-free intent (`Sum`, `Avg`, …). -fn bindable_accuracy_aggregate(node: &QueryExpr) -> Option> { - let QueryExpr::Aggregate { +fn bindable_accuracy_aggregate(node: &OperatorNode) -> Option> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names, filters, having, child, - } = node + }) = node.non_asap() else { return None; }; @@ -278,7 +280,7 @@ fn strictly_tighter(a: &AccuracyTarget, b: &AccuracyTarget) -> bool { /// this strategy from the same post-CSE `Aggregate` sibling set it already /// builds for `RollupStrategy`. pub struct AccuracyReconciliationStrategy { - siblings: Vec>, + siblings: Vec>, } impl AccuracyReconciliationStrategy { @@ -286,7 +288,7 @@ impl AccuracyReconciliationStrategy { /// each as a candidate tighter-accuracy source (or looser-accuracy /// target) — typically the full set of `Aggregate` nodes a workload-wide /// discovery pass already found. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -311,14 +313,14 @@ impl AccuracyReconciliationStrategy { /// reports no unique key — see `cse.rs`'s "Legality" section) would get /// proposed for reconciliation even though nothing guarantees a second /// read of it lines up row-for-row with the first. - fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { let Some((target_reduction, target_intent, target_accuracy, target_names, target_child)) = bindable_accuracy_aggregate(target.root) else { return Vec::new(); }; - let mut sources: Vec<&Rc> = self + let mut sources: Vec<&Rc> = self .siblings .iter() .filter(|candidate| { @@ -335,9 +337,7 @@ impl AccuracyReconciliationStrategy { && (Rc::ptr_eq(child, target_child) || child == target_child) && same_intent_except_accuracy(intent, target_intent) && strictly_tighter(accuracy, target_accuracy) - && candidate - .output_schema() - .is_ok_and(|schema| schema.has_unique_key()) + && candidate.schema.has_unique_key() }) .collect(); sources.sort_by(|a, b| { @@ -369,7 +369,7 @@ impl ReplacementStrategy for AccuracyReconciliationStrategy { .expect("tighter_sources only returns bindable_accuracy_aggregate matches"); ReplacementSubDAG { strategy: self.name(), - replacement: Replacement::Rewrite(Rc::clone(source)), + replacement: Replacement::SubDAG(Rc::clone(source)), provenance: ReplacementProvenance::AccuracyReconciliation, rationale: format!( "reuses a near-duplicate sibling aggregate — identical intent and grouping \ @@ -391,9 +391,9 @@ impl ReplacementStrategy for AccuracyReconciliationStrategy { mod tests { use super::*; use crate::cost_model::{CostModel, DefaultCostModel}; + use asap_types::ir::cse::share_common_sub_dags; + use asap_types::ir::operator_properties::{GroupKeys, Source}; use asap_types::post_asap::SketchAlgorithm; - use asap_types::pre_asap::cse::share_common_sub_dags; - use asap_types::pre_asap::query_expr::{GroupKeys, Source}; use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; /// `[ts(0), value(1), job(2)]`. @@ -401,8 +401,8 @@ mod tests { /// willing to hoist it — see `Schema::has_unique_key`/`cse.rs`'s own /// "Legality" section: a producer with no provable unique key is always /// inserted fresh, never hoisted, regardless of structural equality. - fn metric_scan() -> Rc { - Rc::new(QueryExpr::Scan { + fn metric_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -414,21 +414,23 @@ mod tests { 0, vec![vec![0]], ), - }) + })) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } - fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { agg( vec![2], AggIntent::Quantile { @@ -441,10 +443,14 @@ mod tests { } /// A globally-grouped (`by(vec![])`) quantile — `aggregate_output_schema` - /// reports no unique key for an empty `by` (see `query_expr.rs`'s own + /// reports no unique key for an empty `by` (see `aggregate_schema.rs`'s own /// `unique_keys = if by.is_empty() || has_count_values { vec![] } else /// { .. }`). - fn global_quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn ungrouped_quantile( + q: f64, + accuracy: AccuracyTarget, + child: &Rc, + ) -> Rc { agg( vec![], AggIntent::Quantile { @@ -463,9 +469,9 @@ mod tests { q: f64, accuracy: AccuracyTarget, excluded: Vec, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![AggIntent::Quantile { col: None, @@ -476,7 +482,8 @@ mod tests { filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } // ── dominates / strictly_tighter ───────────────────────────────────── @@ -548,7 +555,7 @@ mod tests { assert!(strategy.matches(&TargetSubDAG::new(&loose))); let replacements = strategy.replacements(&TargetSubDAG::new(&loose)); assert_eq!(replacements.len(), 1); - let Replacement::Rewrite(rc) = &replacements[0].replacement else { + let Replacement::SubDAG(rc) = &replacements[0].replacement else { panic!("expected a Rewrite candidate"); }; assert!(Rc::ptr_eq(rc, &tight)); @@ -580,7 +587,7 @@ mod tests { assert!( loose_group.candidates.iter().any(|candidate| { candidate.strategy == "AccuracyReconciliationStrategy" - && matches!(candidate.replacement, Replacement::Rewrite(_)) + && matches!(candidate.replacement, Replacement::SubDAG(_)) }), "expected an AccuracyReconciliationStrategy candidate for the looser consumer, got: \ {:?}", @@ -661,7 +668,7 @@ mod tests { // ── exact structural equality / share_common_sub_dags is unchanged ──── #[test] - fn share_common_sub_dags_still_never_merges_differing_accuracy() { + fn share_common_subdags_still_never_merges_differing_accuracy() { // The additive guarantee this issue explicitly must not violate: // pre-ASAP CSE's own exact-equality merge stays exact. Two // aggregates differing only in `accuracy` must come back as two @@ -669,8 +676,8 @@ mod tests { // is the *only* place cross-accuracy sharing gets proposed, never // `share_common_sub_dags` itself. let scan = metric_scan(); - let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); + let a = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let b = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!( @@ -684,10 +691,10 @@ mod tests { // The identical scan child, though, is still shared exactly as // before — this module changes nothing about that. - let QueryExpr::Aggregate { child: child_a, .. } = roots[0].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child: child_a, .. }) = roots[0].1.non_asap() else { panic!("expected an Aggregate root"); }; - let QueryExpr::Aggregate { child: child_b, .. } = roots[1].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child: child_b, .. }) = roots[1].1.non_asap() else { panic!("expected an Aggregate root"); }; assert!(Rc::ptr_eq(child_a, child_b)); @@ -700,8 +707,8 @@ mod tests { // exact equality — unrelated to this module, but pins the contrast // with the test above. let scan = metric_scan(); - let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); + let a = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let b = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); @@ -717,11 +724,11 @@ mod tests { // `share_common_sub_dags`/`RollupStrategy` apply, which this // strategy must not bypass (module docs, point 5). let scan = metric_scan(); - let tight = global_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); - let loose = global_quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); + let tight = ungrouped_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let loose = ungrouped_quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); assert!( - !tight.output_schema().unwrap().has_unique_key(), + !tight.schema.clone().has_unique_key(), "fixture sanity: a globally-grouped aggregate has no provable unique key" ); @@ -741,7 +748,7 @@ mod tests { let loose = without_quantile(0.99, AccuracyTarget::Epsilon(0.05), vec![2], &scan); assert!( - !tight.output_schema().unwrap().has_unique_key(), + !tight.schema.clone().has_unique_key(), "fixture sanity: a without(...) aggregate has no provable unique key" ); @@ -831,7 +838,7 @@ mod tests { "global_selection must commit to some candidate for a single-consumer looser target" ); // With no recompute term at all (it never rebuilds `target`), this - // candidate strictly undercuts every SketchAlgorithmStrategy + // candidate strictly undercuts every ASAPStrategies // candidate (which each pay a recompute term on top of their own // maintenance term) under DefaultCostModel's numbers — the sane // direction: reading an already-necessary sibling should be able to diff --git a/crates/asap-aware-mapping/src/analytical_cost.rs b/crates/asap-aware-mapping/src/analytical_cost.rs index ad921d23a..8256d6f10 100644 --- a/crates/asap-aware-mapping/src/analytical_cost.rs +++ b/crates/asap-aware-mapping/src/analytical_cost.rs @@ -2970,7 +2970,7 @@ mod tests { } fn comparison_scope() -> ComparisonScope { - use asap_types::pre_asap::query_expr::Source; + use asap_types::ir::operator_properties::Source; use asap_types::workload::{ DurationMs, QueryRecurrence, QueryTimeScope, RepeatedDemand, RepetitionInterval, TimeSelection, TimestampMs, @@ -3009,9 +3009,7 @@ mod tests { #[test] fn comparison_rejects_different_snapshot_predicate_time_or_horizon() { - use std::rc::Rc; - - use asap_types::pre_asap::query_expr::{Predicate, QueryExpr}; + use asap_types::ir::{Predicate, ScalarExpr}; use asap_types::workload::{DurationMs, TimestampMs}; let raw = comparison_scope(); @@ -3027,7 +3025,7 @@ mod tests { candidate = raw.clone(); candidate.sources[0] .predicates - .push(Predicate(Rc::new(QueryExpr::promql_scalar(1.0)))); + .push(Predicate(ScalarExpr::literal_f64(1.0))); assert_eq!( validate_comparison_scopes(&raw, &candidate), Err(AnalyticalCostError::ComparisonScopeMismatch("sources")) @@ -3247,7 +3245,7 @@ mod tests { operator: PhysicalOperator::Scan, children: vec![], scan_selection: Some(ScanSelection { - source: asap_types::pre_asap::query_expr::Source::Table { + source: asap_types::ir::operator_properties::Source::Table { table_ref: "other_metrics".into(), }, source_snapshot_id: "catalog-version-42".into(), @@ -3312,7 +3310,7 @@ mod tests { let mut scope = comparison_scope(); let coverage = scope.sources[0].clone(); scope.sources.push(ScanSelection { - source: asap_types::pre_asap::query_expr::Source::Table { + source: asap_types::ir::operator_properties::Source::Table { table_ref: "auxiliary".into(), }, source_snapshot_id: "catalog-version-42".into(), diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index 0315e88f1..9fc170666 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -26,7 +26,7 @@ //! than overloading these ones across incompatible `Kind`/`Params` types. //! //! Every entry point that doesn't take an explicit `&dyn CostModel` -//! ([`SketchAlgorithmStrategy::default_cost_model`](crate::replacement::SketchAlgorithmStrategy::default_cost_model), +//! ([`ASAPStrategies::default_cost_model`](crate::replacement::ASAPStrategies::default_cost_model), //! [`search_workload`](crate::replacement::search_workload)) runs against //! [`DefaultCostModel`], so a deployment that never plugs in its own cost //! model keeps today's static-preference-order behavior exactly, byte for @@ -36,7 +36,7 @@ //! //! [`CseCandidate`]/[`ShareDecision`]/[`CostModel::cse_share_decision`] below //! decide whether a CSE-detected shared sub-DAG -//! ([`asap_types::pre_asap::cse::share_common_sub_dags`], issue #223 stages +//! ([`asap_types::ir::cse::share_common_sub_dags`], issue #223 stages //! 1-2, PR #235) is actually worth sharing, via a real Volcano/Cascades-style //! cost comparison rather than a fixed rule. See //! `docs/design_docs/cse-cost-model-decision.md` for the full design discussion (why @@ -48,14 +48,14 @@ use std::rc::Rc; +use crate::exact_composition::ExactOperation; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ - ExactOperation, FieldDataType, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, - SketchParams, SketchStatistic, SummaryExpr, SummaryMaintenanceLifecycleGuarantee, SummaryNode, - SummaryWindowFramework, + FieldDataType, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, SketchParams, + SketchStatistic, SummaryMaintenanceLifecycleGuarantee, SummaryWindowFramework, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::types::AccuracyTarget; use crate::exact_composition::{ExactComposition, OperationPlacement}; @@ -94,7 +94,7 @@ pub struct CostProvenance { /// operator on the update path must never be handed one. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub struct ValueOperationCapabilities { - /// The runtime can apply an exact operator to summary readouts at + /// The runtime can apply an exact operator to summary evaluations at /// query evaluation time. pub query_time: bool, /// The runtime can apply an exact row transform on the update path, @@ -127,17 +127,17 @@ impl ValueOperationCapabilities { /// composes with. #[derive(Debug, Clone, Copy)] pub struct ExactCompositionCostRequest<'a> { - /// The pre-ASAP target the composed candidate replaces. - pub target: &'a QueryExpr, + /// The target the composed candidate replaces. + pub target: &'a OperatorNode, /// The composition itself — placement, operator, child target. pub composition: &'a ExactComposition, /// For [`OperationPlacement::Read`]: the child target's *selected* - /// summary readout candidate the exact operator consumes. For + /// summary evaluation candidate the exact operator consumes. For /// [`OperationPlacement::Maintenance`]: the maintained summary *above* the /// transform that consumes its output (the `SummaryAgg` this transform /// feeds). Either way, the summary whose maintenance/read cost the /// formula charges. - pub summary: &'a SummaryNode, + pub summary: &'a OperatorNode, /// How many times this site actually runs once ancestors' own choices /// are accounted for (see `CandidateLogicalASAPDAGs::global_selection`). pub effective_consumer_count: usize, @@ -147,11 +147,11 @@ pub struct ExactCompositionCostRequest<'a> { /// optional: **an unknown stays `None` — never a zero** — so a formula /// with a missing input yields no rate at all rather than a spuriously /// cheap one, and global selection then keeps the conservative -/// `KeepPreAsap` behavior. A deployment model that wants defaults supplies +/// keep-as-is behavior. A deployment model that wants defaults supplies /// them explicitly by overriding [`CostModel::exact_composition_cost_inputs`]. #[derive(Debug, Clone, PartialEq)] pub struct ExactCompositionCostInputs { - /// Exact operator cost per row it processes — per readout row for a + /// Exact operator cost per row it processes — per evaluation row for a /// read-time operation, per input row for an maintenance-time operation. pub exact_cost_per_row: Option, /// Rows the exact operator consumes per evaluation (read-time operation) or @@ -161,14 +161,14 @@ pub struct ExactCompositionCostInputs { pub expected_output_rows: Option, /// Cost of one update to the composed-with summary's maintained state. pub summary_maintenance_cost_per_update: Option, - /// Cost of one readout of that summary at evaluation time. + /// Cost of one evaluation of that summary at evaluation time. pub summary_read_cost: Option, /// Update (ingest) events per second reaching this site. pub update_rate: Option, /// Evaluations per second across every consumer of this site. pub evaluation_rate: Option, /// Cost of one full raw recompute of the target from pre-ASAP data — - /// the `KeepPreAsap` baseline's per-evaluation cost. + /// the kept-query baseline's per-evaluation cost. pub raw_recompute_cost: Option, /// Recurring formulas require `CostUnitsPerSecond`; totals yield no rate. pub unit: CostUnit, @@ -271,16 +271,16 @@ fn finite_rate(units_per_second: f64) -> Option { /// [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted) /// (via [`crate::replacement`]'s own `cse_preference`) the first time it /// needs a representative bound node for a sub-DAG that -/// [`asap_types::pre_asap::cse::share_common_sub_dags`] already collapsed +/// [`asap_types::ir::cse::share_common_sub_dags`] already collapsed /// onto one `Rc` for two or more workload roots. See /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { - /// The shared pre-ASAP sub-DAG itself. - pub sub_dag: &'a QueryExpr, - /// The `SummaryNode` this sub-DAG bound to — gives the cost model the + /// The shared sub-DAG itself. + pub sub_dag: &'a Rc, + /// The node this sub-DAG bound to — gives the cost model the /// concrete `FieldDataType`/`(kind, params)` actually at stake, not - /// just the pre-ASAP shape. - pub bound_summary: &'a SummaryNode, + /// just the logical shape. + pub bound_summary: &'a OperatorNode, /// How many workload roots reference this exact shared sub-DAG, counted /// once up front over the whole workload (always >= 2 — a candidate is /// only ever constructed for an actually-shared sub-DAG). @@ -302,7 +302,7 @@ pub struct Cost(pub f64); /// node. Node identity is preserved so whole-DAG models can bind per-state /// evidence without relying on traversal order. pub struct CostedSummaryDeployment<'a> { - pub summary: &'a SummaryNode, + pub summary: &'a OperatorNode, pub guarantee: &'a SummaryMaintenanceLifecycleGuarantee, pub selected_cost: Cost, } @@ -359,7 +359,7 @@ impl std::ops::Mul for Cost { /// [`CseCandidate`]. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ShareDecision { - /// Reuse one bound `SummaryNode` across every consumer. + /// Reuse one bound node across every consumer. Share, /// Bind each occurrence independently — the shared-maintenance cost /// isn't worth it for this candidate. @@ -367,23 +367,23 @@ pub enum ShareDecision { } /// Default [`CostModel::cse_recompute_cost`]: a structural-size proxy — the -/// number of *unique* nodes in `sub_dag`'s DAG -/// ([`asap_types::pre_asap::cse::dag_node_count`], the same module this +/// number of *unique* nodes in `sub-DAG`'s DAG +/// ([`asap_types::ir::cse::dag_node_count`], the same module this /// candidate's sharing was detected in). Deliberately **not** a raw -/// `serde_json` serialization length: after CSE, `sub_dag` generally has -/// internal sharing (a `CseCandidate` only exists because something got +/// `serde_json` serialization length: after CSE, `sub-DAG` is generally a +/// DAG, not a tree (a `CseCandidate` only exists because something got /// shared), and a naive full serialization re-serializes — over-counts — -/// any descendant `sub_dag` already shares internally, once per parent +/// any descendant `sub-DAG` already shares internally, once per parent /// that references it, instead of once for the whole DAG. `dag_node_count` /// dedupes by `Rc` pointer identity, so it charges each unique node's /// contribution exactly once regardless of how many places within -/// `sub_dag` reference it. Cheap to compute (one pass, no serialization), +/// `sub-DAG` reference it. Cheap to compute (one pass, no serialization), /// and still scales with real structural complexity — a genuinely tiny /// leaf costs little to recompute, a deep multi-join sub-DAG costs a lot. /// A deployment with real per-row/per-update cost knowledge should /// override [`CostModel::cse_recompute_cost`] instead of relying on this. -pub fn default_cse_recompute_cost(sub_dag: &QueryExpr) -> Cost { - Cost(asap_types::pre_asap::cse::dag_node_count(sub_dag) as f64) +pub fn default_cse_recompute_cost(sub_dag: &Rc) -> Cost { + Cost(asap_types::ir::cse::dag_node_count(sub_dag) as f64) } /// Default [`CostModel::cse_shared_maintenance_cost`]: a small @@ -506,7 +506,7 @@ pub trait CostModel { /// Estimated number of distinct subpopulations produced by `target`'s /// grouping keys. `None` means the deployment has no cardinality estimate; /// grouping alternatives remain legal but keep their discovery order. - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { + fn estimated_subpopulation_count(&self, _target: &OperatorNode) -> Option { None } @@ -518,7 +518,7 @@ pub trait CostModel { candidate: &ReplacementSubDAG, target: &TargetSubDAG<'_>, ) -> Option { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return None; }; let (kind, grouping) = sketch_state(node)?; @@ -545,17 +545,17 @@ pub trait CostModel { Realization::PassThrough } - /// Build the `SummaryEstimate` readout for an `Extension` intent this + /// Build the `SummaryEstimate` evaluation for an `Extension` intent this /// same `CostModel` realized as `Realization::Sketch` via /// [`realize_extension`](Self::realize_extension). Only ever called /// when `realize_extension` returned `Sketch` for the same - /// `(ext_kind, payload)` — `replacement::readout` has no other way to build a + /// `(ext_kind, payload)` — `replacement::evaluation` has no other way to build a /// `SketchStatistic` for a shape core doesn't know. A deployment that /// overrides `realize_extension` to return `Sketch` for some /// `ext_kind` MUST also override this for that same `ext_kind`, or /// this default panics loudly (rather than silently misinterpreting /// `payload`) the first time that intent is actually read out. - fn readout_extension( + fn evaluation_extension( &self, ext_kind: &str, _payload: &serde_json::Value, @@ -563,11 +563,11 @@ pub trait CostModel { ) -> SketchStatistic { unimplemented!( "CostModel::realize_extension returned Sketch for ext_kind={ext_kind:?} but \ - readout_extension wasn't overridden to match" + evaluation_extension wasn't overridden to match" ) } - /// Estimate the one-time cost of recomputing `candidate.sub_dag` + /// Estimate the one-time cost of recomputing `candidate.sub-DAG` /// independently at a single use site. Default: /// [`default_cse_recompute_cost`] (a structural-size proxy). See /// `docs/design_docs/cse-cost-model-decision.md`. @@ -581,7 +581,7 @@ pub trait CostModel { /// weight table), applied to whichever field of /// `candidate.bound_summary`'s output schema actually carries summary /// state (falls back to the cheapest, `Plain`, weight if none does — - /// e.g. `bound_summary` is a passthrough `KeepPreAsap` node with nothing + /// e.g. `bound_summary` is a kept non-ASAP sub-DAG with nothing /// summary-shaped to maintain). See `docs/design_docs/cse-cost-model-decision.md`. fn cse_shared_maintenance_cost(&self, candidate: &CseCandidate) -> Cost { let family = candidate @@ -598,7 +598,7 @@ pub trait CostModel { default_cse_shared_maintenance_cost(&family) } - /// Decide whether to reuse one shared `SummaryNode` across every + /// Decide whether to reuse one shared node across every /// consumer of `candidate`, or bind each occurrence independently — a /// Volcano/Cascades-style cost comparison (issue #237, #223 stage 4; see /// `docs/design_docs/cse-cost-model-decision.md`): share iff the estimated cost of @@ -666,7 +666,7 @@ pub trait CostModel { Cost(1.0) } - /// Cost of recomputing `candidate.sub_dag` once, from the pre-ASAP/raw + /// Cost of recomputing `candidate.sub-DAG` once, from the pre-ASAP/raw /// path. Units: cost units per recomputation — the `raw_recompute_cost` /// term of `recompute_cost_rate`. Default: delegates to /// [`cse_recompute_cost`](Self::cse_recompute_cost) (the same @@ -738,21 +738,21 @@ pub trait CostModel { /// on [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted)), /// not just order candidates against each other — that ordering job /// already belongs to [`rank_candidates`](Self::rank_candidates) (for a - /// [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) + /// [`ASAPStrategies`](crate::replacement::ASAPStrategies) /// group) and [`cse_share_decision`](Self::cse_share_decision) (for a /// [`SharedSubDAGStrategy`](crate::replacement::SharedSubDAGStrategy) /// group). /// /// One method covers both candidate shapes this crate ships: - /// `candidate.replacement`'s [`Replacement::Summary`] arm (a - /// `SketchAlgorithmStrategy` candidate — the bound `SummaryNode` is right - /// there, nothing to reconstruct) and its [`Replacement::Rewrite`] arm + /// `candidate.replacement`'s [`Replacement::SubDAG`] from a summary + /// realization (a `ASAPStrategies` candidate — the bound node is + /// right there, nothing to reconstruct) and the same arm from a rewrite /// (a `SharedSubDAGStrategy` share-vs-recompute candidate — no bound - /// `SummaryNode` of its own, since sharing is a decision about a target + /// summary of its own, since sharing is a decision about a target /// already bound some other way; a representative binding is recovered /// from `target` itself). `target` is threaded through explicitly /// (rather than only ever the target embedded in `candidate` — there - /// isn't one for a `Rewrite`) so both arms have the `consumer_count` + /// isn't one for a rewrite) so both arms have the `consumer_count` /// context a cost estimate needs to be meaningful. /// /// Default: **not a real cost model** — always returns `f64::NAN`. @@ -776,7 +776,7 @@ pub trait CostModel { /// optimistic zeroes. fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs::default() } @@ -786,7 +786,7 @@ pub trait CostModel { /// rate so the horizon integral equals one peak-capacity charge. fn summary_maintenance_lifecycle_cost_inputs_for_horizon( &self, - summary: &SummaryNode, + summary: &OperatorNode, _horizon: Option, ) -> SummaryMaintenanceLifecycleCostInputs { self.summary_maintenance_lifecycle_cost_inputs(summary) @@ -796,7 +796,7 @@ pub trait CostModel { /// conservative default advertises no long-lived maintenance capability. fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities::default() } @@ -807,8 +807,8 @@ pub trait CostModel { /// must not then reuse the partial per-state sum. fn complete_summary_candidate_cost( &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, + _root: &OperatorNode, + _target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], _horizon: Option, _expected_reads: Option, @@ -827,8 +827,8 @@ pub trait CostModel { /// that do not perform either decision. fn complete_summary_candidate_estimate( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -861,7 +861,7 @@ pub trait CostModel { /// Cost of evaluating `target` directly from its logical/raw inputs once. /// When known, lifecycle-aware materialization compares this fallback with /// the aggregate cost of the selected summary deployments. - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &OperatorNode) -> Option { None } @@ -870,7 +870,7 @@ pub trait CostModel { /// cardinality changes between evaluations. fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &OperatorNode, expected_reads: f64, ) -> Option { self.raw_query_recompute_cost(target) @@ -879,7 +879,7 @@ pub trait CostModel { /// Physical feasibility evidence for a complete summary candidate. /// `None` defers admission to physical/deployment compilation; `Some(false)` /// excludes the candidate without changing its computation or parameters. - fn summary_support_evidence(&self, _summary: &SummaryNode) -> Option { + fn summary_support_evidence(&self, _summary: &OperatorNode) -> Option { None } @@ -932,7 +932,7 @@ pub trait CostModel { /// /// Default: every input unknown ([`ExactCompositionCostInputs::unknown`]) /// — unknown is never zero, and with no rate derivable - /// `CandidateLogicalASAPDAGs::global_selection` keeps the conservative `KeepPreAsap` + /// `CandidateLogicalASAPDAGs::global_selection` keeps the conservative keep-as-is /// behavior for the site. A deployment that wants defaults must supply /// them here explicitly. fn exact_composition_cost_inputs( @@ -948,14 +948,16 @@ pub trait CostModel { } fn sketch_state( - node: &SummaryNode, + node: &OperatorNode, ) -> Option<(&asap_types::post_asap::SketchKind, &GroupingStrategy)> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_state(summary_input), - SummaryExpr::SummaryAgg { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + sketch_state(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, grouping), .. - } => Some((kind, grouping)), + }) => Some((kind, grouping)), _ => None, } } @@ -1027,15 +1029,17 @@ impl CostModel for DefaultCostModel { /// `cse_share_decision`'s default body already composes — rather than a /// second formula: /// - /// - [`Replacement::Summary`]: `cse_recompute_cost` (the one-time + /// - A [`ReplacementProvenance::SummaryRealization`] candidate (a + /// `ASAPStrategies` binding): `cse_recompute_cost` (the one-time /// structural cost of building `target` at all) plus /// `cse_shared_maintenance_cost` of the candidate's own bound family /// (a pricier family — a sketch over an exact accumulator, say — /// costs more here, consistent with the per-family weighting /// [`default_cse_shared_maintenance_cost`] already orders candidates /// by). - /// - [`Replacement::Rewrite`]: recovers one representative bound - /// `SummaryNode` for `target` via `realize_child` (the same + /// - Any other [`Replacement::SubDAG`] (a logical rewrite or a CSE + /// share/recompute candidate): recovers one + /// representative bound node for `target` via `realize_child` (the same /// rank-and-take-first helper `replacement::realize_child` reuses for the /// identical need), then charges /// `cse_shared_maintenance_cost` for the candidate that shares @@ -1060,7 +1064,9 @@ impl CostModel for DefaultCostModel { fn estimate_cost(&self, candidate: &ReplacementSubDAG, target: &TargetSubDAG<'_>) -> f64 { let consumer_count = target.consumer_count.max(1); match &candidate.replacement { - Replacement::Summary(node) => { + Replacement::SubDAG(node) + if candidate.provenance == ReplacementProvenance::SummaryRealization => + { let cse = CseCandidate { sub_dag: target.root, bound_summary: node, @@ -1068,7 +1074,7 @@ impl CostModel for DefaultCostModel { }; (self.cse_recompute_cost(&cse) + self.cse_shared_maintenance_cost(&cse)).0 } - Replacement::Rewrite(rc) + Replacement::SubDAG(rc) if candidate.provenance == ReplacementProvenance::AccuracyReconciliation => { let Ok(sibling_bound) = realize_child(rc, self) else { @@ -1087,7 +1093,7 @@ impl CostModel for DefaultCostModel { }; self.cse_shared_maintenance_cost(&cse).0 } - Replacement::Rewrite(rc) => { + Replacement::SubDAG(rc) => { let Ok(bound) = realize_child(target.root, self) else { return f64::NAN; }; @@ -1320,11 +1326,11 @@ mod tests { assert_eq!( DefaultCostModel.value_operation_support_evidence( &ExactOperation::Aggregate { - reduction: asap_types::pre_asap::query_expr::Reduction::by(vec![]), + reduction: asap_types::ir::operator_properties::Reduction::by(vec![]), measures: vec![AggIntent::Max { col: None }], output_names: vec![], - having: None, filters: vec![], + having: None, }, OperationPlacement::Read, ), @@ -1346,14 +1352,15 @@ mod tests { // ── CSE sharing (issue #237, #223 stage 4) ────────────────────────── + use asap_types::ir::operator_properties::Source; + use asap_types::ir::{NonASAPOp, Predicate, ScalarExpr}; use asap_types::post_asap::{ - ExactKind, ExactParams, Field, GroupingStrategy, Schema, SketchKind, SummaryExpr, + ExactKind, ExactParams, Field, GroupingStrategy, Schema, SketchKind, }; - use asap_types::pre_asap::query_expr::Source; use asap_types::pre_asap::schema::DataType; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -1364,37 +1371,39 @@ mod tests { 0, vec![], ), - } - } - - fn summary_node(family: FieldDataType) -> SummaryNode { - SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: std::rc::Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: Schema::lifted(vec![], None), - guarantee: None, + })) + .unwrap() + } + + /// A `SummaryAgg` directly over the kept `scan()` sub-DAG. + fn summary_node(family: FieldDataType) -> Rc { + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: scan(), + family: family.clone(), + input: asap_types::post_asap::SummaryUpdate::column( + asap_types::pre_asap::expr_ir::ColumnRef::Named("value".into()), + ), + reduction: asap_types::ir::operator_properties::Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, }), - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column( - asap_types::pre_asap::expr_ir::ColumnRef::Named("value".into()), - ), - reduction: asap_types::pre_asap::query_expr::Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: None, - } + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(None), + ) } #[test] fn default_recompute_cost_is_positive_and_grows_with_structural_size() { let leaf = scan(); - let nested = QueryExpr::Dedup { - cols: vec![0], - child: std::rc::Rc::new(leaf.clone()), - }; + let nested = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: Rc::clone(&leaf), + })) + .unwrap(); assert!(default_cse_recompute_cost(&leaf) > Cost::ZERO); assert!(default_cse_recompute_cost(&nested) > default_cse_recompute_cost(&leaf)); } @@ -1407,27 +1416,27 @@ mod tests { /// identity-blind recursive walk) would count it. #[test] fn default_recompute_cost_does_not_double_count_an_internally_shared_descendant() { + use asap_types::ir::operator_properties::JoinKind; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::{JoinKind, Predicate}; - let true_pred = || { - Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))) - }; - let shared_leaf = std::rc::Rc::new(scan()); - let no_sharing = QueryExpr::Join { - kind: JoinKind::Inner, - pred: true_pred(), - left: std::rc::Rc::new(scan()), - right: std::rc::Rc::new(scan()), - }; - let with_sharing = QueryExpr::Join { - kind: JoinKind::Inner, - pred: true_pred(), - left: std::rc::Rc::clone(&shared_leaf), - right: std::rc::Rc::clone(&shared_leaf), - }; + let true_pred = || Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))); + let shared_leaf = scan(); + let no_sharing = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left: scan(), + right: scan(), + })) + .unwrap(); + let with_sharing = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left: Rc::clone(&shared_leaf), + right: Rc::clone(&shared_leaf), + })) + .unwrap(); assert_eq!( default_cse_recompute_cost(&no_sharing), Cost(3.0), @@ -1555,20 +1564,20 @@ mod tests { } } - let root = Rc::new(scan()); + let root = scan(); let target = TargetSubDAG::new(&root); let candidate = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(FieldDataType::Plain( + replacement: Replacement::SubDAG(summary_node(FieldDataType::Plain( asap_types::pre_asap::DataType::Float64, - )))), + ))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: "whatever".into(), }; assert!(RankOnly.estimate_cost(&candidate, &target).is_nan()); } - /// `DefaultCostModel::estimate_cost` for a [`Replacement::Summary`] + /// `DefaultCostModel::estimate_cost` for a summary-rooted [`Replacement::SubDAG`] /// candidate reuses [`default_cse_shared_maintenance_cost`]'s own /// per-family ordering: a candidate bound to a cheap-to-maintain family /// (an exact accumulator) must cost less than one bound to an @@ -1578,25 +1587,26 @@ mod tests { /// above. #[test] fn estimate_cost_for_summary_orders_candidates_by_family_cheapest_to_priciest() { - let root = Rc::new(scan()); + let root = scan(); let target = TargetSubDAG::new(&root); let cheap = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node( - FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + replacement: Replacement::SubDAG(summary_node(FieldDataType::ExactAggregate( + ExactKind::Sum, + ExactParams::Sum, ))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: "exact accumulator".into(), }; let pricey = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(FieldDataType::StatModel( + replacement: Replacement::SubDAG(summary_node(FieldDataType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { family: "gaussian_mixture".into(), }, - )))), + ))), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: "fitted statistical model".into(), }; @@ -1614,7 +1624,7 @@ mod tests { ); } - /// `DefaultCostModel::estimate_cost` for a [`Replacement::Rewrite`] pair + /// `DefaultCostModel::estimate_cost` for a relational [`Replacement::SubDAG`] pair /// (the `SharedSubDAGStrategy` share-vs-recompute shape) agrees with /// what `cse_share_decision` would already pick for the same target: with /// many consumers of a cheap-to-recompute leaf, the "share" candidate @@ -1625,18 +1635,18 @@ mod tests { /// directly. #[test] fn estimate_cost_for_rewrite_prefers_sharing_when_recompute_dominates_maintenance() { - let target_root = Rc::new(scan()); + let target_root = scan(); let target = TargetSubDAG::with_consumer_count(&target_root, 20); let share = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::clone(&target_root)), + replacement: Replacement::SubDAG(Rc::clone(&target_root)), provenance: crate::replacement::ReplacementProvenance::CseShare, rationale: "build once and share".into(), }; let recompute = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new((*target_root).clone())), + replacement: Replacement::SubDAG(Rc::new((*target_root).clone())), provenance: crate::replacement::ReplacementProvenance::CseRecompute, rationale: "build independently".into(), }; diff --git a/crates/asap-aware-mapping/src/empirical_comparison.rs b/crates/asap-aware-mapping/src/empirical_comparison.rs index ecd2a6cd4..17e3e9a82 100644 --- a/crates/asap-aware-mapping/src/empirical_comparison.rs +++ b/crates/asap-aware-mapping/src/empirical_comparison.rs @@ -41,7 +41,7 @@ pub struct OfflineExactMeasurement { } /// The companion format binds otherwise query-agnostic sketch primitives to -/// their measured readout and exact reference. Bindings describe state after +/// their measured evaluation and exact reference. Bindings describe state after /// ingestion, without merges or intervening updates during the read sequence. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -247,7 +247,9 @@ pub fn recommend_offline( || error.query.get("value_type").and_then(|v| v.as_str()) != Some(request.query.value_type.as_str()) { - return Err("offline error observation has incompatible readout semantics".into()); + return Err( + "offline error observation has incompatible evaluation semantics".into(), + ); } if error.metric != request.accuracy.metric || error.trials < request.accuracy.minimum_trials @@ -802,14 +804,14 @@ mod tests { } assert!(recommend_offline(&evidence, &request).is_err(), "{case}"); } - for case in ["metric", "trials", "binding", "readout"] { + for case in ["metric", "trials", "binding", "evaluation"] { let mut evidence = evidence.clone(); let mut request = request.clone(); match case { "metric" => request.accuracy.metric = "rank_error".into(), "trials" => request.accuracy.minimum_trials = 100, "binding" => evidence.query_bindings.clear(), - "readout" => { + "evaluation" => { for row in &mut evidence.sketch_evidence.records { row.error.as_mut().unwrap().query["kind"] = serde_json::json!("total_count"); diff --git a/crates/asap-aware-mapping/src/empirical_cost.rs b/crates/asap-aware-mapping/src/empirical_cost.rs index b6d4f9a08..18d1d7ee4 100644 --- a/crates/asap-aware-mapping/src/empirical_cost.rs +++ b/crates/asap-aware-mapping/src/empirical_cost.rs @@ -2,9 +2,8 @@ //! configuration and environment; they are neither runtime feedback nor proofs //! of an accuracy guarantee. CPU quantities are nanoseconds, never CPU operations. -use asap_types::post_asap::{ - FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, -}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; +use asap_types::post_asap::{FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams}; use asap_types::pre_asap::AggIntent; use serde::{Deserialize, Serialize}; @@ -214,13 +213,13 @@ impl EmpiricalEvidenceProvider { /// mixed with an existing deployment's unitless or CPU-operation costs. pub fn lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, GroupingStrategy::PerSubpopulationInstance), grouping: GroupingStrategy::PerSubpopulationInstance, .. - } = &summary.expr + }) = &summary.operator else { return SummaryMaintenanceLifecycleCostInputs::default(); }; @@ -280,7 +279,7 @@ impl CostModel for EmpiricalCostModel { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { self.provider.lifecycle_cost_inputs(summary) } diff --git a/crates/asap-aware-mapping/src/exact_composition.rs b/crates/asap-aware-mapping/src/exact_composition.rs index 54bce26f6..f48bc08b5 100644 --- a/crates/asap-aware-mapping/src/exact_composition.rs +++ b/crates/asap-aware-mapping/src/exact_composition.rs @@ -4,21 +4,23 @@ //! `construct_summary_agg` already nests accumulator realizations, such as //! KLL over exact `Sum` state or a quantile over `Rate` state. This strategy //! covers the more general cases where an exact function must consume a -//! summary readout, or where a maintained summary consumes the values of an +//! summary evaluation, or where a maintained summary consumes the values of an //! exact function that has no accumulator realization. //! -//! Both cases use the general [`SummaryExpr::ValueOperation`] node. Its -//! semantic [`ValueOperation`] is independent from [`ExecutionTiming`], so -//! adding a function does not require adding a new physical node type. +//! Both cases use an ordinary `NonASAPOp::Aggregate` node over the child +//! plan. The node carries no timing: it runs when its consumer runs, so the +//! same operator serves both placements and adding a function does not +//! require adding a new physical node type. [`OperationPlacement`] is the +//! search-time placement choice. //! //! ## Reference, don't select //! //! A composed candidate needs a child plan to compose *with* — the inner -//! quantile's own summary readout, say. This strategy deliberately does +//! quantile's own summary evaluation, say. This strategy deliberately does //! **not** pick that child itself (the way `construct_summary_agg`'s //! `realize_child` takes the head of the child's own ranking): a //! [`Replacement::ExactComposition`] carries only the child *target* -//! (`ExactComposition::child_target`, the same `Rc` whose +//! (`ExactComposition::child_target`, the same `Rc` whose //! `TargetSubDAGCandidates` in `CandidateLogicalASAPDAGs` already holds every candidate for it). It is //! [`CandidateLogicalASAPDAGs::global_selection`](crate::replacement::CandidateLogicalASAPDAGs::global_selection) //! that commits the compatible parent/child pair — so the child's own @@ -34,7 +36,7 @@ //! //! - the target is a single-measure, `HAVING`-free exact aggregate; //! - read-time operation: the child is a bindable aggregate that has at least one -//! readout-producing summary implementation (a sketch/sample/wavelet/ +//! evaluation-producing summary implementation (a sketch/sample/wavelet/ //! model — the shapes a maintained accumulator can't sit above), and the //! target's grouping keys resolve in the child's output schema; //! transform: the target is a per-entity exact function with no @@ -58,18 +60,21 @@ //! - Decide whether a composition is *worth it*: that is //! `global_selection`'s job, using the issue's cost-units-per-second //! formulas (see `crate::cost_model::read_operation_plan_cost_rate` and -//! siblings). Missing statistics keep the conservative `KeepPreAsap`. +//! siblings). Missing statistics keep the conservative kept sub-DAG. +use asap_types::ir::non_asap::any_measure_filtered; use std::rc::Rc; -use asap_types::post_asap::execution_data_state::validate_execution_data_states_at; +use asap_types::ir::aggregate_schema::aggregate_output_schema; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::timing::{planned_data_state, validate_default}; +use asap_types::ir::{NonASAPOp, Operator, OperatorNode, Predicate}; +use asap_types::post_asap::execution_data_state::lift_plain; use asap_types::post_asap::{ - exact_operation_output_schema, produced_data_state, AccuracyError, ExactOperation, - ExecutionDataState, ExecutionDataStateError, ResultGuarantee, Schema, SummaryExpr, SummaryNode, - ValueOperation, + AccuracyError, ExactOperationSchemaError, ExecutionDataState, ExecutionDataStateError, + ResultGuarantee, Schema, }; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, QueryExpr, Reduction}; use asap_types::types::AccuracyTarget; use crate::cost_model::CostModel; @@ -84,9 +89,61 @@ use asap_types::post_asap::ExecutionTiming; /// Which side of the maintenance/read boundary an [`ExactComposition`]'s /// exact function executes on. +/// The exact function an [`ExactComposition`] applies: the parameters of +/// the `NonASAPOp::Aggregate` node the composition builds over its child. +#[derive(Debug, Clone, PartialEq)] +pub enum ExactOperation { + Aggregate { + reduction: Reduction, + measures: Vec, + output_names: Vec, + filters: Vec>, + having: Option, + }, +} + +impl ExactOperation { + /// Output schema of this operation over a child whose edge carries + /// `input` — the same canonical derivation the pre-ASAP `Aggregate` + /// node uses. `Err` when the child carries non-plain state the operator + /// cannot read. + pub fn output_schema(&self, input: &Schema) -> Result { + if !input.is_all_plain() { + return Err(ExactOperationSchemaError::NonPlainInput); + } + let plain = lift_plain(input); + let ExactOperation::Aggregate { + reduction, + measures, + output_names, + .. + } = self; + let out = aggregate_output_schema(&plain, reduction, measures, output_names)?; + Ok(lift_plain(&out)) + } + + fn into_op(self, child: Rc) -> NonASAPOp { + let ExactOperation::Aggregate { + reduction, + measures, + output_names, + filters, + having, + } = self; + NonASAPOp::Aggregate { + reduction, + measures, + output_names, + filters, + having, + child, + } + } +} + #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] pub enum OperationPlacement { - /// After the child's summary readout. + /// After the child's summary evaluation. Read, /// On the maintenance path, feeding /// maintained state above. @@ -120,7 +177,7 @@ pub struct ExactComposition { pub op: ExactOperation, /// The pre-ASAP child the operator consumes; its `TargetSubDAGCandidates` holds the /// candidates `global_selection` may commit this composition with. - pub child_target: Rc, + pub child_target: Rc, /// The composed node's output schema — the target's own pre-ASAP /// output schema, lifted with every column `Plain` (an exact operator /// only ever produces plain values). @@ -128,24 +185,27 @@ pub struct ExactComposition { } impl ExactComposition { + /// The data state `child` produces when this operation (its consumer) + /// runs at the placement's timing. + fn child_data_state(&self, child: &Rc) -> ExecutionDataState { + planned_data_state(child, self.placement.data_state().timing) + } + /// Can `child` legally be this composition's input? Phase legality - /// (the child's produced data_state — a `KeepPreAsap` leaf takes the + /// (the child's produced data_state — a kept pre-ASAP sub-DAG takes the /// phase this edge assigns) plus the plain-operand rule, checked /// through the same schema derivation [`Self::compose`] uses. - pub fn accepts_child(&self, child: &SummaryNode) -> bool { - let phase_ok = match produced_data_state(&child.expr) { - None => true, - Some(avail) => avail == self.placement.data_state(), - }; - phase_ok && exact_operation_output_schema(&self.op, &child.schema).is_ok() + pub fn accepts_child(&self, child: &Rc) -> bool { + self.child_data_state(child) == self.placement.data_state() + && self.op.output_schema(&child.schema).is_ok() } /// Build the composed, data_state-validated node over `child`. Every edge of /// the result (including everything beneath `child`) is checked by - /// `asap_types::post_asap::validate_execution_data_states`; an illegal + /// `asap_types::ir::timing::validate_default`; an illegal /// placement is a typed [`RealizationError::ExecutionDataState`], never deferred to a /// runtime. - pub fn compose(&self, child: Rc) -> Result, RealizationError> { + pub fn compose(&self, child: Rc) -> Result, RealizationError> { self.compose_with_accuracy(child, &DefaultAccuracyModel) } @@ -154,24 +214,23 @@ impl ExactComposition { /// unsupported folds fail closed with a typed accuracy error. pub fn compose_with_accuracy( &self, - child: Rc, + child: Rc, accuracy_model: &dyn AccuracyModel, - ) -> Result, RealizationError> { - if let Some(produced) = produced_data_state(&child.expr) { - if produced != self.placement.data_state() { - let edge = match self.placement { - OperationPlacement::Maintenance => "ValueOperation.child (maintenance time)", - OperationPlacement::Read => "ValueOperation.child (read time)", - }; - return Err(RealizationError::ExecutionDataState( - ExecutionDataStateError::IllegalChildDataState { - edge, - child: produced, - }, - )); - } + ) -> Result, RealizationError> { + let produced = self.child_data_state(&child); + if produced != self.placement.data_state() { + let edge = match self.placement { + OperationPlacement::Maintenance => "exact operation child (maintenance time)", + OperationPlacement::Read => "exact operation child (read time)", + }; + return Err(RealizationError::ExecutionDataState( + ExecutionDataStateError::IllegalChildDataState { + edge, + child: produced, + }, + )); } - let schema = exact_operation_output_schema(&self.op, &child.schema)?; + let schema = self.op.output_schema(&child.schema)?; let guarantee = match &child.guarantee { None => None, Some(input) if input.is_exact() => Some(ResultGuarantee::exact(format!( @@ -198,23 +257,11 @@ impl ExactComposition { None => None, }, }; - let timing = match self.placement { - OperationPlacement::Read => asap_types::post_asap::ExecutionTiming::QueryTime, - OperationPlacement::Maintenance => { - asap_types::post_asap::ExecutionTiming::IngestionTime - } - }; - let expr = SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Exact(self.op.clone()), - timing, - }; - let node = Rc::new(SummaryNode { - expr, - schema, - guarantee, - }); - validate_execution_data_states_at(&node, self.placement.data_state())?; + let node = Rc::new( + OperatorNode::with_schema(Operator::NonASAP(self.op.clone().into_op(child)), schema) + .with_guarantee(guarantee), + ); + validate_default(&node, self.placement.data_state().timing)?; Ok(node) } @@ -227,7 +274,7 @@ impl ExactComposition { } } -/// Which exact reducers may run as a query-time fold over readout rows. +/// Which exact reducers may run as a query-time fold over evaluation rows. /// `Count` only at `Exact` accuracy (an approximate count is a sketch /// target, not an exact fold). fn is_query_time_reducer(intent: &AggIntent) -> bool { @@ -245,9 +292,9 @@ fn is_query_time_reducer(intent: &AggIntent) -> bool { ) } -/// Does `implementation` need a `SummaryEstimate` readout to yield a value +/// Does `implementation` need a `SummaryEstimate` evaluation to yield a value /// — i.e. is it a shape a maintained accumulator can't legally sit above? -fn needs_readout(implementation: &Realization) -> bool { +fn needs_evaluation(implementation: &Realization) -> bool { matches!( implementation, Realization::Sketch(_) @@ -259,17 +306,17 @@ fn needs_readout(implementation: &Realization) -> bool { /// The `(op, child)` of a read-time operation-shaped target, or `None`. fn query_time_shape( - root: &QueryExpr, + root: &OperatorNode, cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { +) -> Option<(ExactOperation, Rc, AggIntent)> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names, filters, having: None, child, - } = root + }) = root.non_asap() else { return None; }; @@ -291,13 +338,12 @@ fn query_time_shape( let child_intent = bindable_intent(child)?; if !realizations_for_intent(child_intent, cost_model) .iter() - .any(needs_readout) + .any(needs_evaluation) { return None; } // Grouping keys must resolve in the child's output schema — the same // derivation the composed node's own schema will use. - root.output_schema().ok()?; Some(( ExactOperation::Aggregate { reduction: reduction.clone(), @@ -314,17 +360,17 @@ fn query_time_shape( /// The `(op, child)` of a function-shaped target — a per-entity exact /// transform with no accumulator form — or `None`. fn ingestion_time_shape( - root: &QueryExpr, + root: &OperatorNode, cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { +) -> Option<(ExactOperation, Rc, AggIntent)> { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, output_names, filters, having: None, child, - } = root + }) = root.non_asap() else { return None; }; @@ -346,7 +392,6 @@ fn ingestion_time_shape( { return None; } - root.output_schema().ok()?; Some(( ExactOperation::Aggregate { reduction: Reduction::PerEntity, @@ -386,10 +431,7 @@ impl<'a> ExactCompositionStrategy<'a> { } fn candidates(&self, target: &TargetSubDAG<'_>) -> Vec { - let Ok(schema) = target.root.output_schema() else { - return Vec::new(); - }; - let schema = asap_types::post_asap::execution_data_state::lift_plain(&schema); + let schema = lift_plain(&target.root.schema); let mut out = Vec::new(); if let Some((op, child, intent)) = query_time_shape(target.root, self.cost_model) { @@ -410,10 +452,10 @@ impl<'a> ExactCompositionStrategy<'a> { }), provenance: ReplacementProvenance::ValueOperationAtQueryTime, rationale: format!( - "{} is an exact fold whose input is the readout of {} — a maintained \ - accumulator cannot consume query-time values, so instead of collapsing \ - the whole DAG into KeepPreAsap this applies the fold as an \ - ExactRead over whichever summary readout global_selection \ + "{} is an exact fold whose input is the evaluation of {} — a maintained \ + accumulator cannot consume query-time values, so instead of keeping \ + the whole tree pre-ASAP this applies the fold as an \ + ExactRead over whichever summary evaluation global_selection \ commits for the child target (asap_aware_mapping::exact_composition)", describe_intent(&intent), child_desc @@ -441,7 +483,7 @@ impl<'a> ExactCompositionStrategy<'a> { "{} is an exact per-entity function with no accumulator form; as an \ explicit ExactMaintenance on the update path its output can feed a \ maintained summary above it instead of being handed over as an opaque \ - raw KeepPreAsap blob (asap_aware_mapping::exact_composition)", + raw kept sub_dag (asap_aware_mapping::exact_composition)", describe_intent(&intent) ), }); @@ -465,59 +507,20 @@ impl ReplacementStrategy for ExactCompositionStrategy<'_> { mod tests { use super::*; use crate::cost_model::{DefaultCostModel, ValueOperationCapabilities}; - use crate::replacement::keep_pre_asap; + use crate::replacement::retain_exact; + use crate::test_support::{agg, agg_per_entity as per_entity, metric_scan, timed}; + use asap_types::ir::ASAPOp; use asap_types::post_asap::{ExecutionDataStateError, FieldDataType, SketchAlgorithm}; use asap_types::pre_asap::agg_intent::default_quantile; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{DataType, Field, Schema}; - - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - - fn per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } /// `max by (zone) (quantile by (zone, host) (m))`. - fn max_over_quantile() -> Rc { + fn max_over_quantile() -> Rc { let inner = agg( vec![2, 3], default_quantile(0.99), metric_scan(&["zone", "host"]), ); - Rc::new(agg(vec![0], AggIntent::Max { col: None }, inner)) + agg(vec![0], AggIntent::Max { col: None }, inner) } #[test] @@ -539,7 +542,7 @@ mod tests { candidates[0].provenance, ReplacementProvenance::ValueOperationAtQueryTime ); - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = root.non_asap() else { unreachable!() }; assert!( @@ -553,7 +556,7 @@ mod tests { #[test] fn proposes_query_time_operation_for_avg_over_quantile_alongside_the_rewrite() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); - let root = Rc::new(agg(vec![0], AggIntent::Avg { col: None }, inner)); + let root = agg(vec![0], AggIntent::Avg { col: None }, inner); let target = TargetSubDAG::new(&root); assert_eq!( ExactCompositionStrategy::default_cost_model() @@ -567,7 +570,7 @@ mod tests { #[test] fn proposes_ingestion_time_operation_for_a_per_entity_pass_through_over_raw_input() { - let root = Rc::new(per_entity(AggIntent::Deriv, metric_scan(&["zone"]))); + let root = per_entity(AggIntent::Deriv, metric_scan(&["zone"])); let target = TargetSubDAG::new(&root); let candidates = ExactCompositionStrategy::default_cost_model().replacements(&target); assert_eq!(candidates.len(), 1); @@ -579,21 +582,21 @@ mod tests { #[test] fn does_not_propose_for_shapes_already_covered_by_accumulators() { - // sum by (zone) over an exact Sum child: the child has no readout, + // sum by (zone) over an exact Sum child: the child has no evaluation, // so SummaryAgg(Sum) over SummaryAgg(Sum) is already legal. let inner = agg( vec![2, 3], AggIntent::Sum { col: None }, metric_scan(&["zone", "host"]), ); - let root = Rc::new(agg(vec![0], AggIntent::Sum { col: None }, inner)); + let root = agg(vec![0], AggIntent::Sum { col: None }, inner); assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&root))); // rate is an exact accumulator — directly nestable, no separate value operation. - let rate = Rc::new(per_entity(AggIntent::Rate, metric_scan(&[]))); + let rate = per_entity(AggIntent::Rate, metric_scan(&[])); assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&rate))); // A sketch-capable outer intent is not an exact fold. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["zone"])); - let root = Rc::new(agg(vec![0], default_quantile(0.99), inner)); + let root = agg(vec![0], default_quantile(0.99), inner); assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&root))); } @@ -618,7 +621,7 @@ mod tests { let strategy = ExactCompositionStrategy::new(&NoMixedExecution); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); - let deriv = Rc::new(per_entity(AggIntent::Deriv, metric_scan(&[]))); + let deriv = per_entity(AggIntent::Deriv, metric_scan(&[])); assert!(!strategy.matches(&TargetSubDAG::new(&deriv))); } @@ -630,12 +633,13 @@ mod tests { let Replacement::ExactComposition(comp) = &candidates[0].replacement else { unreachable!() }; - // A bare SummaryAgg (state, no readout) is not a legal read-time operation + // A bare SummaryAgg (state, no evaluation) is not a legal read-time operation // input — the operator would be consuming sketch state. let state_child = crate::replacement::realize_child(&comp.child_target, &DefaultCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &state_child.expr else { - panic!("expected the child to realize to a readout"); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &state_child.operator + else { + panic!("expected the child to realize to a evaluation"); }; assert!(!comp.accepts_child(summary_input)); assert!(matches!( @@ -644,7 +648,7 @@ mod tests { ExecutionDataStateError::IllegalChildDataState { .. } )) )); - // The readout itself is accepted and composes to a plain schema. + // The evaluation itself is accepted and composes to a plain schema. assert!(comp.accepts_child(&state_child)); let composed = comp.compose(state_child).unwrap(); assert!( @@ -652,12 +656,13 @@ mod tests { "rank error has no registered conversion through max" ); assert!(matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } + composed.operator, + Operator::NonASAP(NonASAPOp::Aggregate { .. }) )); + // Timing is no longer stored by composition: under the default + // lifecycle assignment the composed read-time operation runs at + // query time. + assert_eq!(timed(&composed).timing, Some(ExecutionTiming::QueryTime)); assert!(composed .schema .fields @@ -666,32 +671,78 @@ mod tests { } #[test] - fn compose_rejects_a_readout_child_for_a_ingestion_time_operation() { + fn compose_rejects_a_evaluation_child_for_a_ingestion_time_operation() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); - let root = Rc::new(per_entity(AggIntent::Deriv, inner)); + let root = per_entity(AggIntent::Deriv, inner); let candidates = ExactCompositionStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); let Replacement::ExactComposition(comp) = &candidates[0].replacement else { unreachable!() }; - let readout = + let evaluation = crate::replacement::realize_child(&comp.child_target, &DefaultCostModel).unwrap(); - assert!(!comp.accepts_child(&readout)); + assert!(!comp.accepts_child(&evaluation)); assert!(matches!( - comp.compose(readout), + comp.compose(evaluation), Err(RealizationError::ExecutionDataState( ExecutionDataStateError::IllegalChildDataState { .. } )) )); // Raw update input is fine. - let raw = keep_pre_asap(&comp.child_target).unwrap(); + let raw = retain_exact(&comp.child_target).unwrap(); assert!(comp.accepts_child(&raw)); + // Timing is no longer stored by composition: the composition's + // placement is maintenance time, the composed exact operation is a + // plain Aggregate over the raw rows, and it is legal (and planned to + // run) at ingestion time. + assert_eq!(comp.placement, OperationPlacement::Maintenance); + let composed = comp.compose(raw).unwrap(); assert!(matches!( - comp.compose(raw).unwrap().expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::IngestionTime, - .. - } + composed.operator, + Operator::NonASAP(NonASAPOp::Aggregate { .. }) + )); + validate_default(&composed, ExecutionTiming::IngestionTime).unwrap(); + assert_eq!( + planned_data_state(&composed, ExecutionTiming::IngestionTime).timing, + ExecutionTiming::IngestionTime + ); + } + + fn max_op(by: Vec) -> ExactOperation { + ExactOperation::Aggregate { + reduction: Reduction::by(by), + measures: vec![AggIntent::Max { col: None }], + output_names: vec![], + filters: vec![], + having: None, + } + } + + #[test] + fn exact_operator_schema_matches_pre_asap_aggregate_derivation() { + let child = lift_plain(&metric_scan(&["zone"]).schema); + let out = max_op(vec![2]).output_schema(&child).unwrap(); + let names: Vec<_> = out.fields.iter().map(|f| f.name.as_str()).collect(); + assert_eq!(names, vec!["zone", "max"]); + assert!(out.is_all_plain()); + } + + #[test] + fn exact_operator_rejects_non_plain_input() { + let state = Schema::lifted( + vec![asap_types::pre_asap::Field::new( + "state", + FieldDataType::ExactAggregate( + asap_types::post_asap::ExactKind::Sum, + asap_types::post_asap::ExactParams::Sum, + ), + false, + )], + None, + ); + assert!(matches!( + max_op(vec![]).output_schema(&state), + Err(ExactOperationSchemaError::NonPlainInput) )); } } diff --git a/crates/asap-aware-mapping/src/explanation.rs b/crates/asap-aware-mapping/src/explanation.rs index e7afff2ba..bc67be96f 100644 --- a/crates/asap-aware-mapping/src/explanation.rs +++ b/crates/asap-aware-mapping/src/explanation.rs @@ -31,7 +31,7 @@ //! collapses into a single question this module asks of *that* data instead: //! **for a given `TargetSubDAG`, does its candidate list contain anything //! other than the trivial, no-op realization?** A `TargetSubDAG` whose only -//! candidate is "the one thing `SketchAlgorithmStrategy` would have committed +//! candidate is "the one thing `ASAPStrategies` would have committed //! to anyway, with no alternative" has no optimization to report — that //! candidate isn't an *opportunity*, it's just the target's existing shape //! reflected back. A `TargetSubDAG` with more than one candidate (several @@ -43,9 +43,9 @@ //! [`TargetSubDAGCandidates`]s into that shape: //! //! - [`ExplanationKind::SketchApproximation`] — the `TargetSubDAG`'s -//! candidate list contains at least one [`Replacement::Summary`] that +//! candidate list contains at least one summary-realization [`Replacement::SubDAG`] that //! actually realizes a sketch family (`FieldDataType::Sketch`), i.e. -//! [`SketchAlgorithmStrategy`] found something to offer beyond whatever +//! [`ASAPStrategies`] found something to offer beyond whatever //! exact/pass-through candidate [`crate::replacement`]'s own //! `realizations_for_intent` would have committed to on its own. //! - [`ExplanationKind::CommonSubexpressionReuse`] — the `TargetSubDAG` @@ -103,7 +103,7 @@ //! [`ReplacementStrategy`] already *is* that extension point, one layer //! down, and [`explain_replacements_with`]'s own `strategies` //! parameter is where a caller plugs in a custom one (or a custom -//! `CostModel`, via [`crate::replacement::SketchAlgorithmStrategy::new`]) — the identical spot +//! `CostModel`, via [`crate::replacement::ASAPStrategies::new`]) — the identical spot //! [`crate::replacement::search_workload_with`] itself exposes. //! //! ## Two guarantees the old traversal made, re-verified against the new one @@ -134,7 +134,7 @@ //! ## One thing [`CandidateLogicalASAPDAGs`] doesn't carry that this module still needs: //! human-readable `location` text //! -//! [`TargetSubDAGCandidates`]/[`CandidateLogicalASAPDAGs`] deliberately track only `Rc` +//! [`TargetSubDAGCandidates`]/[`CandidateLogicalASAPDAGs`] deliberately track only `Rc` //! pointer identity — the currency the search itself needs — not //! caller-facing prose. [`ReplacementExplanation::location`] is prose (a //! breadcrumb like `root "dash_a" > lhs`), so this module keeps one small, @@ -161,8 +161,8 @@ //! //! | Catalog entry | Status | Where a future `ExplanationKind` would come from | //! |---|---|---| -//! | Semantic-equivalent rewriting (e.g. `avg` → `sum`/`count`) | [`AvgToSumOverCountStrategy`](crate::rewrite::AvgToSumOverCountStrategy) exists and is wired into `default_strategies()` (issue #253) — but still no `ExplanationKind` of its own below, since this table is about *direct* findings for a catalog entry, and this strategy's whole point is indirect: its `Replacement::Rewrite` candidate exposes `sum`/`count` as independently bindable discovered targets, which can then earn `CommonSubexpressionReuse` findings when the workload actually reuses them | A dedicated variant would need `findings_from_candidate_logical_asap_dags` to recognize a `LogicalRewrite`-provenance candidate as a finding in its own right, not just rely on what it exposes downstream | -//! | Roll-ups (fine-to-coarse group-by reuse) | [`RollupStrategy`](crate::rollup::RollupStrategy), derived from workload siblings after CSE/target discovery (issue #254) | Any `Replacement::Rewrite` candidate that rolls a coarse aggregate up from a compatible finer aggregate | +//! | Semantic-equivalent rewriting (e.g. `avg` → `sum`/`count`) | [`AvgToSumOverCountStrategy`](crate::rewrite::AvgToSumOverCountStrategy) exists and is wired into `default_strategies()` (issue #253) — but still no `ExplanationKind` of its own below, since this table is about *direct* findings for a catalog entry, and this strategy's whole point is indirect: its `Replacement::SubDAG` rewrite candidate exposes `sum`/`count` as independently bindable discovered targets, which can then earn `CommonSubexpressionReuse` findings when the workload actually reuses them | A dedicated variant would need `findings_from_candidate_logical_asap_dags` to recognize a `LogicalRewrite`-provenance candidate as a finding in its own right, not just rely on what it exposes downstream | +//! | Roll-ups (fine-to-coarse group-by reuse) | [`RollupStrategy`](crate::rollup::RollupStrategy), derived from workload siblings after CSE/target discovery (issue #254) | Any `Replacement::SubDAG` rewrite candidate that rolls a coarse aggregate up from a compatible finer aggregate | //! | Wavelets/OMP | Params type exists (`WaveletKind`/`WaveletParams`), reachable only via a deployment `CostModel::realize_extension` (no core `AggIntent` dispatch picks it) | A `ReplacementStrategy` that inspects a deployment's own `CostModel`, once some intent shape actually maps to `Realization::Wavelet` | //! | Sampling | Same story as Wavelets: `SamplingKind`/`SamplingParams` exist, unreachable from core dispatch | Same hook as Wavelets, for `Realization::Sample` | //! | Deep generative compression | No representation at all — no `Realization`/`FieldDataType` variant | Needs a new summary family added to `asap_types::post_asap` first | @@ -175,9 +175,8 @@ //! [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy //! [`ReplacementSubDAG`]: crate::replacement::ReplacementSubDAG //! [`Replacement`]: crate::replacement::Replacement -//! [`Replacement::Summary`]: crate::replacement::Replacement::Summary -//! [`Replacement::Rewrite`]: crate::replacement::Replacement::Rewrite -//! [`SketchAlgorithmStrategy`]: crate::replacement::SketchAlgorithmStrategy +//! [`Replacement::SubDAG`]: crate::replacement::Replacement::SubDAG +//! [`ASAPStrategies`]: crate::replacement::ASAPStrategies //! [`SharedSubDAGStrategy`]: crate::replacement::SharedSubDAGStrategy //! [`CandidateLogicalASAPDAGs`]: crate::replacement::CandidateLogicalASAPDAGs //! [`TargetSubDAGCandidates`]: crate::replacement::TargetSubDAGCandidates @@ -186,9 +185,9 @@ use std::collections::HashMap; use std::fmt::Display; use std::rc::Rc; -use asap_types::post_asap::{FieldDataType, SummaryExpr, SummaryNode}; -use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::cse::{structural_hash, HashCache}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; +use asap_types::post_asap::FieldDataType; use crate::replacement::{ self, CandidateLogicalASAPDAGs, Replacement, ReplacementStrategy, TargetSubDAGCandidates, @@ -206,8 +205,8 @@ use crate::replacement::{ #[non_exhaustive] pub enum ExplanationKind { /// A `TargetSubDAG`'s candidate list contains at least one - /// [`Replacement::Summary`] that realizes a sketch family — - /// [`crate::replacement::SketchAlgorithmStrategy`] found a genuine sketch + /// [`Replacement::SubDAG`] that realizes a sketch family — + /// [`crate::replacement::ASAPStrategies`] found a genuine sketch /// alternative for this `Aggregate`, beyond whatever exact/pass-through /// candidate `crate::replacement`'s own `realizations_for_intent` would /// have committed to on its own. @@ -222,8 +221,8 @@ pub enum ExplanationKind { /// [`Replacement::ExactComposition`] — /// [`crate::exact_composition::ExactCompositionStrategy`] found an exact /// operator that can be composed with a summary plan across an explicit - /// update/readout boundary instead of collapsing the whole DAG into - /// `KeepPreAsap` (issue #171). + /// update/evaluation boundary instead of keeping the whole tree as it is + /// (issue #171). ExactComposition, } @@ -233,11 +232,11 @@ pub enum ExplanationKind { /// not machine parsing — literally the matching candidate's own /// [`crate::replacement::ReplacementSubDAG::rationale`]). /// -/// `node_hash` is [`structural_hash`](asap_types::pre_asap::cse::structural_hash) +/// `node_hash` is [`structural_hash`](asap_types::ir::cse::structural_hash) /// of the `TargetSubDAG`'s own `target` sub-DAG — the same function, on the -/// same `Rc` shape, that [`asap_types::dag_export::DAGNode::hash`] +/// same `Rc` shape, that [`asap_types::dag_export::DAGNode::hash`] /// is computed with. A downstream consumer that independently exported the -/// same `QueryExpr` (e.g. via `asap_types::dag_export::export`) can match +/// same node (e.g. via `asap_types::dag_export::export`) can match /// this explanation to a `DAGNode` by first comparing hashes and then /// confirming structural equality with [`ReplacementExplanation::target`]. #[derive(Debug, Clone, PartialEq)] @@ -249,7 +248,7 @@ pub struct ReplacementExplanation { /// The exact target expression the explanation describes. Reporting /// integrations use this together with `node_hash`: the hash narrows the /// search, and structural equality makes the final match collision-safe. - pub target: Rc, + pub target: Rc, } /// Explain every replacement [`crate::replacement::search_workload`] finds @@ -265,7 +264,7 @@ pub struct ReplacementExplanation { /// candidate-plan space, then reads findings off it — see the module docs' /// "The reframing" section for what that translation actually checks. pub fn explain_replacements( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, Rc)>, ) -> Vec { explain_replacements_with(roots, &replacement::default_strategies()) } @@ -274,17 +273,17 @@ pub fn explain_replacements( /// instead of [`crate::replacement::default_strategies`] — the extension /// point for a deployment-specific [`ReplacementStrategy`], or a custom /// `CostModel` plugged into -/// [`crate::replacement::SketchAlgorithmStrategy::new`] (e.g. via +/// [`crate::replacement::ASAPStrategies::new`] (e.g. via /// [`crate::replacement::default_strategies_with`]). /// /// [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy pub fn explain_replacements_with<'s, Id: Display>( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, Rc)>, strategies: &[Box], ) -> Vec { - let ided: Vec<(String, Rc)> = roots + let ided: Vec<(String, Rc)> = roots .into_iter() - .map(|(id, expr)| (id.to_string(), Rc::new(expr))) + .map(|(id, expr)| (id.to_string(), expr)) .collect(); let space = replacement::search_workload_with(ided, strategies); findings_from_candidate_logical_asap_dags(&space) @@ -375,7 +374,7 @@ fn sketch_finding_reason(group: &TargetSubDAGCandidates) -> Option { .candidates .iter() .filter( - |c| matches!(&c.replacement, Replacement::Summary(node) if is_sketch_realization(node)), + |c| matches!(&c.replacement, Replacement::SubDAG(node) if is_sketch_realization(node)), ) .map(|c| c.rationale.as_str()) .collect(); @@ -387,7 +386,7 @@ fn sketch_finding_reason(group: &TargetSubDAGCandidates) -> Option { } /// Does `group` have two or more consumers *and* a "build once and share" -/// candidate (the [`Replacement::Rewrite`] whose `Rc` is the group's own +/// candidate (the [`Replacement::SubDAG`] whose `Rc` is the group's own /// `target`) in its candidate list? If so, the finding's `reason` is that /// candidate's own `rationale`. fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option { @@ -398,7 +397,7 @@ fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option Option bool { +fn is_sketch_realization(node: &OperatorNode) -> bool { if node .guarantee .as_ref() @@ -417,9 +416,13 @@ fn is_sketch_realization(node: &SummaryNode) -> bool { { return false; } - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => is_sketch_realization(summary_input), - SummaryExpr::SummaryAgg { family, .. } => matches!(family, FieldDataType::Sketch(..)), + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + is_sketch_realization(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => { + matches!(family, FieldDataType::Sketch(..)) + } _ => false, } } @@ -432,8 +435,10 @@ fn is_sketch_realization(node: &SummaryNode) -> bool { /// every breadcrumb path that reaches a given `Rc`, not just the first: a /// shared node referenced from two workload roots (or two branches of one /// root) needs both breadcrumbs in its finding's `location`, not just one. -fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryExpr, Vec> { - let mut locations: HashMap<*const QueryExpr, Vec> = HashMap::new(); +fn collect_locations( + roots: &[(String, Rc)], +) -> HashMap<*const OperatorNode, Vec> { + let mut locations: HashMap<*const OperatorNode, Vec> = HashMap::new(); for (id, root) in roots { visit(root, format!("root {id:?}"), &mut locations); } @@ -444,9 +449,9 @@ fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryE /// through its children. A shared ancestor is intentionally traversed once /// per incoming path so every descendant receives every valid breadcrumb. fn visit( - node: &Rc, + node: &Rc, label: String, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const OperatorNode, Vec>, ) { let ptr = Rc::as_ptr(node); locations.entry(ptr).or_default().push(label.clone()); @@ -455,20 +460,24 @@ fn visit( /// `node`'s own **relational-skeleton** operator children — the same scope /// `crate::replacement`'s own target-discovery `walk_children` (and -/// `asap_types::pre_asap::cse::share_common_sub_dags`'s `rebuild_children`) -/// use. Exhaustive over every `QueryExpr` variant: a new variant fails to -/// compile here until this match is extended too. +/// `asap_types::ir::cse::share_common_sub_dags`) use. Exhaustive over every +/// `NonASAPOp` variant: a new variant fails to compile here until this match +/// is extended too. An ASAP node never occurs in a workload root. fn visit_children( - node: &QueryExpr, + node: &OperatorNode, label: &str, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const OperatorNode, Vec>, ) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { + use asap_types::ir::{NonASAPOp::*, ScalarExpr}; + let Operator::NonASAP(op) = &node.operator else { + return; + }; + match op { + Scan { .. } | Values { .. } => {} + PromqlVectorFromScalar(ScalarExpr::PromqlScalarFromVector(c)) => { visit(c, format!("{label} > child"), locations) } + PromqlVectorFromScalar(_) => {} PromqlRelabel { child, .. } | PromqlInfoEnrich { child, .. } | PromqlSeriesSample { child, .. } @@ -484,7 +493,7 @@ fn visit_children( | Limit { child, .. } => visit(child, format!("{label} > child"), locations), Concat { children, .. } => { for (i, c) in children.iter().enumerate() { - visit_children(c, &format!("{label} > concat[{i}]"), locations); + visit(c, format!("{label} > concat[{i}]"), locations); } } Join { left, right, .. } | SetOp { left, right, .. } => { @@ -495,31 +504,20 @@ fn visit_children( visit(lhs, format!("{label} > lhs"), locations); visit(rhs, format!("{label} > rhs"), locations); } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} } } #[cfg(test)] mod tests { use super::*; + use asap_types::ir::operator_properties::{BinaryOpKind, Reduction, Source}; + use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, Predicate, ScalarExpr}; use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; - use asap_types::pre_asap::query_expr::{Reduction, Source}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; + use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { + fn metric_scan(labels: &[&str]) -> Rc { let mut columns = vec![ Field::plain("ts", DataType::Timestamp, false), Field::plain("value", DataType::Float64, false), @@ -529,22 +527,43 @@ mod tests { .iter() .map(|n| Field::plain(*n, DataType::Utf8, true)), ); - QueryExpr::Scan { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), - } + })) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], filters: vec![], having: None, - child: Rc::new(child), - } + child, + })) + .unwrap() + } + + fn binary( + kind: BinaryOpKind, + lhs: Rc, + rhs: Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs, + rhs, + })) + .unwrap() } // ── SketchApproximation ────────────────────────────────────────────── @@ -567,7 +586,7 @@ mod tests { } /// `node_hash` must be the literal `structural_hash` a downstream - /// consumer would compute over the *same* `QueryExpr` sub-DAG via + /// consumer would compute over the *same* `OperatorNode` sub-DAG via /// `asap_types::dag_export::export` — the whole point of carrying it is /// that two independent exports of the same DAG agree, with no /// string-matching against `location` required. @@ -586,7 +605,7 @@ mod tests { Some(sketch.node_hash), expected_hash, "ReplacementExplanation::node_hash must match dag_export's DAGNode::hash \ - for the same QueryExpr sub-DAG" + for the same OperatorNode sub_dag" ); } @@ -641,21 +660,18 @@ mod tests { /// A sketch-applicable `Aggregate` reachable via two paths that CSE /// collapses onto one `Rc` — the same `median(x) == median(x)` shape - /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_sub_dag` + /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_sub-DAG` /// test uses — must be reported once, not once per path: it is exactly /// one [`crate::replacement::TargetSubDAGCandidates`], keyed by `Rc` pointer identity, /// not one per path that reaches it. #[test] fn a_shared_sketchable_aggregate_is_reported_only_once() { let quantile = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let root = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( - asap_types::pre_asap::expr_ir::CompareOpKind::Eq, - ), - lhs: Rc::new(quantile.clone()), - rhs: Rc::new(quantile), - vector_match: None, - }; + let root = binary( + BinaryOpKind::Compare(asap_types::pre_asap::expr_ir::CompareOpKind::Eq), + Rc::clone(&quantile), + quantile, + ); let findings = explain_replacements(vec![("ratio", root)]); let sketch: Vec<_> = findings .iter() @@ -695,8 +711,8 @@ mod tests { #[test] fn descendant_of_a_shared_root_keeps_every_root_breadcrumb() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let outer = agg(vec![2], AggIntent::Sum { col: None }, inner); - let findings = explain_replacements(vec![("dash_a", outer.clone()), ("dash_b", outer)]); + let outer = agg(vec![0], AggIntent::Sum { col: None }, inner); + let findings = explain_replacements(vec![("dash_a", Rc::clone(&outer)), ("dash_b", outer)]); let inner_sketch = findings .iter() .find(|f| { @@ -742,14 +758,11 @@ mod tests { // The same shared branch appearing twice within one query (an `a/a` // shape) — single-query CSE. let branch = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let q = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Arithmetic( - asap_types::pre_asap::expr_ir::ArithmeticOpKind::Div, - ), - lhs: Rc::new(branch.clone()), - rhs: Rc::new(branch), - vector_match: None, - }; + let q = binary( + BinaryOpKind::Arithmetic(asap_types::pre_asap::expr_ir::ArithmeticOpKind::Div), + Rc::clone(&branch), + branch, + ); let findings = explain_replacements(vec![("ratio", q)]); let reuse: Vec<_> = findings .iter() @@ -766,7 +779,7 @@ mod tests { /// A shared node nested three levels under two *different*, unshared /// `Filter` parents (mirrors `crate::replacement::tests:: - /// nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered`) + /// nested_shared_sub-DAG_below_an_unshared_parent_is_still_discovered`) /// must still be exactly one finding — the maximal-`TargetSubDAG` /// guarantee the module docs describe, now provided by /// `crate::replacement`'s own target discovery rather than this module's @@ -774,17 +787,20 @@ mod tests { #[test] fn a_deeply_shared_sub_dag_under_different_parents_is_reported_once() { use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::new(shared.clone()), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::new(shared), - }; + let root_a = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: Rc::clone(&shared), + })) + .unwrap(); + let root_b = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: shared, + })) + .unwrap(); let findings = explain_replacements(vec![("a", root_a), ("b", root_b)]); let reuse: Vec<_> = findings .iter() @@ -825,7 +841,7 @@ mod tests { let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let custom_model = AlwaysDDSketch; let strategies: Vec> = vec![Box::new( - crate::replacement::SketchAlgorithmStrategy::new(&custom_model), + crate::replacement::ASAPStrategies::new(&custom_model), )]; let findings = explain_replacements_with(vec![("q", q)], &strategies); assert_eq!(findings.len(), 1); diff --git a/crates/asap-aware-mapping/src/grouping.rs b/crates/asap-aware-mapping/src/grouping.rs index fc4abfe62..bbbaad37f 100644 --- a/crates/asap-aware-mapping/src/grouping.rs +++ b/crates/asap-aware-mapping/src/grouping.rs @@ -7,7 +7,7 @@ //! //! ## Placement: planning metadata and edge-state type //! -//! `SummaryExpr::SummaryAgg` carries the grouping choice next to the +//! `ASAPOp::SummaryAgg` carries the grouping choice next to the //! `Reduction` whose `by` keys determine legality. The same choice is also //! committed to `FieldDataType::Sketch` on the aggregate's output edge. //! That duplication is intentional: the node field makes the choice easy to @@ -41,7 +41,7 @@ //! An earlier draft of this module (written against the very first draft of //! #251) reused a `CostModel`-wrapping adapter that "steered" a //! whole-recursive-bind decision procedure toward a specific `SketchKind`, -//! the same pattern [`crate::replacement::SketchAlgorithmStrategy`]'s own module +//! the same pattern [`crate::replacement::ASAPStrategies`]'s own module //! docs explain was deliberately deleted from this crate as an anti-pattern: //! forcing a choice via a whole-DAG `CostModel` adapter had a real bug where //! the forced choice could leak into a target's own nested aggregates. This @@ -53,7 +53,7 @@ //! passes that exact, //! already-decided `Realization` to //! [`crate::replacement::construct_summary`] — the same first-class, -//! one-candidate-at-a-time primitive [`crate::replacement::SketchAlgorithmStrategy`] +//! one-candidate-at-a-time primitive [`crate::replacement::ASAPStrategies`] //! itself calls once per candidate. No adapter, no steering, no risk of a //! forced choice leaking into nested aggregates. //! @@ -71,13 +71,14 @@ use std::rc::Rc; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ default_hydra_params, hydra_kind_for, AccuracyError, BoundExpr, CompositionOperator, FieldDataType, GroupingStrategy, GuaranteeSource, HydraKind, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, + SketchAlgorithm, SketchParams, }; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; use crate::accuracy::{ AccuracyBudgetAllocator, AccuracyEvidenceProvider, AccuracyModel, PropagationStats, @@ -110,15 +111,15 @@ pub fn has_subpopulations(reduction: &Reduction) -> bool { /// A single static instance so [`HydraGroupingStrategy::default_cost_model`] /// can hand out a `&'static dyn CostModel` without heap-allocating one — same -/// pattern [`crate::replacement::SketchAlgorithmStrategy`] uses. +/// pattern [`crate::replacement::ASAPStrategies`] uses. static DEFAULT_COST_MODEL: DefaultCostModel = DefaultCostModel; /// Wraps the `GroupingStrategy` axis (issue #256) as a -/// [`ReplacementStrategy`]: for a target [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) +/// [`ReplacementStrategy`]: for a target [`ASAPStrategies`](crate::replacement::ASAPStrategies) /// already has an opinion on, offers an additional /// `GroupingStrategy::SharedMultiSubpopulation` candidate wherever the /// legality conditions in the module docs above hold — alongside, not -/// instead of, the per-subpopulation candidates `SketchAlgorithmStrategy` +/// instead of, the per-subpopulation candidates `ASAPStrategies` /// itself enumerates. The workload search composes both strategies over the /// same target, so it sees every summary-family alternative *and* the Hydra /// alternative; the built-in workload search registers both strategies, and @@ -132,7 +133,7 @@ pub struct HydraGroupingStrategy<'a> { impl HydraGroupingStrategy<'static> { /// A strategy that ranks/binds via the built-in [`DefaultCostModel`] — /// what a deployment gets with no custom cost model plugged in, the same - /// default [`crate::replacement::SketchAlgorithmStrategy::default_cost_model`] + /// default [`crate::replacement::ASAPStrategies::default_cost_model`] /// offers. pub fn default_cost_model() -> Self { Self { @@ -144,7 +145,7 @@ impl HydraGroupingStrategy<'static> { impl<'a> HydraGroupingStrategy<'a> { /// A strategy that ranks/binds via `cost_model` instead of the built-in /// static preference order — the same customization point - /// [`crate::replacement::SketchAlgorithmStrategy::new`] already offers. + /// [`crate::replacement::ASAPStrategies::new`] already offers. pub fn new(cost_model: &'a dyn CostModel) -> Self { Self { planning_inputs: CandidatePlanningInputs::with_default_accuracy(cost_model), @@ -173,7 +174,7 @@ impl<'a> HydraGroupingStrategy<'a> { /// variant modeled. fn hydra_proposals(&self, target: &TargetSubDAG<'_>) -> Proposals { let mut proposals = Proposals::default(); - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = target.root.non_asap() else { return proposals; }; if !has_subpopulations(reduction) { @@ -208,11 +209,11 @@ impl<'a> HydraGroupingStrategy<'a> { /// `PerSubpopulationInstance` to /// `SharedMultiSubpopulation { kind: hydra_kind, .. }` — reusing the /// entire bind decision procedure (schema derivation, column resolution, - /// readout construction) unchanged, patching only the one field this + /// evaluation construction) unchanged, patching only the one field this /// axis owns. fn build_candidate( &self, - root: &Rc, + root: &Rc, intent: &AggIntent, sketch_kind: SketchAlgorithm, hydra_kind: HydraKind, @@ -233,12 +234,12 @@ impl<'a> HydraGroupingStrategy<'a> { params, }; - let (family, query) = match &node.expr { - SummaryExpr::SummaryEstimate { + let (family, query) = match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => match &summary_input.expr { - SummaryExpr::SummaryAgg { family, .. } => (family, Some(query)), + }) => match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => (family, Some(query)), _ => return None, }, _ => return None, @@ -295,7 +296,7 @@ impl<'a> HydraGroupingStrategy<'a> { } Some(ReplacementSubDAG { strategy: "HydraGroupingStrategy", - replacement: Replacement::Summary(patched), + replacement: Replacement::SubDAG(patched), provenance: crate::replacement::ReplacementProvenance::SummaryRealization, rationale: format!( "{} realizes as a shared {hydra_kind:?} structure over {sketch_kind:?} \ @@ -313,7 +314,7 @@ impl<'a> HydraGroupingStrategy<'a> { impl ReplacementStrategy for HydraGroupingStrategy<'_> { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = target.root.non_asap() else { return false; }; if !has_subpopulations(reduction) { @@ -357,15 +358,15 @@ impl ReplacementStrategy for HydraGroupingStrategy<'_> { /// destructures the right variant for `kind`; this function's only job is /// to find whatever `SketchParams` the bind decision already committed to /// and hand the whole thing over unchanged. -fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { +fn per_subpopulation_sketch_params(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { per_subpopulation_sketch_params(summary_input) } - SummaryExpr::SummaryAgg { + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } => Some(kind.params().clone()), + }) => Some(kind.params().clone()), _ => None, } } @@ -373,33 +374,35 @@ fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { /// Rebuild `node`, replacing its `SummaryAgg`'s `grouping` field with /// `grouping` — patching the one field this axis owns onto an /// already-correctly-bound node rather than re-deriving the rest of it. -/// Recurses through a `SummaryEstimate` readout wrapper (the shape every +/// Recurses through a `SummaryEstimate` evaluation wrapper (the shape every /// sketch candidate this module builds actually has) to reach the /// `SummaryAgg` underneath. fn with_grouping( - node: Rc, + node: Rc, grouping: GroupingStrategy, stats: &PropagationStats, -) -> Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { +) -> Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: with_grouping(Rc::clone(summary_input), grouping, stats), - query: query.clone(), - }, - schema: node.schema.clone(), - guarantee: node.guarantee.as_ref().map(|g| hydra_guarantee(g, stats)), - }), - SummaryExpr::SummaryAgg { + }) => std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: with_grouping(Rc::clone(summary_input), grouping, stats), + query: query.clone(), + }), + node.schema.clone(), + ) + .with_guarantee(node.guarantee.as_ref().map(|g| hydra_guarantee(g, stats))), + ), + Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } => { + }) => { let grouped_family = match family { FieldDataType::Sketch(kind, _) => { FieldDataType::Sketch(kind.clone(), grouping.clone()) @@ -412,18 +415,20 @@ fn with_grouping( field.dtype = FieldDataType::Sketch(kind.clone(), grouping.clone()); } } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(child), - family: grouped_family, - input: input.clone(), - reduction: reduction.clone(), - grouping, - filter: None, - }, - schema: grouped_schema, - guarantee: None, - }) + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(child), + family: grouped_family, + input: input.clone(), + reduction: reduction.clone(), + grouping, + filter: None, + }), + grouped_schema, + ) + .with_guarantee(None), + ) } // Never reached by this module's own callers (they only ever pass a // node `construct_summary_with` just bound for a `Sketch` @@ -492,51 +497,11 @@ fn hydra_guarantee(inner: &ResultGuarantee, stats: &PropagationStats) -> ResultG mod tests { use super::*; use crate::accuracy::{DefaultAccuracyModel, EqualSplitAllocator}; + use crate::test_support::{agg, agg_per_entity, metric_scan}; use asap_types::post_asap::ErrorMetric; use asap_types::pre_asap::agg_intent::{default_cardinality, default_quantile}; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - // ── has_subpopulations ──────────────────────────────────────────────── #[test] @@ -556,7 +521,7 @@ mod tests { #[test] fn without_grouping_has_a_subpopulation_concept_even_when_empty() { - use asap_types::pre_asap::query_expr::GroupKeys; + use asap_types::ir::operator_properties::GroupKeys; // `without([])` groups by every remaining label — a real // subpopulation concept, unlike `by([])`'s genuine full reduction. assert!(has_subpopulations(&Reduction::Reduce(GroupKeys::without( @@ -603,7 +568,7 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); assert!(HydraGroupingStrategy::default_cost_model().matches(&target)); } @@ -611,7 +576,7 @@ mod tests { #[test] fn does_not_match_an_empty_by_aggregate() { // Global reduction — no subpopulation concept, no Hydra alternative. - let q = Rc::new(agg(vec![], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -620,10 +585,7 @@ mod tests { #[test] fn does_not_match_a_per_entity_aggregate() { - let q = Rc::new(agg_per_entity( - default_quantile(0.99), - metric_scan(&["job"]), - )); + let q = agg_per_entity(default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -632,14 +594,14 @@ mod tests { #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); assert!(!HydraGroupingStrategy::default_cost_model().matches(&target)); } #[test] fn quantile_has_no_hydra_candidate_without_a_modeled_error_bound() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let replacements = HydraGroupingStrategy::default_cost_model().replacements(&target); assert!(replacements.is_empty(), "{replacements:?}"); @@ -653,13 +615,13 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let replacements = HydraGroupingStrategy::default_cost_model().replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|guarantee| guarantee.bound.evaluate().is_none() && guarantee.failure_probability.evaluate().is_none()) @@ -691,7 +653,7 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, @@ -702,7 +664,7 @@ mod tests { assert_eq!(replacements.len(), 2, "{replacements:?}"); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|g| g.bound.evaluate().is_some() && g.failure_probability.evaluate().is_some()) @@ -731,7 +693,7 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, @@ -771,7 +733,7 @@ mod tests { } } } - let q = Rc::new(agg( + let q = agg( vec![2], AggIntent::Count { accuracy: AccuracyTarget::EpsilonDelta { @@ -780,7 +742,7 @@ mod tests { }, }, metric_scan(&["job"]), - )); + ); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, @@ -801,7 +763,7 @@ mod tests { // summary_candidates(Cardinality) = [Hll, Theta, Kmv] — none have a // modeled Hydra variant, so no candidate at all (not an error, just // an empty result, same conservatism as every other strategy here). - let q = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let q = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -817,7 +779,7 @@ mod tests { q: 0.99, accuracy: AccuracyTarget::Exact, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -828,11 +790,7 @@ mod tests { fn exact_mergeable_intent_has_no_hydra_candidate() { // Sum's exact accumulator has no candidate summary families at all // (summary_candidates only covers approximate-capable intents). - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let strategy = HydraGroupingStrategy::default_cost_model(); assert!(!strategy.matches(&target)); @@ -843,14 +801,16 @@ mod tests { fn does_not_match_a_multi_intent_or_having_aggregate() { let strategy = HydraGroupingStrategy::default_cost_model(); - let multi = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }); + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&multi); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); @@ -859,7 +819,7 @@ mod tests { /// A custom `CostModel` doesn't change *which* candidate is offered — /// only which sketch candidate `realizations_for_intent` itself would /// have ranked first, and how that candidate's own params are sized — - /// same guarantee `SketchAlgorithmStrategy` makes for its own candidates. + /// same guarantee `ASAPStrategies` makes for its own candidates. struct PreferDDSketch; impl CostModel for PreferDDSketch { fn rank_candidates( @@ -878,7 +838,7 @@ mod tests { #[test] fn custom_cost_model_cannot_enable_unproven_hydra_kll() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let custom = PreferDDSketch; let replacements = HydraGroupingStrategy::new(&custom).replacements(&target); diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 95e50a552..15fe06d2b 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -1,18 +1,18 @@ //! `asap-plan` — the cost-aware optimizer layer over the pre-ASAP intent algebra. //! //! This crate sits between the language-agnostic IR ([`asap_ir`]) and -//! any runtime: it consumes pre-ASAP [`QueryExpr`](asap_types::pre_asap::QueryExpr) +//! any runtime: it consumes pre-ASAP [`OperatorNode`](asap_types::ir::OperatorNode) //! DAGs and makes the cost-aware decisions the pre-ASAP IR deliberately //! leaves open — which sketch (if any) realises each approximate intent. //! //! **Common sub-expression elimination (CSE) is not this crate's job.** -//! Detection is a primary pass over the pre-ASAP `QueryExpr` IR itself -//! (`asap_types::pre_asap`, design tracked in issue #223), run before a -//! DAG ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 +//! Detection is a primary pass over the pre-ASAP operator IR itself +//! (`asap_types::ir::cse`, design tracked in issue #223), run before a +//! tree ever reaches [`replacement::ASAPStrategies`] — see issue #222 //! for why (batch query optimization needs to see shared work across a //! `QueryWorkload` before summary binding, not after). This crate may //! eventually run a second, narrower CSE pass of its own over an -//! already-bound `SummaryExpr`/`SummaryNode` DAG, recognizing sharing that's invisible +//! already-bound post-ASAP `OperatorNode` DAG, recognizing sharing that's invisible //! at the pre-ASAP level by construction — e.g. `Quantile(x, 0.99)` and //! `Quantile(x, 0.95)` are structurally distinct `AggIntent`s but can //! still share one built sketch, read out twice. That post-ASAP pass is @@ -92,7 +92,7 @@ //! #33) is an additional `ReplacementStrategy`: the orthogonal //! `GroupingStrategy` axis (one summary instance per `by` subpopulation //! versus one shared Hydra-family structure serving all of them), offered -//! alongside the candidates [`replacement::SketchAlgorithmStrategy`] +//! alongside the candidates [`replacement::ASAPStrategies`] //! enumerates for the same target. //! - [`rewrite`] — the "semantic-equivalent rewriting (e.g. `avg` → //! `sum`/`count`) to increase how often the [sharing/sketch] optimizations @@ -117,7 +117,7 @@ //! |---|---|---| //! | Schema resolution | Derive input schemas and resolve column names to positions | `asap_types::pre_asap::SchemaResolver::resolve_schema`, `resolve_root` | //! | Realization | Enumerate ranked physical forms for one aggregate intent | `replacement::realizations_for_intent` | -//! | Replacement | Construct each candidate summary sub-DAG | [`replacement::SketchAlgorithmStrategy`] | +//! | Replacement | Construct each candidate summary sub-DAG | [`replacement::ASAPStrategies`] | //! | Search | Enumerate and compare alternatives across a workload | [`replacement::search_workload`] | //! | Runtime placement | Choose deployment locations and concrete executors | Downstream physical plan providers | //! @@ -207,12 +207,12 @@ pub use recurrence::{ UpdateRate, }; pub use replacement::{ - default_strategies, default_strategies_with, search_workload, search_workload_with, - search_workload_with_targets, summary_candidates, CandidateLogicalASAPDAGs, - CompositionDecision, GlobalSelection, Matcher, Proposals, RankedTargetSubDAGCandidates, - Realization, RealizationError, RecurrenceProfileMap, RejectedCandidate, Replacement, - ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubDAGStrategy, - SketchAlgorithmStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, + default_strategies, default_strategies_with, is_logical_rewrite, search_workload, + search_workload_with, search_workload_with_targets, summary_candidates, ASAPStrategies, + CandidateLogicalASAPDAGs, CompositionDecision, GlobalSelection, Matcher, Proposals, + RankedTargetSubDAGCandidates, Realization, RealizationError, RecurrenceProfileMap, + RejectedCandidate, Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, + SharedSubDAGStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, MAX_SEARCH_ITERATIONS, }; pub use rewrite::{AvgToSumOverCountStrategy, SemanticEquivalentRewriteStrategy}; @@ -222,14 +222,15 @@ pub use summary_maintenance_dag_export::{ }; pub use summary_maintenance_lifecycle::{ assemble_selected_dag_with_summary_maintenance_lifecycles, - enumerate_summary_maintenance_lifecycles, global_selection_with_summary_maintenance_lifecycles, - plan_summary_maintenance_lifecycles, SummaryMaintenanceCapabilities, - SummaryMaintenanceDeployment, SummaryMaintenanceLifecycleAlternative, - SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCandidates, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleChoiceError, - SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecyclePlan, - SummaryMaintenanceLifecyclePlanError, SummaryMaintenanceLifecycleRejection, - SummaryMaintenanceLifecycleSelectionError, SummaryMaintenanceTimingError, WorkloadDemand, + enumerate_summary_maintenance_lifecycles, execution_timed_workload_dag, + global_selection_with_summary_maintenance_lifecycles, plan_summary_maintenance_lifecycles, + SummaryMaintenanceCapabilities, SummaryMaintenanceDeployment, + SummaryMaintenanceLifecycleAlternative, SummaryMaintenanceLifecycleAssemblyError, + SummaryMaintenanceLifecycleCandidates, SummaryMaintenanceLifecycleCapabilities, + SummaryMaintenanceLifecycleChoiceError, SummaryMaintenanceLifecycleCostInputs, + SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecyclePlanError, + SummaryMaintenanceLifecycleRejection, SummaryMaintenanceLifecycleSelectionError, + SummaryMaintenanceTimingError, WorkloadDemand, }; pub use topk_reuse::TopKLimitReuseStrategy; diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index b749a6967..cb1236fd7 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -2,22 +2,14 @@ use crate::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; -use asap_types::post_asap::{ - maintained_population::*, ExecutionTiming, ResultGuarantee, SummaryExpr, SummaryNode, - ValueOperation, -}; -use asap_types::pre_asap::{ - any_measure_filtered, AggIntent, CompareOpKind, DataType, QueryExpr, Reduction, ScalarValue, - Schema, Source, -}; +use asap_types::ir::non_asap::any_measure_filtered; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; +use asap_types::post_asap::{maintained_population::*, ResultGuarantee}; +use asap_types::pre_asap::{AggIntent, CompareOpKind, DataType, Reduction, ScalarValue, Source}; use std::rc::Rc; -fn plain(schema: Schema) -> Schema { - Schema::lifted(schema.fields, schema.time_index) -} - -fn strip_projection(mut root: &QueryExpr) -> &QueryExpr { - while let QueryExpr::Project { child, .. } = root { +fn strip_projection(mut root: &OperatorNode) -> &OperatorNode { + while let Some(NonASAPOp::Project { child, .. }) = root.non_asap() { root = child; } root @@ -26,17 +18,20 @@ fn strip_projection(mut root: &QueryExpr) -> &QueryExpr { /// Skips a projection that keeps every column in place, such as the one /// `SELECT *` lowers to: it changes neither the rows nor the column positions /// a Sort key refers to. -fn strip_identity_projection(expr: &Rc) -> &Rc { - if let QueryExpr::Project { +fn strip_identity_projection(expr: &Rc) -> &Rc { + if let Some(NonASAPOp::Project { cols, qualifier: None, child, - } = expr.as_ref() + }) = expr.non_asap() { - let width = child.output_schema().map(|schema| schema.fields.len()); + let width = child + .operator + .output_schema() + .map(|schema| schema.fields.len()); if width.ok() == Some(cols.len()) && cols.iter().enumerate().all(|(i, item)| { - item.alias.is_none() && matches!(item.expr, QueryExpr::Column(c) if c == i) + item.alias.is_none() && matches!(item.expr, ScalarExpr::Column(c) if c == i) }) { return child; @@ -46,11 +41,11 @@ fn strip_identity_projection(expr: &Rc) -> &Rc { } fn recognize( - root: &QueryExpr, -) -> Option<(MaintainedPopulation, PopulationStatistic, Rc)> { + root: &OperatorNode, +) -> Option<(MaintainedPopulation, PopulationStatistic, Rc)> { let root = strip_projection(root); - let (source, grouping, readout, value_column) = match root { - QueryExpr::Aggregate { + let (source, grouping, evaluation, value_column) = match root.non_asap()? { + NonASAPOp::Aggregate { child, reduction: Reduction::Reduce(grouping), measures, @@ -64,7 +59,7 @@ fn recognize( if any_measure_filtered(filters) { return None; } - let (col, readout) = match intent { + let (col, evaluation) = match intent { AggIntent::Quantile { q, col, .. } if q.is_finite() => { (*col, PopulationStatistic::Quantile { q: *q }) } @@ -74,29 +69,30 @@ fn recognize( AggIntent::Avg { col } => (*col, PopulationStatistic::Average), _ => return None, }; - let schema = child.output_schema().ok()?; + let schema = &child.schema; if col.is_some_and(|c| schema.fields.get(c).is_none()) { return None; } - (child, grouping, readout, col) + (child, grouping, evaluation, col) } - QueryExpr::Limit { - n, + NonASAPOp::Limit { + n: Some(n), offset: 0, child, + .. } => { - let QueryExpr::Sort { + let Some(NonASAPOp::Sort { child, keys, partition_by, - } = child.as_ref() + }) = child.non_asap() else { return None; }; let [key] = keys.as_slice() else { return None; }; - let QueryExpr::Column(col) = &key.expr else { + let ScalarExpr::Column(col) = &key.expr else { return None; }; if key.ascending { @@ -111,11 +107,11 @@ fn recognize( } _ => return None, }; - if let QueryExpr::Scan { + if let Some(NonASAPOp::Scan { source: Source::Table { .. }, schema, .. - } = source.as_ref() + }) = source.non_asap() { let value_column = value_column.or_else(|| { schema @@ -132,29 +128,29 @@ fn recognize( max_k: 0, quantiles: false, }; - if !schema.closed || !population.matches_input(source) { + if !schema.closed || !population.matches_node(source) { return None; } - return Some((population, readout, Rc::clone(source))); + return Some((population, evaluation, Rc::clone(source))); } // A bare PromQL selector carries the declared ingestion interval as a // temporal input scope. Membership must expire at that horizon; retain // the wrapper as the maintained input so validation can check agreement. - let (series_source, lookback_ms) = match source.as_ref() { - QueryExpr::TimeRange { range, child } => { + let (series_source, lookback_ms) = match source.non_asap() { + Some(NonASAPOp::TimeRange { range, child, .. }) => { let ms = u64::try_from(range.as_millis()).ok()?; if ms == 0 || std::time::Duration::from_millis(ms) != *range { return None; } (child.as_ref(), ms) } - other => (other, 300_000), + _ => (source.as_ref(), 300_000), }; - let QueryExpr::Scan { + let Some(NonASAPOp::Scan { source: Source::TimeSeries { metric }, predicates, schema, - } = series_source + }) = series_source.non_asap() else { return None; }; @@ -174,10 +170,13 @@ fn recognize( }; let mut matchers = Vec::new(); for predicate in predicates { - let QueryExpr::Compare { left, op, right } = predicate.0.as_ref() else { + let ScalarExpr::Compare { + left, op, right, .. + } = &predicate.0 + else { return None; }; - let (QueryExpr::Column(col), QueryExpr::Literal(ScalarValue::Utf8(value))) = + let (ScalarExpr::Column(col), ScalarExpr::Literal(ScalarValue::Utf8(value))) = (left.as_ref(), right.as_ref()) else { return None; @@ -216,46 +215,46 @@ fn recognize( max_k: 0, quantiles: false, }, - readout, + evaluation, Rc::clone(source), )) } -/// Workload-aware rule: compatible readouts share one retractable population. +/// Workload-aware rule: compatible evaluations share one retractable population. /// Deployments opt in by registering this strategy when they can maintain complete -/// population updates and price the maintenance/readout boundary. -/// The population is exact; max_k bounds the shared readout cache, not its members. +/// population updates and price the maintenance/evaluation boundary. +/// The population is exact; max_k bounds the shared evaluation cache, not its members. pub struct MaintainedPopulationStrategy { - roots: Vec>, + roots: Vec>, } impl MaintainedPopulationStrategy { - pub fn new(roots: &[Rc]) -> Self { + pub fn new(roots: &[Rc]) -> Self { Self { roots: roots.to_vec(), } } - pub fn candidate(&self, root: &Rc) -> Option> { - if let QueryExpr::Project { + pub fn candidate(&self, root: &Rc) -> Option> { + if let Some(NonASAPOp::Project { cols, qualifier, child, - } = root.as_ref() + }) = root.non_asap() { let child = self.candidate(child)?; - return Some(Rc::new(SummaryNode { - guarantee: child.guarantee.clone(), - schema: plain(root.output_schema().ok()?), - expr: SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { + let guarantee = child.guarantee.clone(); + return Some(Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Project { cols: cols.clone(), qualifier: qualifier.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - })); + child, + }), + root.schema.clone(), + ) + .with_guarantee(guarantee), + )); } - let (mut population, readout, source) = recognize(root)?; + let (mut population, evaluation, source) = recognize(root)?; let identity = population.clone(); for other in self.roots.iter().chain(std::iter::once(root)) { if let Some((p, r, _)) = recognize(other) { @@ -272,36 +271,39 @@ impl MaintainedPopulationStrategy { } } } - let input_schema = plain(source.output_schema().ok()?); - let scan = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(source), - schema: input_schema.clone(), - guarantee: Some(ResultGuarantee::exact("source samples")), - }); - // Query time is only the initial layout: whether the population is - // retained at ingestion or rebuilt per query is its lifecycle choice - // (`SummaryMaintenanceLifecyclePlan::execution_timed_dag`). The readout - // and projection above it are query-time by construction. - let maintained = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: scan, - operation: ValueOperation::MaintainPopulation { population }, - timing: ExecutionTiming::QueryTime, - }, - schema: input_schema, - guarantee: Some(ResultGuarantee::exact( + let input_schema = source.schema.clone(); + // The source node itself is the maintained input (a non-ASAP node + // keeps its derived schema), kept with its exact guarantee. + let scan = Rc::new( + source + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("source samples"))), + ); + let maintained = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::MaintainPopulation { + child: scan, + population, + }), + input_schema, + ) + .with_guarantee(Some(ResultGuarantee::exact( "exact members under the declared population semantics", - )), - }); - Some(Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: maintained, - operation: ValueOperation::ReadPopulation { readout }, - timing: ExecutionTiming::QueryTime, - }, - schema: plain(root.output_schema().ok()?), - guarantee: Some(ResultGuarantee::exact("exact current-population readout")), - })) + ))), + ); + Some(std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::EvaluatePopulation { + child: maintained, + evaluation, + }), + root.schema.clone(), + ) + .with_guarantee(Some(ResultGuarantee::exact( + "exact current-population evaluation", + ))), + )) } } impl ReplacementStrategy for MaintainedPopulationStrategy { @@ -312,10 +314,10 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { self.candidate(target.root) .map(|node| ReplacementSubDAG { strategy: "MaintainedPopulationStrategy", - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), provenance: ReplacementProvenance::SummaryRealization, rationale: - "share an exact maintained population across compatible aggregate readouts" + "share an exact maintained population across compatible aggregate evaluations" .into(), }) .into_iter() @@ -327,10 +329,26 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { mod tests { use super::*; use crate::test_support::lower_promql; - use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_sub_dags}; + use asap_types::ir::cse::share_common_sub_dags; + use asap_types::ir::physical_export::compile_physical_asap_dag as export_timed; + use asap_types::ir::timing::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; - fn lower(q: &str) -> Rc { - Rc::new(lower_promql(q, asap_types::types::AccuracyTarget::Exact)) + /// Time `root` under the default lifecycle assignment (which runs the + /// data-state / population-contract validation) and export it. + fn compile_physical_asap_dag(root: &Rc) -> Result<(), String> { + root.validate_structure().map_err(|e| e.to_string())?; + let timed = apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .map_err(|e| format!("{e:?}"))?; + export_timed(&timed).map_err(|e| format!("{e:?}"))?; + Ok(()) + } + + fn lower(q: &str) -> Rc { + lower_promql(q, asap_types::types::AccuracyTarget::Exact) } // Instant scalar aggregations share the same retractable series population. @@ -351,11 +369,11 @@ mod tests { let candidate = rule .candidate(&root) .expect("current-series rule candidate"); - compile_post_asap_dag(&candidate).expect("typed post-ASAP DAG"); + compile_physical_asap_dag(&candidate).expect("typed post-ASAP DAG"); } } - // Different readout parameters retain one shared maintenance producer in the DAG. + // Different evaluation parameters retain one shared maintenance producer in the DAG. #[test] fn quantiles_and_topk_share_a_planner_population() { let roots: Vec<_> = [ @@ -379,7 +397,7 @@ mod tests { .target_subdag_candidates() .flat_map(|g| &g.candidates) .any(|c| c.strategy == "MaintainedPopulationStrategy")); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -388,19 +406,11 @@ mod tests { ); let mut producers = Vec::new(); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::ReadPopulation { .. }, - .. - } = &plan.expr - else { - panic!("missing typed readout") + compile_physical_asap_dag(plan).unwrap(); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &plan.operator else { + panic!("missing typed evaluation") }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!("missing maintained population") }; @@ -426,14 +436,10 @@ mod tests { let (p, _, _) = recognize(&roots[0]).unwrap(); assert!(matches!(p.input, PopulationInput::CurrentSeries(ref s) if s.grouping.is_empty())); let candidate = strategy.candidate(&roots[0]).unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &candidate.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &candidate.operator else { unreachable!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { unreachable!() }; assert_eq!(population.max_k, 5); @@ -460,58 +466,50 @@ mod tests { assert_eq!(p.matchers[0].operation, CurrentSeriesMatch::Regex); } // Population timing is a lifecycle choice: a retained or rebuilt - // population both validate, while its readout must stay at query time. + // population both validate, while its evaluation must stay at query time. #[test] fn population_timing_is_not_structural() { let root = lower("topk(5,a)"); let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); - let with_timings = |population: ExecutionTiming, readout: ExecutionTiming| { + use asap_types::post_asap::ExecutionTiming; + let with_timings = |population: ExecutionTiming, evaluation: ExecutionTiming| { let mut node = (*candidate).clone(); - let SummaryExpr::ValueOperation { child, timing, .. } = &mut node.expr else { - unreachable!() - }; - *timing = readout; - let SummaryExpr::ValueOperation { timing, .. } = &mut Rc::make_mut(child).expr else { + node.timing = Some(evaluation); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &mut node.operator + else { unreachable!() }; - *timing = population; - compile_post_asap_dag(&Rc::new(node)) + Rc::make_mut(child).timing = Some(population); + compile_physical_asap_dag(&Rc::new(node)) }; use ExecutionTiming::{IngestionTime, QueryTime}; assert!(with_timings(IngestionTime, QueryTime).is_ok()); assert!(with_timings(QueryTime, QueryTime).is_ok()); assert!(with_timings(IngestionTime, IngestionTime).is_err()); } - // A readout cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. + // A evaluation cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. #[test] fn malformed_population_dags_fail_closed() { let root = lower("topk(5,a)"); let strategy = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)); let candidate = strategy.candidate(&root).unwrap(); + compile_physical_asap_dag(&candidate).expect("the unmodified candidate is legal"); let mut bad = (*candidate).clone(); - let SummaryExpr::ValueOperation { operation, .. } = &mut bad.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { evaluation, .. }) = &mut bad.operator + else { unreachable!() }; - *operation = ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { k: 6 }, - }; - assert!(compile_post_asap_dag(&Rc::new(bad.clone())).is_err()); - let SummaryExpr::ValueOperation { - child, operation, .. - } = &mut bad.expr + *evaluation = PopulationStatistic::TopK { k: 6 }; + assert!(compile_physical_asap_dag(&Rc::new(bad.clone())).is_err()); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, evaluation }) = &mut bad.operator else { unreachable!() }; - *operation = ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { k: 5 }, - }; + *evaluation = PopulationStatistic::TopK { k: 5 }; let producer = Rc::make_mut(child); - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &mut producer.expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &mut producer.operator else { unreachable!() }; @@ -519,6 +517,6 @@ mod tests { unreachable!() }; spec.metric = "b".into(); - assert!(compile_post_asap_dag(&Rc::new(bad)).is_err()); + assert!(compile_physical_asap_dag(&Rc::new(bad)).is_err()); } } diff --git a/crates/asap-aware-mapping/src/pane_sharing.rs b/crates/asap-aware-mapping/src/pane_sharing.rs index b5f5bff3a..0d944b2da 100644 --- a/crates/asap-aware-mapping/src/pane_sharing.rs +++ b/crates/asap-aware-mapping/src/pane_sharing.rs @@ -1,6 +1,6 @@ //! Costed reuse of compatible physical pane producers. The executor supplies //! an equality key covering source, state, phase and evidence. This pass never -//! changes logical readout windows or assumes compatibility from metric names. +//! changes logical evaluation windows or assumes compatibility from metric names. /// A concrete mergeable-pane implementation and its horizon costs. #[derive(Debug, Clone)] @@ -9,9 +9,9 @@ pub struct PaneReuseCandidate { pub lookback_ms: u64, /// Build, update, residency and retirement for this producer. Candidates /// with the same key must use the same unit costs and pane width, making - /// the longest-lived producer sufficient for every readout in the group. + /// the longest-lived producer sufficient for every evaluation in the group. pub producer_cost: f64, - /// Readout cost for all consumers of this distinct producer. + /// Evaluation cost for all consumers of this distinct producer. pub read_cost: f64, } @@ -84,7 +84,7 @@ mod tests { read_cost: 2.0, } } - // Share source work once while retaining both readout charges and longest history. + // Share source work once while retaining both evaluation charges and longest history. #[test] fn shares_compatible_windows() { assert_eq!( diff --git a/crates/asap-aware-mapping/src/pass/major.rs b/crates/asap-aware-mapping/src/pass/major.rs index 3671ece58..7be53e0f0 100644 --- a/crates/asap-aware-mapping/src/pass/major.rs +++ b/crates/asap-aware-mapping/src/pass/major.rs @@ -6,10 +6,10 @@ //! it *one* pass rather than *the* algorithm. `ReplacementStrategy` is //! therefore a concept of this pass, not of the optimization interface. +use asap_types::ir::cse::share_common_sub_dags; use std::rc::Rc; -use asap_types::post_asap::{share_common_summary_sub_dags, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use super::{OptimizationInput, OptimizationPass, OptimizeError, PlanOutput, QueryLifecyclePlan}; @@ -39,10 +39,10 @@ impl OptimizationPass for MajorPass { // search result carries the workload binding the lifecycle stage and // the output both need. CSE may make two identical queries share one // `Rc`, but it never drops or reorders a root, so this stays aligned. - let roots: Vec<(usize, Rc, Option)> = workload + let roots: Vec<(usize, Rc, Option)> = workload .entries() - .enumerate() - .map(|(index, (entry, expr))| { + .zip(workload.operator_indices().iter().copied()) + .map(|((entry, expr), index)| { ( index, Rc::clone(expr), @@ -57,7 +57,7 @@ impl OptimizationPass for MajorPass { // One index per root, in `CandidateLogicalASAPDAGs::roots` order — which is the order // the roots went in, which is `entries()` order. - let entry_indices: Vec = (0..workload.len()).collect(); + let entry_indices = workload.operator_indices().to_vec(); let demand = WorkloadDemand { workload: workload.query_workload(), data_workload: workload.data_workload(), @@ -94,8 +94,7 @@ impl OptimizationPass for MajorPass { .ok_or_else(|| self.missing_group(*entry_index))?; assembled.push(dag); } - let interned = - share_common_summary_sub_dags(assembled.iter().cloned().enumerate().collect()); + let interned = share_common_sub_dags(assembled.iter().cloned().enumerate().collect()); let states: Vec<_> = interned .iter() .map(|(_, dag)| summary_states(dag)) @@ -105,7 +104,7 @@ impl OptimizationPass for MajorPass { // their entries, in every plan that reaches it, so each plan picks // the same lifecycle for it. When that union cannot be costed the // roots keep their own, unshared DAG and entries. - let mut shared_entries: Vec<(Rc, Option>)> = Vec::new(); + let mut shared_entries: Vec<(Rc, Option>)> = Vec::new(); for (position, (entry_index, root)) in space.roots.iter().enumerate() { for state in &states[position] { if shared_entries.iter().any(|(s, _)| Rc::ptr_eq(s, state)) { @@ -183,7 +182,9 @@ impl OptimizationPass for MajorPass { plan, }); } - Ok(PlanOutput::new(plans)) + let mut output = PlanOutput::new(plans); + output.scalar_roots = workload.scalar_roots().to_vec(); + Ok(output) } } diff --git a/crates/asap-aware-mapping/src/pass/mod.rs b/crates/asap-aware-mapping/src/pass/mod.rs index a0b993d97..df14ab1d7 100644 --- a/crates/asap-aware-mapping/src/pass/mod.rs +++ b/crates/asap-aware-mapping/src/pass/mod.rs @@ -17,8 +17,8 @@ mod major; use std::collections::BTreeMap; use std::rc::Rc; +use asap_types::ir::OperatorNode; use asap_types::parsed_workload::ParsedWorkload; -use asap_types::post_asap::SummaryNode; use asap_types::workload::WorkloadError; use crate::accuracy::{ @@ -173,7 +173,7 @@ pub struct QueryLifecyclePlan { pub plan: SummaryMaintenanceLifecyclePlan, } -/// One plan per workload entry, in `QueryWorkload::entries()` order; +/// One multi-root workload DAG with query/lifecycle bindings in entry order; /// [`check_contract`] enforces that. /// /// Plans are not deduplicated across entries: a summary state that several @@ -184,25 +184,96 @@ pub struct QueryLifecyclePlan { #[non_exhaustive] pub struct PlanOutput { pub plans: Vec, + /// Exact scalar expressions, keyed by workload entry; embedded plan reads remain visible. + pub scalar_roots: Vec<(usize, asap_types::ir::ScalarExpr)>, } impl PlanOutput { pub fn new(plans: Vec) -> Self { - Self { plans } + Self { + plans, + scalar_roots: Vec::new(), + } } /// Entry indices in output order. pub fn entry_indices(&self) -> Vec { - self.plans.iter().map(|p| p.entry_index).collect() + let mut indices: Vec<_> = self + .plans + .iter() + .map(|p| p.entry_index) + .chain(self.scalar_roots.iter().map(|(i, _)| *i)) + .collect(); + if !self.scalar_roots.is_empty() { + indices.sort_unstable(); + } + indices } - /// The selected DAG root per query. - pub fn dags(&self) -> Vec> { + /// All query roots in workload order, including standalone scalars. + pub fn roots(&self) -> Vec { + let mut roots: Vec<_> = self + .plans + .iter() + .map(|p| { + ( + p.entry_index, + asap_types::ir::QueryRoot::Operator(Rc::clone(&p.plan.root)), + ) + }) + .chain( + self.scalar_roots + .iter() + .map(|(i, expr)| (*i, asap_types::ir::QueryRoot::Scalar(expr.clone()))), + ) + .collect(); + roots.sort_by_key(|(i, _)| *i); + roots.into_iter().map(|(_, root)| root).collect() + } + + /// The selected operator roots. Use `roots()` to include scalar queries. + pub fn operator_roots(&self) -> Vec> { self.plans.iter().map(|p| Rc::clone(&p.plan.root)).collect() } + /// Unique operators in the entire workload DAG, including scalar-plan dependencies. + /// Several query roots can reach the same operator; it is returned once. + pub fn operators(&self) -> Vec> { + let mut seen = std::collections::HashSet::new(); + let mut nodes = Vec::new(); + for root in self.roots() { + let inputs = match root { + asap_types::ir::QueryRoot::Operator(node) => vec![node], + asap_types::ir::QueryRoot::Scalar(expr) => { + expr.operator_refs().into_iter().cloned().collect() + } + }; + for input in inputs { + for node in OperatorNode::reachable(&input) { + if seen.insert(Rc::as_ptr(&node)) { + nodes.push(node); + } + } + } + } + nodes + } + + /// The workload as one physical ASAP DAG: a root per operator query, in + /// plan order, with shared sub-DAGs exported once. Standalone scalar + /// roots have no physical form yet and are left out. + pub fn execution_timed_dag( + &self, + ) -> Result< + asap_types::ir::physical_export::PhysicalASAPDAG, + crate::SummaryMaintenanceTimingError, + > { + let plans: Vec<_> = self.plans.iter().map(|p| &p.plan).collect(); + crate::execution_timed_workload_dag(&plans) + } + pub fn len(&self) -> usize { - self.plans.len() + self.plans.len() + self.scalar_roots.len() } pub fn is_empty(&self) -> bool { diff --git a/crates/asap-aware-mapping/src/physical_operator_statistics.rs b/crates/asap-aware-mapping/src/physical_operator_statistics.rs index 7be1d292c..1d2b9ebb0 100644 --- a/crates/asap-aware-mapping/src/physical_operator_statistics.rs +++ b/crates/asap-aware-mapping/src/physical_operator_statistics.rs @@ -6,7 +6,9 @@ use std::collections::HashMap; -use asap_types::pre_asap::query_expr::{InfoMatcher, Predicate, Source}; +use asap_types::ir::operator_properties::{InfoMatcher, Source}; +use asap_types::ir::Predicate; + use asap_types::workload::{ DataArrival, DataWorkload, DurationMs, QueryRecurrence, QueryWorkloadEntry, RepeatedDemand, TimeSelection, TimestampMs, @@ -278,8 +280,8 @@ pub struct PartitionStatistics { /// is the authoritative operator vocabulary: every one of its variants has a /// matching statistics variant here. /// -/// This enum intentionally does not mirror either logical IR. `QueryExpr` and -/// `SummaryExpr` are inputs to physical lowering, and one logical node may +/// This enum intentionally does not mirror the logical IR. `OperatorNode`s +/// are inputs to physical lowering, and one logical node may /// expand into several physical nodes or choose among several algorithms. /// Physical configuration such as a Top-K limit or hash-join build side lives /// on `PhysicalOperator`; this enum contains only workload/catalog evidence diff --git a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs index 509e6a13a..d846f907b 100644 --- a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs +++ b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs @@ -2,8 +2,9 @@ use std::{cell::RefCell, rc::Rc}; -use asap_types::post_asap::{SketchAlgorithm, SummaryExpr, SummaryNode}; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::ir::OperatorNode; +use asap_types::post_asap::SketchAlgorithm; +use asap_types::pre_asap::AggIntent; use asap_types::resources::CacheProfile; use crate::analytical_cost::{ @@ -40,7 +41,7 @@ pub struct PhysicalEvidenceSnapshot { /// operator. Post-ASAP summary operators need a physical plan provider because their /// implementation, placement, and retained-state layout are deployment /// choices; that provider must return the complete summary DAG, including any -/// embedded `KeepPreAsap` work. +/// non-ASAP work kept inside it. pub trait PlannerPhysicalPlanProvider { /// Atomically captures the comparison scope and evidence generation. fn capture_evidence_snapshot( @@ -57,7 +58,7 @@ pub trait PlannerPhysicalPlanProvider { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result; } @@ -91,7 +92,7 @@ pub struct PhysicalPlanCostModel<'a> { } struct CachedTargetEvidence { - root: Rc, + root: Rc, consumer_count: usize, snapshot: PhysicalEvidenceSnapshot, raw: PhysicalDAG, @@ -188,15 +189,14 @@ impl<'a> PhysicalPlanCostModel<'a> { Replacement::ExactComposition(_) => { return Err(AnalyticalCostError::UnsupportedCandidate) } - Replacement::Rewrite(query) => lower_query_physical_dag(query, scope, &evidence)?, - Replacement::Summary(summary) => match &summary.expr { - SummaryExpr::KeepPreAsap(query) => { - lower_query_physical_dag(query, scope, &evidence)? - } - _ => self - .provider - .summary_physical_dag(&snapshot, summary, target)?, - }, + // A sub-DAG without summary state is the planner's own query + // lowering; anything with summary state is deployment-provided. + Replacement::SubDAG(sub_dag) if !sub_dag.contains_asap() => { + lower_query_physical_dag(sub_dag, scope, &evidence)? + } + Replacement::SubDAG(summary) => self + .provider + .summary_physical_dag(&snapshot, summary, target)?, }; let resources = estimate_physical_dag_comparison( PhysicalDAGEstimateRequest { @@ -359,7 +359,8 @@ mod tests { use std::cell::Cell; use std::collections::HashMap; - use asap_types::pre_asap::{DataType, Field, QueryExpr, Reduction, Schema, Source}; + use asap_types::ir::{NonASAPOp, OperatorNode}; + use asap_types::pre_asap::{DataType, Field, Reduction, Schema, Source}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, @@ -405,8 +406,16 @@ mod tests { } } - fn query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn query() -> Rc { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "events".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + })) + .unwrap(); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), @@ -414,14 +423,9 @@ mod tests { output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "events".into(), - }, - predicates: vec![], - schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }), - }) + child: scan, + })) + .unwrap() } fn scope() -> ComparisonScope { @@ -583,7 +587,7 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, ) -> Result { assert_eq!(snapshot.version, "test-snapshot-1"); @@ -687,7 +691,7 @@ mod tests { version: "unused-base-v1".into(), }; let candidates = - crate::replacement::SketchAlgorithmStrategy::default_cost_model().replacements(&target); + crate::replacement::ASAPStrategies::default_cost_model().replacements(&target); provider.storage_io = Some(profile.clone()); let model = PhysicalPlanCostModel::new(&provider, base.clone()).unwrap(); let estimate = model.estimate_candidate(&candidates[0], &target).unwrap(); @@ -718,7 +722,7 @@ mod tests { let root = query(); let target = TargetSubDAG::new(&root); let candidates = - crate::replacement::SketchAlgorithmStrategy::default_cost_model().replacements(&target); + crate::replacement::ASAPStrategies::default_cost_model().replacements(&target); let provider = TestProvider::new(true, 800); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); let estimate = model.estimate_candidate(&candidates[0], &target).unwrap(); @@ -913,7 +917,7 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result { self.0.summary_physical_dag(snapshot, summary, target) @@ -1001,7 +1005,7 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result { let mut dag = self.0.summary_physical_dag(snapshot, summary, target)?; @@ -1015,7 +1019,7 @@ mod tests { } let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = crate::replacement::ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)); let provider = WrongScope(TestProvider::new(true, 800)); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); @@ -1053,7 +1057,7 @@ mod tests { fn summary_physical_dag( &self, _snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, ) -> Result { panic!("blank snapshot versions must fail before summary binding") @@ -1061,7 +1065,7 @@ mod tests { } let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = crate::replacement::ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)); let model = PhysicalPlanCostModel::new(&BlankVersionProvider, calibration()).unwrap(); assert_eq!( @@ -1088,7 +1092,7 @@ mod tests { #[test] fn sibling_candidates_share_one_scope_and_raw_baseline() { let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = crate::replacement::ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)); assert!(candidates.len() >= 2); let provider = TestProvider::new(true, 800); diff --git a/crates/asap-aware-mapping/src/query_physical_lowering.rs b/crates/asap-aware-mapping/src/query_physical_lowering.rs index 65b7fab93..055d19c79 100644 --- a/crates/asap-aware-mapping/src/query_physical_lowering.rs +++ b/crates/asap-aware-mapping/src/query_physical_lowering.rs @@ -1,7 +1,9 @@ -//! Recursive lowering from the canonical query IR to evidenced physical DAGs. +//! Recursive lowering from the operator IR to evidenced physical DAGs. use std::rc::Rc; +use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; + use crate::analytical_cost::{ validate_operator_semantics, AnalyticalCostError, EvidenceBackedPhysicalDAG, ExecutionMultiplicity, HashJoinBuildSide, PhysicalDAGNode, PhysicalNodeEvidence, @@ -13,7 +15,7 @@ use crate::physical_operator_statistics::{ }; pub struct PhysicalNodeRequest<'a> { - pub logical_node: &'a asap_types::pre_asap::QueryExpr, + pub logical_node: &'a OperatorNode, pub operator: PhysicalOperator, pub occurrence: usize, pub synthetic: bool, @@ -40,19 +42,19 @@ where } } -/// Lower a resolved query operator DAG to the physical operators understood by +/// Lower a non-ASAP operator DAG to the physical operators understood by /// this cost model. The authoritative provider supplies statistics by the /// stable physical IDs owned by that provider; missing evidence makes the /// complete query unavailable. Scalar expressions remain part of their -/// containing operator's local cost. +/// containing operator's local cost. An ASAP node is unsupported here. pub fn lower_query_physical_dag( - root: &Rc, + root: &Rc, scope: &ComparisonScope, evidence: &dyn PhysicalNodeEvidenceProvider, ) -> Result { use std::collections::HashMap; - use asap_types::pre_asap::{GroupKeys, QueryExpr, RelationalSetOpKind}; + use asap_types::pre_asap::{GroupKeys, RelationalSetOpKind}; scope.validate()?; @@ -65,7 +67,7 @@ pub fn lower_query_physical_dag( } impl Lowerer<'_> { - fn lower(&mut self, query: &QueryExpr) -> Result { + fn lower(&mut self, query: &OperatorNode) -> Result { let occurrence = self.next_id; self.next_id += 1; self.lower_new(query, occurrence) @@ -73,7 +75,7 @@ pub fn lower_query_physical_dag( fn resolve( &self, - query: &QueryExpr, + query: &OperatorNode, operator: PhysicalOperator, occurrence: usize, synthetic: bool, @@ -128,12 +130,23 @@ pub fn lower_query_physical_dag( fn lower_unary( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &OperatorNode, ) -> Result { let child_id = self.lower(child)?; + self.push_unary(query, occurrence, operator, child_id) + } + + /// `operator` over an already-lowered child. + fn push_unary( + &mut self, + query: &OperatorNode, + occurrence: usize, + operator: PhysicalOperator, + child_id: String, + ) -> Result { let children = vec![child_id.clone()]; let evidence = self.resolve(query, operator, occurrence, false, &children, None)?; let statistics = &evidence.statistics; @@ -151,17 +164,17 @@ pub fn lower_query_physical_dag( fn lower_promql_unary( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &OperatorNode, ) -> Result { self.lower_unary(query, occurrence, operator, child) } fn lower_promql_scalar_leaf( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, ) -> Result { let operator = PhysicalOperator::PromqlScalarLeaf; @@ -171,6 +184,28 @@ pub fn lower_query_physical_dag( self.push(evidence, operator, vec![], None) } + /// Lower the owned scalar operand of `vector(s)`. A literal or + /// `time()` is a physical scalar leaf; `scalar(v)` reads its vector + /// through `PromqlVectorToScalar`. + fn lower_scalar_operand( + &mut self, + query: &OperatorNode, + occurrence: usize, + expr: &ScalarExpr, + ) -> Result { + match expr { + ScalarExpr::Literal(asap_types::pre_asap::ScalarValue::Float64(_)) + | ScalarExpr::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), + ScalarExpr::PromqlScalarFromVector(vector) => self.lower_promql_unary( + query, + occurrence, + PhysicalOperator::PromqlVectorToScalar, + vector, + ), + _ => Err(AnalyticalCostError::UnsupportedQueryOperator), + } + } + fn node_statistics(&self, id: &str) -> Result<&OperatorStatistics, AnalyticalCostError> { self.evidence .get(id) @@ -182,11 +217,14 @@ pub fn lower_query_physical_dag( fn lower_new( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, ) -> Result { - match query { - QueryExpr::Scan { + let Some(op) = query.non_asap() else { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + }; + match op { + NonASAPOp::Scan { source, predicates, .. } => { let coverage = bind_scan_coverage( @@ -252,13 +290,13 @@ pub fn lower_query_physical_dag( require_operator_statistics(filter_operator, &filter_evidence.statistics)?; self.push(filter_evidence, filter_operator, children, None) } - QueryExpr::Filter { pred, child } => { + NonASAPOp::Filter { pred, child } => { let operator = PhysicalOperator::Filter { predicate_operations_per_row: scalar_operation_count(&pred.0)?.max(1), }; self.lower_unary(query, occurrence, operator, child) } - QueryExpr::Project { cols, child, .. } => { + NonASAPOp::Project { cols, child, .. } => { let expression_operations_per_row = cols .iter() .try_fold(0_u64, |total, item| { @@ -280,7 +318,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Aggregate { + NonASAPOp::Aggregate { reduction, measures, filters, @@ -289,7 +327,7 @@ pub fn lower_query_physical_dag( .. } => { if having.is_some() - || asap_types::pre_asap::any_measure_filtered(filters) + || asap_types::ir::non_asap::any_measure_filtered(filters) || measures.is_empty() { return Err(AnalyticalCostError::UnsupportedQueryOperator); @@ -338,13 +376,9 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Dedup { cols, child } => { + NonASAPOp::Dedup { cols, child } => { let key_count = if cols.is_empty() { - child - .output_schema() - .map_err(|_| AnalyticalCostError::UnsupportedQueryOperator)? - .fields - .len() + child.schema.fields.len() } else { cols.len() }; @@ -361,7 +395,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Sort { + NonASAPOp::Sort { keys, partition_by, child, @@ -381,13 +415,26 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Limit { n, offset, child } => { - if let QueryExpr::Sort { + NonASAPOp::Limit { + n, + offset, + partition_by: limit_partition_by, + child, + } => { + // Offset-only and per-group limits have no physical + // operator here. + let Some(n) = n else { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + }; + if limit_partition_by != &GroupKeys::none() { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + } + if let Some(NonASAPOp::Sort { keys, partition_by, child: sorted_child, .. - } = child.as_ref() + }) = child.non_asap() { if !keys.is_empty() && partition_by == &GroupKeys::none() { let child_id = self.lower(sorted_child)?; @@ -442,7 +489,7 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::SQLWindowFunc { + NonASAPOp::SQLWindowFunc { func, partition_by, order_by, @@ -472,7 +519,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::TimeRange { range, child } => { + NonASAPOp::TimeRange { range, child, .. } => { let range_millis = duration_millis(*range, "range")?; self.lower_promql_unary( query, @@ -481,7 +528,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlSubquery { + NonASAPOp::PromqlSubquery { range, resolution, child, @@ -513,7 +560,7 @@ pub fn lower_query_physical_dag( } Ok(id) } - QueryExpr::PromqlRelabel { value, child, .. } => self.lower_promql_unary( + NonASAPOp::PromqlRelabel { value, child, .. } => self.lower_promql_unary( query, occurrence, PhysicalOperator::PromqlRelabel { @@ -521,7 +568,7 @@ pub fn lower_query_physical_dag( }, child, ), - QueryExpr::PromqlSeriesSample { + NonASAPOp::PromqlSeriesSample { by, kind, child, .. } => { if by.is_without() { @@ -557,7 +604,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlInfoEnrich { selector, child } => { + NonASAPOp::PromqlInfoEnrich { selector, child } => { let left_id = self.lower(child)?; let coverage = bind_info_coverage( &format!("occurrence-{occurrence}-info"), @@ -609,34 +656,13 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, + NonASAPOp::BinaryOp { + operator, lhs, rhs, .. } => { - let left_scalar = is_promql_scalar(lhs); - let right_scalar = is_promql_scalar(rhs); - if left_scalar && right_scalar { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } + let op = &operator.kind; + let vector_match = &operator.vector_match; let operation = promql_binary_operation(op); - if (left_scalar || right_scalar) - && !matches!(operation, PromqlBinaryOperation::ArithmeticOrComparison) - { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } - let operand_mode = match (left_scalar, right_scalar) { - (false, false) => PromqlBinaryOperandMode::VectorVector, - (false, true) => PromqlBinaryOperandMode::VectorScalar, - (true, false) => PromqlBinaryOperandMode::ScalarVector, - (true, true) => unreachable!("scalar/scalar returned above"), - }; - if operand_mode != PromqlBinaryOperandMode::VectorVector - && vector_match.is_some() - { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } + let operand_mode = PromqlBinaryOperandMode::VectorVector; let cardinality = promql_vector_cardinality(vector_match.as_ref()); let left_id = self.lower(lhs)?; let right_id = self.lower(rhs)?; @@ -670,41 +696,31 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::PromqlVectorFromScalar(child) => self.lower_promql_unary( - query, - occurrence, - PhysicalOperator::PromqlScalarToVector, - child, - ), - QueryExpr::PromqlScalarFromVector(child) => self.lower_promql_unary( - query, - occurrence, - PhysicalOperator::PromqlVectorToScalar, - child, - ), - QueryExpr::PromqlScalarBridge(inner) - if matches!( - inner.as_ref(), - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Float64(_)) - ) => - { - self.lower_promql_scalar_leaf(query, occurrence) + NonASAPOp::PromqlVectorFromScalar(scalar) => { + let scalar_occurrence = self.next_id; + self.next_id += 1; + let child_id = self.lower_scalar_operand(query, scalar_occurrence, scalar)?; + self.push_unary( + query, + occurrence, + PhysicalOperator::PromqlScalarToVector, + child_id, + ) } - QueryExpr::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), - QueryExpr::TimeShift { shift, child } => { + NonASAPOp::TimeShift { shift, child } => { if !shift.is_identity() { return Err(AnalyticalCostError::UnsupportedQueryOperator); } self.lower_unary(query, occurrence, PhysicalOperator::PassThrough, child) } - QueryExpr::Concat { children, .. } => { + NonASAPOp::Concat { children, .. } => { let child_ids = children .iter() .map(|child| self.lower(child)) .collect::, _>>()?; self.lower_concat(query, occurrence, child_ids) } - QueryExpr::SetOp { + NonASAPOp::SetOp { kind: RelationalSetOpKind::Union, all: true, left, @@ -714,7 +730,7 @@ pub fn lower_query_physical_dag( let right_id = self.lower(right)?; self.lower_concat(query, occurrence, vec![left_id, right_id]) } - QueryExpr::Join { + NonASAPOp::Join { kind, pred, left, @@ -764,7 +780,7 @@ pub fn lower_query_physical_dag( fn lower_concat( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, child_ids: Vec, ) -> Result { @@ -957,7 +973,7 @@ fn require_operator_statistics( fn bind_scan_coverage( node_id: &str, source: &asap_types::pre_asap::Source, - predicates: &[asap_types::pre_asap::Predicate], + predicates: &[asap_types::ir::Predicate], scope: &ComparisonScope, ) -> Result { let mut matches = scope.sources.iter().filter(|coverage| { @@ -1051,17 +1067,14 @@ fn promql_vector_cardinality( } fn hash_join_key_count( - expr: &asap_types::pre_asap::QueryExpr, - left: &asap_types::pre_asap::QueryExpr, - right: &asap_types::pre_asap::QueryExpr, + expr: &ScalarExpr, + left: &OperatorNode, + right: &OperatorNode, ) -> Option { - use asap_types::pre_asap::{CompareOpKind, QueryExpr}; + use asap_types::pre_asap::CompareOpKind; - let (Ok(left_schema), Ok(right_schema)) = (left.output_schema(), right.output_schema()) else { - return None; - }; - let left_width = left_schema.fields.len(); - let total_width = left_width.saturating_add(right_schema.fields.len()); + let left_width = left.schema.fields.len(); + let total_width = left_width.saturating_add(right.schema.fields.len()); fn column_side(column: usize, left_width: usize, total_width: usize) -> Option { if column < left_width { @@ -1073,14 +1086,15 @@ fn hash_join_key_count( } } - fn predicate(expr: &QueryExpr, left_width: usize, total_width: usize) -> Option { + fn predicate(expr: &ScalarExpr, left_width: usize, total_width: usize) -> Option { match expr { - QueryExpr::Compare { + ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, + .. } => match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(left), QueryExpr::Column(right)) => match ( + (ScalarExpr::Column(left), ScalarExpr::Column(right)) => match ( column_side(*left, left_width, total_width), column_side(*right, left_width, total_width), ) { @@ -1089,7 +1103,7 @@ fn hash_join_key_count( }, _ => None, }, - QueryExpr::BoolAnd(parts) if !parts.is_empty() => { + ScalarExpr::BoolAnd(parts) if !parts.is_empty() => { parts.iter().try_fold(0_u64, |count, part| { count.checked_add(predicate(part, left_width, total_width)?) }) @@ -1101,12 +1115,8 @@ fn hash_join_key_count( predicate(expr, left_width, total_width) } -fn scalar_operation_count( - expr: &asap_types::pre_asap::QueryExpr, -) -> Result { - use asap_types::pre_asap::QueryExpr; - - let add = |parts: &[&QueryExpr]| { +fn scalar_operation_count(expr: &ScalarExpr) -> Result { + let add = |parts: &[&ScalarExpr]| { parts.iter().try_fold(0_u64, |total, part| { total .checked_add(scalar_operation_count(part)?) @@ -1119,14 +1129,14 @@ fn scalar_operation_count( .ok_or(AnalyticalCostError::Overflow) }; match expr { - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => Ok(0), - QueryExpr::Compare { left, right, .. } | QueryExpr::Arithmetic { left, right, .. } => { + ScalarExpr::Column(_) + | ScalarExpr::Literal(_) + | ScalarExpr::EvalTimestamp + | ScalarExpr::CurrentTimestamp => Ok(0), + ScalarExpr::Compare { left, right, .. } | ScalarExpr::Arithmetic { left, right, .. } => { with_local(&[left, right]) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { let children = parts.iter().collect::>(); add(&children)? .checked_add( @@ -1135,12 +1145,11 @@ fn scalar_operation_count( ) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::Not(child) - | QueryExpr::IsNull(child) - | QueryExpr::IsNotNull(child) - | QueryExpr::PromqlScalarBridge(child) => with_local(&[child]), - QueryExpr::Cast { expr, .. } => with_local(&[expr]), - QueryExpr::InList { expr, list, .. } => { + ScalarExpr::Not(child) | ScalarExpr::IsNull(child) | ScalarExpr::IsNotNull(child) => { + with_local(&[child]) + } + ScalarExpr::Cast { expr, .. } => with_local(&[expr]), + ScalarExpr::InList { expr, list, .. } => { let mut children = Vec::with_capacity(list.len() + 1); children.push(expr.as_ref()); children.extend(list.iter()); @@ -1148,11 +1157,11 @@ fn scalar_operation_count( .checked_add(u64::try_from(list.len()).map_err(|_| AnalyticalCostError::Overflow)?) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::FunctionCall { args, .. } => { + ScalarExpr::FunctionCall { args, .. } => { let children = args.iter().collect::>(); with_local(&children) } - QueryExpr::Case { + ScalarExpr::Case { operand, branches, else_expr, @@ -1246,16 +1255,6 @@ fn fixed_state_per_series_intent(intent: &asap_types::pre_asap::AggIntent) -> bo ) } -fn is_promql_scalar(query: &asap_types::pre_asap::QueryExpr) -> bool { - use asap_types::pre_asap::QueryExpr; - matches!( - query, - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp - ) -} - #[cfg(test)] mod tests { use super::*; @@ -1266,6 +1265,7 @@ mod tests { validate_comparison_scopes, BinaryEdgeStatistics, PartitionStatistics, PromqlEdgeStatistics, PromqlUnaryEdgeStatistics, PromqlValueKind, UnaryEdgeStatistics, }; + use asap_types::ir::{BinaryOperator, ExprSemantics, Predicate, SortKey, TimeRangeKind}; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; @@ -1401,10 +1401,7 @@ mod tests { } } - fn coverage( - source: asap_types::pre_asap::Source, - predicates: Vec, - ) -> ScanSelection { + fn coverage(source: asap_types::pre_asap::Source, predicates: Vec) -> ScanSelection { ScanSelection { source, source_snapshot_id: "snapshot-1".into(), @@ -1451,27 +1448,30 @@ mod tests { // Correlation can be costed as an exact hash aggregate using provider-supplied state size. #[test] fn correlation_lowers_to_physical_hash_aggregate() { - use asap_types::pre_asap::{ - AggIntent, DataType, Field, QueryExpr, Reduction, Schema, Source, - }; + use asap_types::pre_asap::{AggIntent, DataType, Field, Reduction, Schema, Source}; let source = Source::Table { table_ref: "pairs".into(), }; - let root = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], - output_names: vec!["r".into()], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::Scan { - source: source.clone(), - predicates: vec![], - schema: Schema::new(vec![ - Field::plain("x", DataType::Float64, true), - Field::plain("y", DataType::Float64, true), - ]), - }), - }); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], + output_names: vec!["r".into()], + filters: vec![], + having: None, + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::Scan { + source: source.clone(), + predicates: vec![], + schema: Schema::new(vec![ + Field::plain("x", DataType::Float64, true), + Field::plain("y", DataType::Float64, true), + ]), + }, + )) + .unwrap(), + })) + .unwrap(); let scope = scope(vec![coverage(source, vec![])]); let provided = HashMap::from([ ( @@ -1501,51 +1501,57 @@ mod tests { #[test] fn query_lowering_recurses_and_fuses_global_sort_limit() { - use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr, Reduction, SortKey, Source}; + use asap_types::pre_asap::{AggIntent, GroupKeys, Reduction, Source}; use asap_types::pre_asap::{DataType, Field, Schema}; use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, - predicates: vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + predicates: vec![Predicate(ScalarExpr::Literal( + asap_types::pre_asap::ScalarValue::Boolean(true), ))], schema: Schema::new(vec![ Field::plain("service", DataType::Utf8, false), Field::plain("value", DataType::Float64, false), ]), - }); - let aggregate = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![0]), - measures: vec![AggIntent::Sum { col: Some(1) }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::clone(&scan), - }); - let sort = Rc::new(QueryExpr::Sort { + })) + .unwrap(); + let aggregate = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![0]), + measures: vec![AggIntent::Sum { col: Some(1) }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::clone(&scan), + })) + .unwrap(); + let sort = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: ScalarExpr::Column(0), ascending: false, nulls_first: false, }], partition_by: GroupKeys::none(), child: aggregate, - }); - let root = Rc::new(QueryExpr::Limit { - n: 10, + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), offset: 5, + partition_by: GroupKeys::none(), child: sort, - }); + })) + .unwrap(); let scan_coverage = coverage( Source::Table { table_ref: "events".into(), }, - vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + vec![Predicate(ScalarExpr::Literal( + asap_types::pre_asap::ScalarValue::Boolean(true), ))], ); let scope = scope(vec![scan_coverage]); @@ -1638,26 +1644,29 @@ mod tests { #[test] fn query_lowering_shares_only_provider_identified_physical_nodes() { use asap_types::pre_asap::{CompareOpKind, DataType, Field, Schema}; - use asap_types::pre_asap::{JoinKind, Predicate, QueryExpr, Source}; + use asap_types::pre_asap::{JoinKind, Source}; use std::rc::Rc; - let shared = Rc::new(QueryExpr::Scan { + let shared = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "dimensions".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let root = Rc::new(QueryExpr::Join { + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })), + right: Box::new(ScalarExpr::Column(1)), + semantics: ExprSemantics::Sql, + }), left: Rc::clone(&shared), right: Rc::clone(&shared), - }); + })) + .unwrap(); let scan_selection = coverage( Source::Table { table_ref: "dimensions".into(), @@ -1781,16 +1790,19 @@ mod tests { )) ); - let invalid = Rc::new(QueryExpr::Join { - kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(0)), - })), - left: Rc::clone(&shared), - right: Rc::clone(&shared), - }); + let invalid = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(0)), + semantics: ExprSemantics::Sql, + }), + left: Rc::clone(&shared), + right: Rc::clone(&shared), + })) + .unwrap(); assert_eq!( lower_query_physical_dag(&invalid, &shared_scope, &shared_provider), Err(AnalyticalCostError::UnsupportedQueryOperator) @@ -1800,62 +1812,73 @@ mod tests { #[test] fn query_lowering_covers_relational_unary_operators() { use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; - use asap_types::pre_asap::{ - GroupKeys, Predicate, QueryExpr, SortKey, Source, TimeShift, WindowFuncKind, - }; - use std::rc::Rc; + use asap_types::pre_asap::{GroupKeys, Source, TimeShift, WindowFuncKind}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: scan, - }); - let project = Rc::new(QueryExpr::Project { - cols: vec![], - qualifier: None, - child: filter, - }); - let dedup = Rc::new(QueryExpr::Dedup { + })) + .unwrap(); + let filter = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: scan, + })) + .unwrap(); + let project = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols: vec![], + qualifier: None, + child: filter, + })) + .unwrap(); + let dedup = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { cols: vec![0], child: project, - }); - let window = Rc::new(QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::none(), - order_by: vec![SortKey { - expr: QueryExpr::Column(0), - ascending: true, - nulls_first: false, - }], - frame: None, - output_name: "rn".into(), - child: dedup, - }); - let sort = Rc::new(QueryExpr::Sort { + })) + .unwrap(); + let window = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::SQLWindowFunc { + func: WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::none(), + order_by: vec![SortKey { + expr: ScalarExpr::Column(0), + ascending: true, + nulls_first: false, + }], + frame: None, + output_name: "rn".into(), + child: dedup, + }, + )) + .unwrap(); + let sort = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: ScalarExpr::Column(0), ascending: true, nulls_first: false, }], partition_by: GroupKeys::by(vec![0]), child: window, - }); - let limit = Rc::new(QueryExpr::Limit { - n: 20, + })) + .unwrap(); + let limit = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(20), offset: 0, + partition_by: GroupKeys::none(), child: sort, - }); - let root = Rc::new(QueryExpr::TimeShift { - shift: TimeShift::default(), - child: limit, - }); + })) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeShift { + shift: TimeShift::default(), + child: limit, + })) + .unwrap(); let scan_selection = coverage( Source::Table { @@ -1974,22 +1997,25 @@ mod tests { #[test] fn query_lowering_maps_concat_and_union_all_but_rejects_distinct_set_ops() { use asap_types::pre_asap::{DataType, Field, Schema}; - use asap_types::pre_asap::{QueryExpr, RelationalSetOpKind, Source}; - use std::rc::Rc; + use asap_types::pre_asap::{RelationalSetOpKind, Source}; - let scan = |name: &str| QueryExpr::Scan { - source: Source::Table { - table_ref: name.into(), - }, - predicates: vec![], - schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + let scan = |name: &str| { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: name.into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + })) + .unwrap() }; - let union = Rc::new(QueryExpr::SetOp { + let union = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::SetOp { kind: RelationalSetOpKind::Union, all: true, - left: Rc::new(scan("a")), - right: Rc::new(scan("b")), - }); + left: scan("a"), + right: scan("b"), + })) + .unwrap(); let scope = scope(vec![ coverage( Source::Table { @@ -2033,19 +2059,23 @@ mod tests { )) ); - let concat = Rc::new(QueryExpr::Concat { - children: vec![scan("a"), scan("b")], - discriminator_unique_key: None, - }); + let concat = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Concat { + children: vec![scan("a"), scan("b")], + discriminator_unique_key: None, + })) + .unwrap(); let dag = lower_query_physical_dag(&concat, &scope, &scripted(&provided)).unwrap(); assert_eq!(dag.nodes.last().unwrap().operator, PhysicalOperator::Concat); - let distinct_union = Rc::new(QueryExpr::SetOp { - kind: RelationalSetOpKind::Union, - all: false, - left: Rc::new(scan("a")), - right: Rc::new(scan("b")), - }); + let distinct_union = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::SetOp { + kind: RelationalSetOpKind::Union, + all: false, + left: scan("a"), + right: scan("b"), + })) + .unwrap(); assert_eq!( lower_query_physical_dag(&distinct_union, &scope, &scripted(&provided)), Err(AnalyticalCostError::UnsupportedQueryOperator) @@ -2054,22 +2084,24 @@ mod tests { #[test] fn query_lowering_fails_closed_for_missing_or_inconsistent_statistics() { + use asap_types::pre_asap::Source; use asap_types::pre_asap::{DataType, Field, Schema}; - use asap_types::pre_asap::{QueryExpr, Source}; - use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let root = Rc::new(QueryExpr::Project { - cols: vec![], - qualifier: None, - child: scan, - }); + })) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols: vec![], + qualifier: None, + child: scan, + })) + .unwrap(); let comparison_scope = scope(vec![coverage( Source::Table { @@ -2171,25 +2203,29 @@ mod tests { #[test] fn query_lowering_accepts_a_consistently_empty_edge() { use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; - use asap_types::pre_asap::{Predicate, QueryExpr, Source}; - use std::rc::Rc; + use asap_types::pre_asap::{GroupKeys, Source}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(false)))), - child: scan, - }); - let root = Rc::new(QueryExpr::Limit { - n: 10, + })) + .unwrap(); + let filter = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(false))), + child: scan, + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), offset: 0, + partition_by: GroupKeys::none(), child: filter, - }); + })) + .unwrap(); let scope = scope(vec![coverage( Source::Table { @@ -2230,59 +2266,70 @@ mod tests { #[test] fn query_lowering_rejects_aggregates_without_a_hash_implementation() { - use asap_types::pre_asap::{ - AggIntent, GroupKeys, QueryExpr, Reduction, Source, WindowFuncKind, - }; + use asap_types::pre_asap::{AggIntent, GroupKeys, Reduction, Source, WindowFuncKind}; use asap_types::pre_asap::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; - use std::rc::Rc; let scan = || { - Rc::new(QueryExpr::Scan { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }) + })) + .unwrap() }; - let exact_quantile = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Quantile { - col: Some(0), - q: 0.99, - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: scan(), - }); - let empty_sort_limit = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child: Rc::new(QueryExpr::Sort { - keys: vec![], + let exact_quantile = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![AggIntent::Quantile { + col: Some(0), + q: 0.99, + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + })) + .unwrap(); + let empty_sort_limit = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), + offset: 0, partition_by: GroupKeys::none(), + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::Sort { + keys: vec![], + partition_by: GroupKeys::none(), + child: scan(), + }, + )) + .unwrap(), + })) + .unwrap(); + let unsupported_window = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::SQLWindowFunc { + func: WindowFuncKind::Lag, + args: vec![ScalarExpr::Column(0)], + partition_by: GroupKeys::none(), + order_by: vec![], + frame: None, + output_name: "lag".into(), child: scan(), - }), - }); - let unsupported_window = Rc::new(QueryExpr::SQLWindowFunc { - func: WindowFuncKind::Lag, - args: vec![QueryExpr::Column(0)], - partition_by: GroupKeys::none(), - order_by: vec![], - frame: None, - output_name: "lag".into(), - child: scan(), - }); - let shifted = Rc::new(QueryExpr::TimeShift { - shift: asap_types::pre_asap::TimeShift { - offset_ms: 60_000, - at: None, }, - child: scan(), - }); + )) + .unwrap(); + let shifted = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeShift { + shift: asap_types::pre_asap::TimeShift { + offset_ms: 60_000, + at: None, + }, + child: scan(), + })) + .unwrap(); let scope = scope(vec![coverage( Source::Table { table_ref: "events".into(), @@ -2305,40 +2352,42 @@ mod tests { #[test] fn scalar_work_counts_every_local_predicate_operation() { - use asap_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use asap_types::pre_asap::{CompareOpKind, ScalarValue}; - let comparison = || QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let comparison = || ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(1))), + semantics: ExprSemantics::Sql, }; - let predicate = QueryExpr::BoolAnd(vec![comparison(), comparison()]); + let predicate = ScalarExpr::BoolAnd(vec![comparison(), comparison()]); assert_eq!(scalar_operation_count(&predicate), Ok(3)); } #[test] fn promql_presence_is_lowered_with_a_per_step_output_bound() { - use asap_types::pre_asap::{ - AggIntent, DataType, Field, QueryExpr, Reduction, Schema, Source, - }; + use asap_types::pre_asap::{AggIntent, DataType, Field, Reduction, Schema, Source}; let source = Source::TimeSeries { metric: "missing".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }); - let root = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Absent], - output_names: vec![], - filters: vec![], - having: None, - child: scan, - }); + })) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![AggIntent::Absent], + output_names: vec![], + filters: vec![], + having: None, + child: scan, + })) + .unwrap(); let vector = promql_edge(0, 2, PromqlValueKind::Vector); let scan_statistics = OperatorStatistics::Scan { edges: promql_unary_edges(edge(0, 0), edge(0, 0), vector, vector), @@ -2387,24 +2436,31 @@ mod tests { #[test] fn promql_range_and_subquery_preserve_internal_steps() { - use asap_types::pre_asap::{DataType, Field, QueryExpr, Schema, Source}; + use asap_types::pre_asap::{DataType, Field, Schema, Source}; use std::time::Duration; let source = Source::TimeSeries { metric: "m".into() }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }); - let range = Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: scan, - }); - let root = Rc::new(QueryExpr::PromqlSubquery { - range: Duration::from_secs(300), - resolution: Some(Duration::from_secs(60)), - child: range, - }); + })) + .unwrap(); + let range = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + range: Duration::from_secs(300), + kind: TimeRangeKind::Range, + child: scan, + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlSubquery { + range: Duration::from_secs(300), + resolution: Some(Duration::from_secs(60)), + child: range, + }, + )) + .unwrap(); let vector = promql_edge(10, 6, PromqlValueKind::Vector); let range_vector = promql_edge(10, 6, PromqlValueKind::RangeVector); let outer_range = promql_edge(10, 1, PromqlValueKind::RangeVector); @@ -2462,32 +2518,40 @@ mod tests { #[test] fn promql_binary_lowering_keeps_operation_and_matching_cardinality() { use asap_types::pre_asap::{ - ArithmeticOpKind, BinaryOpKind, DataType, Field, GroupSide, QueryExpr, Schema, Source, + ArithmeticOpKind, BinaryOpKind, DataType, Field, GroupSide, Schema, Source, VectorGrouping, VectorMatch, VectorMatchKind, }; let left_source = Source::TimeSeries { metric: "a".into() }; let right_source = Source::TimeSeries { metric: "b".into() }; let scan = |source| { - Rc::new(QueryExpr::Scan { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source, predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }) + })) + .unwrap() }; - let root = Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: scan(left_source.clone()), - rhs: scan(right_source.clone()), - vector_match: Some(VectorMatch { - kind: VectorMatchKind::On, - labels: vec!["service".into()], - grouping: Some(VectorGrouping { - side: GroupSide::Left, - labels: vec!["region".into()], - }), - }), - }); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + vector_match: Some(VectorMatch { + kind: VectorMatchKind::On, + labels: vec!["service".into()], + grouping: Some(VectorGrouping { + side: GroupSide::Left, + labels: vec!["region".into()], + }), + }), + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: scan(left_source.clone()), + rhs: scan(right_source.clone()), + })) + .unwrap(); let left_promql = promql_edge(10, 10, PromqlValueKind::Vector); let right_promql = promql_edge(5, 10, PromqlValueKind::Vector); let output_promql = promql_edge(8, 10, PromqlValueKind::Vector); @@ -2545,36 +2609,45 @@ mod tests { #[test] fn promql_relabel_sample_and_per_series_lower_as_a_complete_chain() { use asap_types::pre_asap::{ - AggIntent, DataType, Field, GroupKeys, QueryExpr, Reduction, SampleKind, ScalarValue, - Schema, Source, + AggIntent, DataType, Field, GroupKeys, Reduction, SampleKind, ScalarValue, Schema, + Source, }; let source = Source::TimeSeries { metric: "requests".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }); - let relabel = Rc::new(QueryExpr::PromqlRelabel { - dst: "service".into(), - value: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api".into()))), - child: scan, - }); - let sample = Rc::new(QueryExpr::PromqlSeriesSample { - by: GroupKeys::none(), - kind: SampleKind::LimitK(5), - child: relabel, - }); - let root = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: sample, - }); + })) + .unwrap(); + let relabel = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlRelabel { + dst: "service".into(), + value: ScalarExpr::Literal(ScalarValue::Utf8("api".into())), + child: scan, + }, + )) + .unwrap(); + let sample = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlSeriesSample { + by: GroupKeys::none(), + kind: SampleKind::LimitK(5), + child: relabel, + }, + )) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: sample, + })) + .unwrap(); let input = edge(100, 1_600); let sampled = edge(50, 800); diff --git a/crates/asap-aware-mapping/src/recurrence.rs b/crates/asap-aware-mapping/src/recurrence.rs index 660787cdb..835a35a65 100644 --- a/crates/asap-aware-mapping/src/recurrence.rs +++ b/crates/asap-aware-mapping/src/recurrence.rs @@ -394,7 +394,7 @@ pub enum RootRecurrence { // ── Explanation ────────────────────────────────────────────────────────── -/// The full readout [`CostModel::cse_share_decision_with_recurrence`] +/// The full evaluation [`CostModel::cse_share_decision_with_recurrence`] /// returns: which alternative was selected, both compared cost rates /// (and, when a [`Horizon`] was supplied, both compared totals), every /// input that went into them, their units, and provenance — meant to be @@ -779,17 +779,20 @@ mod tests { // ── decide (structural fallback) ───────────────────────────────────── use crate::cost_model::CseCandidate; + use asap_types::ir::operator_properties::{Reduction, Source}; + use asap_types::ir::{ + ASAPOp, BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, + }; use asap_types::post_asap::{ ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, ResultGuarantee, Schema, - SummaryExpr, SummaryNode, }; use asap_types::pre_asap::expr_ir::ColumnRef; - use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; use asap_types::pre_asap::schema::DataType; + use std::rc::Rc; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -800,28 +803,34 @@ mod tests { 0, vec![], ), - } + })) + .unwrap() } - fn summary_node(family: FieldDataType) -> SummaryNode { - SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: Schema::lifted(vec![], None), - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), + /// A summary of `family` over the kept pre-ASAP scan. + fn summary_node(family: FieldDataType) -> Rc { + let kept = Rc::new( + scan() + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("RetainedExact"))), + ); + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: kept, + family: family.clone(), + input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( + "value".into(), + )), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, }), - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( - "value".into(), - )), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: None, - } + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(None), + ) } #[test] @@ -1120,9 +1129,9 @@ mod tests { // ── multiple roots sharing a sub-DAG, via CandidateLogicalASAPDAGs ────────────────── use crate::replacement::search_workload; + use asap_types::ir::operator_properties::Reduction as QueryReduction; use asap_types::pre_asap::agg_intent::AggIntent; - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::{Predicate, Reduction as QueryReduction}; + use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; /// Like `scan()`, plus a "job" label column to group by — CSE's /// sharing legality gate requires a provable unique key @@ -1132,8 +1141,8 @@ mod tests { /// real one, matching the pattern /// `replacement.rs`'s own CSE fixtures already use (`metric_scan`/`agg` /// grouped by a label column). - fn labeled_scan() -> QueryExpr { - QueryExpr::Scan { + fn labeled_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -1145,18 +1154,20 @@ mod tests { 0, vec![], ), - } + })) + .unwrap() } - fn sum_agg() -> QueryExpr { - QueryExpr::Aggregate { + fn sum_agg() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: QueryReduction::by(vec![2]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(labeled_scan()), - } + child: labeled_scan(), + })) + .unwrap() } /// A root wrapping a fresh, independently-built (but structurally @@ -1167,13 +1178,19 @@ mod tests { /// `shared_aggregate_across_two_roots_gets_both_strategies_candidates`'s /// own doc) while letting `share_common_sub_dags` unify their /// identical `sum_agg()` children onto one shared `Rc`. - fn filtered_root(distinguishing_literal: i64) -> QueryExpr { - QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64( - distinguishing_literal, - )))), - child: Rc::new(sum_agg()), - } + fn filtered_root(distinguishing_literal: i64) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(1)), + op: CompareOpKind::Gt, + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64( + distinguishing_literal, + ))), + semantics: ExprSemantics::Sql, + }), + child: sum_agg(), + })) + .unwrap() } /// Three workload roots share one underlying `sum_agg()` sub-DAG: two @@ -1185,10 +1202,10 @@ mod tests { /// roots sharing a sub-DAG" acceptance criteria. #[test] fn recurrence_profiles_aggregates_mixed_intervals_across_roots_sharing_a_subdag() { - let roots: Vec<(&str, Rc)> = vec![ - ("root_a", Rc::new(filtered_root(1))), - ("root_b", Rc::new(filtered_root(2))), - ("root_c", Rc::new(filtered_root(3))), + let roots: Vec<(&str, Rc)> = vec![ + ("root_a", filtered_root(1)), + ("root_b", filtered_root(2)), + ("root_c", filtered_root(3)), ]; let space = search_workload(roots); @@ -1206,7 +1223,7 @@ mod tests { ); let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("the shared sum_agg() is a discovered target"); assert_eq!(shared_group.consumer_count, 3, "shared by all 3 roots"); @@ -1250,14 +1267,11 @@ mod tests { #[test] fn plan_selection_uses_recurrence_profiles_for_cse_choices() { - let roots = vec![ - ("a", Rc::new(filtered_root(1))), - ("b", Rc::new(filtered_root(2))), - ]; + let roots = vec![("a", filtered_root(1)), ("b", filtered_root(2))]; let space = search_workload(roots); let shared = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("the aggregate is shared by both roots"); let update_rate = Some(UpdateRate(10.0)); @@ -1315,8 +1329,8 @@ mod tests { #[test] fn recurrence_profiles_rejects_an_invalid_evaluation_rate() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space .recurrence_profiles(&[RootRecurrence::Repeating(EvaluationRate(f64::NAN))], None) @@ -1329,8 +1343,8 @@ mod tests { /// signature promises a `Result`. #[test] fn recurrence_profiles_reports_a_root_count_mismatch_as_an_error_not_a_panic() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space.recurrence_profiles(&[], None).unwrap_err(); assert_eq!( @@ -1344,8 +1358,8 @@ mod tests { #[test] fn recurrence_profiles_rejects_an_invalid_update_rate() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); let err = space .recurrence_profiles( @@ -1373,23 +1387,25 @@ mod tests { /// `consumer_count`. #[test] fn recurrence_profiles_does_not_stamp_update_rate_on_a_site_unreachable_from_any_root() { - let avg_root = QueryExpr::Aggregate { - reduction: QueryReduction::by(vec![]), - measures: vec![AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let roots: Vec<(&str, Rc)> = vec![("q", Rc::new(avg_root))]; + let avg_root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: QueryReduction::by(vec![]), + measures: vec![AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + })) + .unwrap(); + let roots: Vec<(&str, Rc)> = vec![("q", avg_root)]; let space = search_workload(roots); let count_group = space .target_subdag_candidates() .find(|g| { matches!( - g.target.as_ref(), - QueryExpr::Aggregate { measures, .. } + g.target.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if measures.iter().any(|m| matches!(m, AggIntent::Count { .. })) ) }) @@ -1428,19 +1444,26 @@ mod tests { /// reachability-set walk would (wrongly) collapse it to. #[test] fn recurrence_profiles_credits_a_direct_repeated_reference_by_its_multiplicity() { - let root = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( - asap_types::pre_asap::expr_ir::CompareOpKind::Eq, - ), - lhs: Rc::new(sum_agg()), - rhs: Rc::new(sum_agg()), - vector_match: None, - }; - let space = search_workload(vec![("q", Rc::new(root))]); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: asap_types::ir::operator_properties::BinaryOpKind::Compare( + CompareOpKind::Eq, + ), + vector_match: None, + }, + return_bool: false, + lhs: sum_agg(), + rhs: sum_agg(), + })) + .unwrap(); + let space = search_workload(vec![("q", root)]); let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("sum_agg() should merge onto one shared Rc, referenced twice from BinaryOp"); assert_eq!( shared_group.consumer_count, 2, @@ -1463,7 +1486,7 @@ mod tests { let scan_group = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Scan { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Scan { .. }))) .expect("the shared aggregate has a scan descendant"); assert_eq!( profiles diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 44ffef189..a3ff4dbd4 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -3,9 +3,9 @@ //! under "Key concepts (not yet implemented)", implemented for real (issue //! #251, part of #33). //! -//! ## One step, not two: `SketchAlgorithmStrategy::replacements()` decides *and* builds +//! ## One step, not two: `ASAPStrategies::replacements()` decides *and* builds //! -//! For a bindable `Aggregate`, `SketchAlgorithmStrategy::replacements()` is the +//! For a bindable `Aggregate`, `ASAPStrategies::replacements()` is the //! single place this crate both decides what an `AggIntent` may become and //! turns each of those candidates into a real, executable //! [`ReplacementSubDAG`]: @@ -18,8 +18,8 @@ //! `CostModel::size_params`, not a placeholder filled in later). //! 2. **Build**: for each candidate in that list, [`construct_summary`] //! mechanically turns the already-decided `(kind, params)` into a real -//! [`SummaryNode`] — derives the child schema, resolves the summarized -//! column, builds the readout query, recurses into the child (via +//! [`OperatorNode`] — derives the child schema, resolves the summarized +//! column, builds the evaluation query, recurses into the child (via //! [`realize_child`], so a nested aggregate gets its own //! independent enumeration, never the outer target's forced choice), and //! assembles the `SummaryAgg`/`SummaryEstimate` node. @@ -31,13 +31,13 @@ //! has to run regardless of how `(kind, params)` were chosen, so it lives //! directly inside the one method that needs it. //! -//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`QueryExpr`] node that is a +//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`OperatorNode`] that is a //! candidate for replacement, plus how many places in the workload already //! reference it (its `consumer_count`) — the one piece of cross-node //! context [`SharedSubDAGStrategy`] needs that a bare node reference alone //! doesn't carry. //! - [`ReplacementSubDAG`] — one candidate replacement for a `TargetSubDAG`: -//! either a fully bound [`SummaryNode`] or a pre-ASAP [`QueryExpr`] rewrite +//! either a fully bound summary sub-DAG or a pre-ASAP logical rewrite //! (still logical, structurally different from the target but semantically //! equivalent) — see [`Replacement`] — plus a human-readable `rationale`. //! - [`ReplacementStrategy`] — `matches` + `replacements`, the same @@ -72,7 +72,7 @@ //! //! ## The two strategies, and why these two //! -//! - [`SketchAlgorithmStrategy`] wraps [`realizations_for_intent`]'s exhaustive, +//! - [`ASAPStrategies`] wraps [`realizations_for_intent`]'s exhaustive, //! ranked list directly: for the same bindable-`Aggregate` shape this crate //! binds (single intent, no `HAVING`), every entry becomes its own bound //! candidate. @@ -80,7 +80,7 @@ //! `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing decision. //! Wherever a [`TargetSubDAG`] already has two or more consumers (i.e. //! `share_common_sub_dags` already collapsed two or more workload -//! locations onto the same `Rc` — [`discover_targets`] below +//! locations onto the same `Rc` — [`discover_targets`] below //! does the identical workload-wide discovery for [`search_workload_with`]; //! this module's own tests reuse the same dedup logic to build realistic //! fixtures), it reports the two-way candidate CSE's own detection pass @@ -98,7 +98,7 @@ //! unchanged.** Same inputs still produce the same exhaustive, ranked //! list — only its home moved (from a separate `implementation` module //! into this one) and its own visibility dropped to module-private, since -//! [`SketchAlgorithmStrategy`] is now its only caller. +//! [`ASAPStrategies`] is now its only caller. //! //! ## Workload-wide search — merged in from the former `search.rs` (issue #252, part of #33) //! @@ -143,12 +143,12 @@ //! //! 1. **Per-target candidates, not flat plans.** [`TargetSubDAGCandidates`] //! stores the alternatives for one distinct [`TargetSubDAG`] (identified by -//! its own `Rc` pointer identity — the same currency +//! its own `Rc` pointer identity — the same currency //! [`asap_types::pre_asap::cse::share_common_sub_dags`] already //! established across the workload) holding every //! [`ReplacementSubDAG`] alternative discovered for it. [`CandidateLogicalASAPDAGs`] is //! a collection of these groups, keyed by `TargetSubDAG` — a candidate -//! "plan" is never materialized as a distinct top-level `Rc` +//! "plan" is never materialized as a distinct top-level `Rc` //! at all; two logically-different overall choices at two different //! targets are just two different entries in two different groups, //! sharing every other node in the workload by construction (they *are* @@ -157,7 +157,7 @@ //! discipline.** [`asap_types::pre_asap::cse::structural_hash`] (made //! `pub` for exactly this reuse) is only ever a candidate-narrowing //! filter; [`TargetSubDAGCandidates::add_candidate`]'s actual duplicate check is -//! `QueryExpr`'s derived `PartialEq` — the same "hash is a filter, +//! `OperatorNode`'s derived `PartialEq` — the same "hash is a filter, //! `PartialEq` is the decision, no exceptions" rule `cse.rs`'s own //! "Correctness" section states and this module inherits rather than //! reinvents. See [`is_duplicate_rewrite`] for the one deliberate @@ -212,10 +212,10 @@ //! an alternative *for* the target just processed, not a new target of its //! own; see [`discover_new_descendant_targets`]) are scanned for pointers //! not already known, and any found become next round's frontier. Both shipped -//! strategies are idempotent in exactly this sense: [`SketchAlgorithmStrategy`] -//! produces terminal [`Replacement::Summary`] candidates (no `QueryExpr` -//! children to scan at all), and [`SharedSubDAGStrategy`]'s two -//! [`Replacement::Rewrite`] candidates both reuse the target's own +//! strategies are idempotent in exactly this sense: [`ASAPStrategies`] +//! produces terminal bound-summary [`Replacement::SubDAG`] candidates (no +//! logical-rewrite children to scan at all), and [`SharedSubDAGStrategy`]'s +//! two logical-rewrite [`Replacement::SubDAG`] candidates both reuse the target's own //! already-known child `Rc`s verbatim (`Rc::clone`/a shallow top-level //! `.clone()` — see that strategy's own doc). So for both, the frontier is //! always empty after round one: real workloads converge in exactly one @@ -248,7 +248,7 @@ //! share-vs-recompute pair is ranked by calling //! [`CostModel::cse_share_decision`] via this module's own //! [`cse_preference`] — rather than re-deriving a competing comparison. -//! - A group whose candidates are [`SketchAlgorithmStrategy`]'s sketch-family +//! - A group whose candidates are [`ASAPStrategies`]'s sketch-family //! candidates is ranked via [`CostModel::rank_candidates`] (the same hook //! `realizations_for_intent` itself consults), applied to the //! candidates' own [`SketchAlgorithm`]s. @@ -317,8 +317,8 @@ //! documented follow-up rather than silently overclaimed: //! //! - [`CostModel::rank_candidates`]/[`CostModel::size_params`] — the hooks -//! [`SketchAlgorithmStrategy`] groups rank by — take no `consumer_count` -//! parameter at all today, so a `SketchAlgorithmStrategy` group's selection +//! [`ASAPStrategies`] groups rank by — take no `consumer_count` +//! parameter at all today, so a `ASAPStrategies` group's selection //! here still falls back to [`rank_group`]'s ordinary (consumer-count- //! blind) local ranking, even though its own //! [`TargetSubDAGSelection::effective_consumer_count`] is computed and exposed @@ -345,31 +345,33 @@ use crate::accuracy::estimators::{ cms::{cms_depth, cms_width}, saturating_ceil, }; +use asap_types::ir::non_asap::any_measure_filtered; +use asap_types::pre_asap::resolve_column_ref; use std::cell::RefCell; use std::collections::{HashMap, HashSet, VecDeque}; +use asap_types::ir::cse::{share_common_sub_dags, structural_hash, HashCache}; +use asap_types::ir::operator_properties::{BinaryOpKind, JoinKind, Reduction}; +use asap_types::ir::timing::validate_default; +use asap_types::ir::SchemaDerivationError; +use asap_types::ir::{ + ASAPOp, BinaryOperator, NonASAPOp, Operator, OperatorNode, Predicate, ProjectItem, ScalarExpr, + SortKey, +}; +use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; use asap_types::post_asap::{ - validate_execution_data_states_at, EntityIdentity, ExactKind, ExactOperation, - ExactOperationSchemaError, ExactParams, ExecutionDataState, ExecutionDataStateError, + EntityIdentity, ExactKind, ExactOperationSchemaError, ExactParams, ExecutionDataStateError, ExecutionTiming, Field, FieldDataType, GroupingStrategy, NonNegativeWeightProof, SamplingKind, SamplingParams, Schema, SketchAlgorithm, SketchKind, SketchParams, - SketchStatistic as PostAsapSketchStatistic, StatModelKind, StatModelParams, SummaryExpr, - SummaryInputExpr, SummaryNode, SummaryUpdate, ValueOperation, WaveletKind, WaveletParams, - WeightDomain, + SketchStatistic as PostAsapSketchStatistic, StatModelKind, StatModelParams, SummaryInputExpr, + SummaryUpdate, WaveletKind, WaveletParams, WeightDomain, }; -use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; use asap_types::pre_asap::agg_intent::{agg_is_mergeable, AggIntent}; -use asap_types::pre_asap::column_resolution::resolve_column_ref; -use asap_types::pre_asap::cse::{share_common_sub_dags, structural_hash, HashCache}; use asap_types::pre_asap::expr_ir::{ArithmeticOpKind, ColumnRef}; -use asap_types::pre_asap::query_expr::any_measure_filtered; -use asap_types::pre_asap::query_expr::{ - BinaryOpKind, Predicate, QueryExpr, QueryExprError, Reduction, -}; use asap_types::pre_asap::schema::ColumnId; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataWorkload, QueryRecurrence, QueryWorkload, RepeatedDemand}; -use std::rc::Rc; +use std::rc::{Rc, Weak}; use thiserror::Error; use crate::accuracy::reconciliation::AccuracyReconciliationStrategy; @@ -391,20 +393,20 @@ use crate::rollup::RollupStrategy; use crate::topk_reuse::TopKLimitReuseStrategy; /// Errors from the pre-ASAP → post-ASAP replacement/construction path -/// ([`realize_child`] and [`keep_pre_asap`]). Moved here from the former +/// ([`realize_child`] and [`retain_exact`]). Moved here from the former /// `bind.rs` (issue #251): this is what a [`ReplacementStrategy`] /// implementor's own construction path can realistically fail with — -/// schema derivation over a pre-ASAP [`QueryExpr`] — not something specific -/// to workload-wide orchestration. +/// schema derivation over a pre-ASAP [`OperatorNode`] sub-DAG — not +/// something specific to workload-wide orchestration. #[derive(Debug, Error)] pub enum RealizationError { /// Schema derivation failed while lifting an edge to `Schema`. #[error("schema derivation failed during pre-ASAP → post-ASAP binding: {0}")] - Schema(#[from] QueryExprError), + Schema(#[from] SchemaDerivationError), /// The candidate is accuracy-illegal (issue #172): its composed /// guarantee has no sound propagation rule, or misses the applicable /// `AccuracyTarget`. Fail-closed — the candidate is never constructed - /// with the child "treated as exact". [`SketchAlgorithmStrategy::propose`] + /// with the child "treated as exact". [`ASAPStrategies::propose`] /// records it as a [`RejectedCandidate`] instead of a candidate. #[error("accuracy-illegal candidate: {0}")] Accuracy(#[from] AccuracyError), @@ -414,8 +416,8 @@ pub enum RealizationError { /// would change its semantics. #[error("unsupported physical summary realization: {0}")] PhysicalRealization(&'static str), - /// A constructed plan violates the update/readout phase contract - /// (issue #171) — e.g. a summary readout placed beneath a maintained + /// A constructed plan violates the update/evaluation phase contract + /// (issue #171) — e.g. a summary evaluation placed beneath a maintained /// `SummaryAgg`. Detected at construction, never at runtime. #[error("execution-data_state violation in post-ASAP plan: {0}")] ExecutionDataState(#[from] ExecutionDataStateError), @@ -427,10 +429,10 @@ pub enum RealizationError { /// A pre-ASAP sub-DAG a [`ReplacementStrategy`] knows how to replace. /// -/// `root` is a reference into the workload's own [`QueryExpr`] DAG (an -/// `Rc`, the same currency [`search_workload`] and -/// `asap_types::pre_asap::cse::share_common_sub_dags` already thread through -/// this crate's public API — not a bare `&QueryExpr` — so a strategy that +/// `root` is a reference into the workload's own [`OperatorNode`] DAG (an +/// `Rc`, the same currency [`search_workload`] and +/// `asap_types::ir::cse::share_common_sub_dags` already thread through +/// this crate's public API — not a bare `&OperatorNode` — so a strategy that /// needs the node's own `Rc` identity, not just its shape, has it available /// without the caller re-deriving it). /// @@ -439,16 +441,16 @@ pub enum RealizationError { /// [`search_workload_with`] computes the workload-wide value during target /// discovery. [`TargetSubDAG::new`] defaults it to `1` for callers invoking a /// strategy against one node in isolation. A strategy that only cares about -/// `root`'s shape (for example, [`SketchAlgorithmStrategy`]) can ignore the +/// `root`'s shape (for example, [`ASAPStrategies`]) can ignore the /// count; [`SharedSubDAGStrategy`] consults it directly. /// /// `strictest_sibling_accuracy` is the strictest accuracy among workload /// siblings that read the same summary input as `root`, when stricter than -/// `root`'s own. [`search_workload_with`] sets it; [`SketchAlgorithmStrategy`] +/// `root`'s own. [`search_workload_with`] sets it; [`ASAPStrategies`] /// also sizes a candidate to it. #[derive(Debug, Clone, Copy)] pub struct TargetSubDAG<'a> { - pub root: &'a Rc, + pub root: &'a Rc, pub consumer_count: usize, pub strictest_sibling_accuracy: Option<&'a AccuracyTarget>, } @@ -456,7 +458,7 @@ pub struct TargetSubDAG<'a> { impl<'a> TargetSubDAG<'a> { /// A target assumed to have exactly one consumer — the common case for a /// caller that isn't already tracking cross-workload sharing. - pub fn new(root: &'a Rc) -> Self { + pub fn new(root: &'a Rc) -> Self { Self { root, consumer_count: 1, @@ -466,7 +468,7 @@ impl<'a> TargetSubDAG<'a> { /// A target with an explicit `consumer_count`, used by workload discovery /// and by callers that already know how many locations reference `root`. - pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { + pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { Self { root, consumer_count, @@ -482,25 +484,35 @@ impl<'a> TargetSubDAG<'a> { /// — into "one candidate among several", each with its own /// [`ReplacementSubDAG`]. #[derive(Debug, Clone)] +#[allow(clippy::large_enum_variant)] // Keep the public strategy API value-based. pub enum Replacement { - /// A fully bound post-ASAP summary decision, for one particular - /// candidate realization of the target. - Summary(Rc), - /// A pre-ASAP rewrite: still a logical [`QueryExpr`], structurally - /// different from the target's own `root` (e.g. sharing vs. not sharing - /// a sub-DAG) but semantically equivalent to it. - Rewrite(Rc), + /// A sub-DAG that replaces the target: either a bound summary decision + /// (a DAG containing ASAP operators, for one particular candidate + /// realization of the target) or a pre-ASAP rewrite (a logical sub-DAG + /// with no ASAP operator, structurally different from the target's own + /// `root` — e.g. sharing vs. not sharing a sub-DAG — but semantically + /// equivalent to it). [`is_logical_rewrite`] tells the two apart. + SubDAG(Rc), /// An exact operator composed over another target's *own* selected - /// decision across an explicit update/readout boundary (issue #171): - /// `ValueOperationAtQueryTime` over a child's summary readout, or + /// decision across an explicit update/evaluation boundary (issue #171): + /// `ValueOperationAtQueryTime` over a child's summary evaluation, or /// `ValueOperationAtIngestionTime` feeding a maintained summary above. Carries only a /// reference to the child target — [`CandidateLogicalASAPDAGs::global_selection`] /// commits the compatible parent/child pair and /// [`GlobalSelection::assemble_selected_dag`] links it into one validated - /// `SummaryNode`. See [`crate::exact_composition`]. + /// `OperatorNode` DAG. See [`crate::exact_composition`]. ExactComposition(ExactComposition), } +/// Whether a [`Replacement::SubDAG`] is a pure logical rewrite: a sub-DAG +/// with no ASAP operator and no guarantee established yet (the shape every +/// front end emits and every rewrite strategy builds). A bound summary +/// decision contains an ASAP operator, or is a kept pre-ASAP sub-DAG that +/// already carries its exact guarantee. +pub fn is_logical_rewrite(node: &OperatorNode) -> bool { + node.guarantee.is_none() && !node.contains_asap() +} + /// One candidate replacement for a [`TargetSubDAG`], plus a human-readable /// `rationale` explaining why it's a valid candidate (meant for a /// report/log/debugging a search engine's choices, not machine parsing — @@ -523,11 +535,12 @@ pub struct ReplacementSubDAG { impl ReplacementSubDAG { /// Whether this summary still needs accuracy/domain evidence before it can /// be treated as certified. A missing guarantee on any summary candidate - /// is unknown; exact `KeepPreAsap` carries an explicit exact guarantee. + /// (a sub-DAG whose root is an ASAP operator) is unknown; a kept + /// pre-ASAP sub-DAG carries an explicit exact guarantee. pub fn has_missing_accuracy_evidence(&self) -> bool { matches!( &self.replacement, - Replacement::Summary(node) if has_missing_accuracy_evidence(node) + Replacement::SubDAG(node) if !is_logical_rewrite(node) && has_missing_accuracy_evidence(node) ) } @@ -539,8 +552,12 @@ impl ReplacementSubDAG { Replacement::ExactComposition(composition) => { cost_model.value_operation_support_evidence(&composition.op, composition.placement) } - Replacement::Summary(node) => cost_model.summary_support_evidence(node), - Replacement::Rewrite(_) => Some(true), + // Any summary decision, including one rooted in a relational + // operator above its evaluations, asks the deployment for support. + Replacement::SubDAG(node) if !is_logical_rewrite(node) => { + cost_model.summary_support_evidence(node) + } + Replacement::SubDAG(_) => Some(true), } } } @@ -574,7 +591,7 @@ pub enum ReplacementProvenance { /// A finalized whole-query result over rows carrying the PromQL series /// identity, which the logical root does not expose (see /// [`ReplacementStrategy::propose_for_root`]). Default selection never - /// commits it, because its readout must be validated and priced by + /// commits it, because its evaluation must be validated and priced by /// deployment; otherwise it would silently replace the logical plan. RootPhysicalRealization, } @@ -615,7 +632,7 @@ pub struct Proposals { /// of this trait or any existing strategy required. /// /// `replacements` is only meaningful when `matches` would return `true` for -/// the same target; both [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] +/// the same target; both [`ASAPStrategies`] and [`SharedSubDAGStrategy`] /// return an empty `Vec` rather than panicking when called on a target they /// don't match, so a caller that skips the `matches` check first still gets a /// safe (merely uninformative) answer instead of a crash. @@ -657,7 +674,7 @@ pub trait ReplacementStrategy { /// (for example, the PromQL series identity), so /// [`search_workload_with_targets`] asks only workload roots, once each. /// They decide what to compute, never placement. Default: none. - fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { + fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { Proposals::default() } } @@ -674,41 +691,41 @@ pub trait ReplacementStrategy { /// [`realizations_for_intent`] is where every valid realization gets /// enumerated, exhaustive and ranked (most-preferred first) — this crate has /// no separate function that computes just "the one" `Realization` -/// independently of that list. [`SketchAlgorithmStrategy`] is the sole +/// independently of that list. [`ASAPStrategies`] is the sole /// consumer: it wraps every entry of this list into its own bound -/// [`SummaryNode`] and returns all of them, ranked — a caller wanting a +/// [`OperatorNode`] and returns all of them, ranked — a caller wanting a /// single answer keeps the first one itself (see the module docs above). #[derive(Debug, Clone, PartialEq)] pub enum Realization { /// An exact **mergeable** accumulator (partial state ≡ the value /// itself: `Sum` / `Count` / `Min` / `Max` / `Rate` / `Increase`). The - /// built state *is* the answer already — no `SummaryEstimate` readout + /// built state *is* the answer already — no `SummaryEstimate` evaluation /// step. ExactAggregate { kind: ExactKind, params: ExactParams, }, /// An approximate sketch sized to the intent's [`AccuracyTarget`]. - /// Needs a `SummaryEstimate` readout to recover a value. Already + /// Needs a `SummaryEstimate` evaluation to recover a value. Already /// classified into its [`SketchKind`] category (`SketchKind::new` /// having been called) — construction always goes through that /// classifier, never this variant directly. Sketch(SketchKind), /// A sampling-based summary (a retained row subset). Needs a - /// `SummaryEstimate` readout. Not chosen by any core `AggIntent` + /// `SummaryEstimate` evaluation. Not chosen by any core `AggIntent` /// dispatch today — see the module docs. Sample { kind: SamplingKind, params: SamplingParams, }, - /// A wavelet-transform summary. Needs a `SummaryEstimate` readout. Not + /// A wavelet-transform summary. Needs a `SummaryEstimate` evaluation. Not /// chosen by any core `AggIntent` dispatch today — see the module docs. Wavelet { kind: WaveletKind, params: WaveletParams, }, /// A fitted statistical/parametric-model summary. Needs a - /// `SummaryEstimate` readout. Not chosen by any core `AggIntent` + /// `SummaryEstimate` evaluation. Not chosen by any core `AggIntent` /// dispatch today — see the module docs. StatModel { kind: StatModelKind, @@ -827,7 +844,7 @@ pub fn accuracy_target(intent: &AggIntent) -> Option<&AccuracyTarget> { /// (most-preferred first via `cost_model`) — the *only* place this crate /// decides what an `AggIntent` may become. Nothing in this crate computes /// "the one" `Realization` independently of this list: -/// [`SketchAlgorithmStrategy`] keeps every entry as a candidate, and a caller +/// [`ASAPStrategies`] keeps every entry as a candidate, and a caller /// that wants a single executable answer takes the head of *that* strategy's /// output itself. /// @@ -835,7 +852,7 @@ pub fn accuracy_target(intent: &AggIntent) -> Option<&AccuracyTarget> { /// explicit realization is a compile error, and the coverage-matrix test pins /// each variant's category. /// -/// `pub(crate)`: [`SketchAlgorithmStrategy::replacements`] is this module's +/// `pub(crate)`: [`ASAPStrategies::replacements`] is this module's /// own caller; `grouping::HydraGroupingStrategy` (issue #256) is the one /// caller outside it, needing the exact same already-ranked candidate list /// to find the `Realization::Sketch` matching the Hydra-eligible kind it @@ -1190,9 +1207,9 @@ pub fn posterior_aware_size_params( } } -// ── SketchAlgorithmStrategy ───────────────────────────────────────────────── +// ── ASAPStrategies ───────────────────────────────────────────────── -/// A single static instance so [`SketchAlgorithmStrategy::default_cost_model`] +/// A single static instance so [`ASAPStrategies::default_cost_model`] /// can hand out a `&'static dyn CostModel` without heap-allocating one — /// `DefaultCostModel` is a unit struct with no state, so one instance serves /// every caller. @@ -1227,14 +1244,17 @@ impl<'a> CandidatePlanningInputs<'a> { } } -/// Wraps [`realizations_for_intent`]'s exhaustive, ranked list directly: for -/// a bindable `Aggregate`, every valid candidate summary realization as its -/// own [`ReplacementSubDAG`]. +/// Proposes the supported ASAP realizations for a bindable aggregate, including +/// exact accumulators, approximate sketches, and supported maintained populations. +/// Each valid realization becomes its own [`ReplacementSubDAG`]. +/// +/// [`realizations_for_intent`] enumerates summary families; extension hooks can +/// supply additional supported families. This is not limited to sketch algorithms. /// /// Ranked (only to *order the enumeration*, never to drop a candidate) via a /// [`CostModel`] — [`DefaultCostModel`] unless constructed with -/// [`SketchAlgorithmStrategy::new`] — so a deployment-specific cost model's -/// other hooks (`size_params`, `realize_extension`, `readout_extension`) are +/// [`ASAPStrategies::new`] — so a deployment-specific cost model's +/// other hooks (`size_params`, `realize_extension`, `evaluation_extension`) are /// still consulted while binding each candidate. /// /// The one thing that *does* drop a candidate is accuracy legality (issue @@ -1245,11 +1265,11 @@ impl<'a> CandidatePlanningInputs<'a> { /// [`ReplacementStrategy::propose`] as a [`RejectedCandidate`]. See /// [`crate::accuracy`]'s module docs for the rules and the precedence /// between root and per-node targets. -pub struct SketchAlgorithmStrategy<'a> { +pub struct ASAPStrategies<'a> { planning_inputs: CandidatePlanningInputs<'a>, } -impl SketchAlgorithmStrategy<'static> { +impl ASAPStrategies<'static> { /// A strategy that ranks/binds via the built-in [`DefaultCostModel`] — /// what a deployment gets with no custom cost model plugged in. pub fn default_cost_model() -> Self { @@ -1259,7 +1279,7 @@ impl SketchAlgorithmStrategy<'static> { } } -impl<'a> SketchAlgorithmStrategy<'a> { +impl<'a> ASAPStrategies<'a> { /// A strategy that ranks/binds via `cost_model` instead of the built-in /// static preference order — the same customization point /// [`realizations_for_intent`] already offers. Accuracy legality stays @@ -1314,34 +1334,33 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// treats a range of historical samples as the instant vector. pub fn current_series_topk_candidates( &self, - root: &Rc, + root: &Rc, accuracy: &AccuracyTarget, ) -> Proposals { - let QueryExpr::Limit { - n, + let Some(NonASAPOp::Limit { + n: Some(n), offset: 0, child, - } = root.as_ref() + .. + }) = root.non_asap() else { return Proposals::default(); }; - let QueryExpr::Sort { + let Some(NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + }) = child.non_asap() else { return Proposals::default(); }; let [key] = keys.as_slice() else { return Proposals::default(); }; - let QueryExpr::Column(value) = key.expr else { - return Proposals::default(); - }; - let Ok(schema) = child.output_schema() else { + let ScalarExpr::Column(value) = key.expr else { return Proposals::default(); }; + let schema = &child.schema; if key.ascending || key.nulls_first || partition_by.is_without() @@ -1354,20 +1373,107 @@ impl<'a> SketchAlgorithmStrategy<'a> { { return Proposals::default(); } - let ranked = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(partition_by.clone()), - measures: vec![AggIntent::TopK { - k: *n, - accuracy: accuracy.clone(), - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::clone(child), - }); + let Ok(ranked) = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(partition_by.clone()), + measures: vec![AggIntent::TopK { + k: *n, + accuracy: accuracy.clone(), + }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::clone(child), + })) + else { + return Proposals::default(); + }; self.propose_with(&ranked, None, None) } + /// Fixed-window maintenance can finalize each series' counter state and + /// build a fresh heap or grouped Sum for that evaluation window. Deployment must provide + /// a complete, synchronized population and bind the matching window; this + /// candidate never incrementally adds one window's rates to another. + /// + pub fn fixed_window_rate_candidates(&self, root: &Rc) -> Proposals { + fn place(node: &Rc) -> Option> { + retime_rate_finalize(node, ExecutionTiming::IngestionTime, true) + } + let timed = |node: &Rc| { + asap_types::ir::timing::apply_lifecycle_timings( + node, + &asap_types::ir::timing::LifecycleAssignment::default_maintained(), + &mut asap_types::ir::timing::TimingMemo::new(), + ) + .ok() + .and_then(|timed| { + asap_types::ir::physical_export::compile_physical_asap_dag(&timed).ok() + }) + }; + let mut proposals = self.propose_with(root, None, None); + proposals.candidates.retain_mut(|candidate| { + let Replacement::SubDAG(node) = &candidate.replacement else { + return false; + }; + let Some(dag) = timed(node) else { + return false; + }; + if !dag.nodes.iter().any(|node| match &node.payload { + asap_types::ir::physical_export::PhysicalASAPOperatorPayload::ASAP( + asap_types::ir::ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + }, + ) => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + asap_types::ir::physical_export::PhysicalASAPOperatorPayload::ASAP( + asap_types::ir::ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), + .. + }, + ) => true, + _ => false, + }) { + return false; + } + let Some(placed) = place(node) else { + return false; + }; + if timed(&placed).is_none() { + return false; + } + let Ok(placed) = finalize_query_candidate(placed, root) else { + return false; + }; + candidate.replacement = Replacement::SubDAG(placed); + candidate + .rationale + .push_str("; fixed-window precompute over complete per-series counter states"); + true + }); + proposals + } + + /// Retain grouped Sum after a per-series Rate evaluation as a query-time + /// candidate alongside its complete-window maintenance placement. + /// + pub fn query_time_rate_aggregation_candidates(&self, root: &Rc) -> Proposals { + let mut proposals = self.fixed_window_rate_candidates(root); + proposals.candidates.retain_mut(|candidate| { + let Replacement::SubDAG(node) = &candidate.replacement else { return false }; + if !matches!(&node.operator, Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!(&child.operator, Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. }))) { return false; } + let Some(query_time) = retime_rate_finalize(node, ExecutionTiming::QueryTime, false) else { return false }; + candidate.replacement = Replacement::SubDAG(query_time); + candidate.rationale = "query-time grouped Sum over complete per-series Rate evaluations".into(); + true + }); + proposals + } + pub(crate) fn from_planning_inputs(planning_inputs: CandidatePlanningInputs<'a>) -> Self { Self { planning_inputs } } @@ -1380,20 +1486,21 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// with the sibling that needs it. fn propose_with( &self, - root: &Rc, + root: &Rc, intent_override: Option<&AggIntent>, strictest_sibling: Option<&AccuracyTarget>, ) -> Proposals { let mut proposals = Proposals::default(); - // A selected logical rewrite otherwise remains KeepPreAsap during DAG - // assembly. Also expose its concrete summary realization for selection. + // A selected logical rewrite otherwise stays a kept pre-ASAP sub-DAG + // during DAG assembly. Also expose its concrete summary realization + // for selection. if intent_override.is_none() { if let Some(rewritten) = crate::rewrite::composed_aggregate_rewrite(root) { if let Ok(node) = realize_child_with(&rewritten, self.planning_inputs, None) { - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { + if node.contains_asap() { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "realize a schema-preserving composition of temporal and grouped accumulators".into(), }); @@ -1403,8 +1510,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { } if let Ok(Some(node)) = exact_topk_over_temporal_values(root, self.planning_inputs) { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "select exact Top-K from independently maintained temporal values" .into(), @@ -1413,8 +1520,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { if intent_override.is_none() { if let Ok(Some(node)) = realize_temporal_average(root, self.planning_inputs, None) { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "read temporal average from sum/count only within the finite arithmetic domain; otherwise execute the original average".into(), }); @@ -1428,8 +1535,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { "preserve exact PromQL arithmetic over independently realized summary operands" }; proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: rationale.into(), }); @@ -1516,17 +1623,17 @@ impl<'a> SketchAlgorithmStrategy<'a> { let Some(child) = aggregate_child(root) else { continue; }; - let QueryExpr::Aggregate { reduction, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = root.non_asap() else { continue; }; let Ok(input) = realize_physical_summary_input(intent, &family, reduction, child) else { continue; }; - let readout_query = readout(intent, &input.input, planning_inputs.cost); + let evaluation_query = evaluation(intent, &input.input, planning_inputs.cost); let Some(local) = planning_inputs .accuracy - .local_guarantee(&family, &readout_query) + .local_guarantee(&family, &evaluation_query) else { continue; }; @@ -1537,7 +1644,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { let allocations = planning_inputs.allocator.allocations(target, &shape); if allocations.is_empty() { proposals.rejected.push(RejectedCandidate { - strategy: "SketchAlgorithmStrategy", + strategy: "ASAPStrategies", description: rationale.clone(), error: AccuracyError::NoLegalAllocation { target: target.clone(), @@ -1591,10 +1698,10 @@ impl<'a> SketchAlgorithmStrategy<'a> { } if proposals.candidates.is_empty() { if let Some(error) = &proposals.domain_error { - if let Ok(node) = keep_pre_asap(root) { + if let Ok(node) = retain_exact(root) { proposals.candidates.push(ReplacementSubDAG { - strategy: "SketchAlgorithmStrategy", - replacement: Replacement::Summary(node), + strategy: "ASAPStrategies", + replacement: Replacement::SubDAG(node), provenance: ReplacementProvenance::SummaryRealization, rationale: format!( "{} stays pre-ASAP because summary construction crosses an illegal \ @@ -1613,16 +1720,16 @@ impl Proposals { /// File one construction attempt: a legal node becomes a candidate, an /// [`RealizationError::Accuracy`] becomes a [`RejectedCandidate`], and a /// schema-derivation failure is skipped exactly as it always was. - fn record(&mut self, rationale: String, built: Result, RealizationError>) { + fn record(&mut self, rationale: String, built: Result, RealizationError>) { match built { Ok(node) => self.candidates.push(ReplacementSubDAG { - strategy: "SketchAlgorithmStrategy", - replacement: Replacement::Summary(node), + strategy: "ASAPStrategies", + replacement: Replacement::SubDAG(node), provenance: ReplacementProvenance::SummaryRealization, rationale, }), Err(RealizationError::Accuracy(error)) => self.rejected.push(RejectedCandidate { - strategy: "SketchAlgorithmStrategy", + strategy: "ASAPStrategies", description: rationale, error, }), @@ -1639,14 +1746,14 @@ impl Proposals { } /// The `child` of a [`bindable_intent`]-shaped `Aggregate`. -fn aggregate_child(node: &QueryExpr) -> Option<&Rc> { - match node { - QueryExpr::Aggregate { child, .. } => Some(child), +fn aggregate_child(node: &OperatorNode) -> Option<&Rc> { + match node.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) => Some(child), _ => None, } } -impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { +impl ReplacementStrategy for ASAPStrategies<'_> { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { bindable_intent(target.root).is_some() || is_supported_exact_binary(target.root) } @@ -1665,24 +1772,24 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { /// for the identity-carrying root. Placement variants (for example, /// fixed-window or query-time Rate aggregation) are not listed here: the /// lifecycle assigns timing and the physical compiler reads it. - fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { - let Ok(typed) = asap_types::pre_asap::schema::with_promql_series_identity(root) else { + fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { + let Ok(typed) = asap_types::ir::schema_support::with_promql_series_identity(root) else { return Proposals::default(); }; - let typed = Rc::new(typed); + let mut proposals = self.current_series_topk_candidates(&typed, target); for mut candidate in std::mem::take(&mut proposals.candidates) { - let Replacement::Summary(node) = candidate.replacement else { + let Replacement::SubDAG(node) = candidate.replacement else { continue; }; let Ok(node) = finalize_query_candidate(node, &typed) else { continue; }; let duplicate = proposals.candidates.iter().any(|existing| { - matches!(&existing.replacement, Replacement::Summary(other) if *other == node) + matches!(&existing.replacement, Replacement::SubDAG(other) if *other == node) }); if !duplicate { - candidate.replacement = Replacement::Summary(node); + candidate.replacement = Replacement::SubDAG(node); candidate.provenance = ReplacementProvenance::RootPhysicalRealization; proposals.candidates.push(candidate); } @@ -1755,11 +1862,11 @@ pub(crate) fn describe_intent(intent: &AggIntent) -> String { } } -// ── realize_child / keep_pre_asap: rank-and-take-first, and its fallback ── +// ── realize_child / retain_exact: rank-and-take-first, and its fallback ── -/// Rank-and-take-first selector for a single [`QueryExpr`] node: enumerate -/// every candidate via [`SketchAlgorithmStrategy::replacements`], keep the -/// `cost_model`-preferred (first) one, and fall back to [`keep_pre_asap`] +/// Rank-and-take-first selector for a single [`OperatorNode`]: enumerate +/// every candidate via [`ASAPStrategies::replacements`], keep the +/// `cost_model`-preferred (first) one, and fall back to [`retain_exact`] /// when there's no candidate at all — **not** a general single-answer API /// for a whole workload. Use [`CandidateLogicalASAPDAGs::global_selection`] and DAG assembly /// for coordinated logical selection; physical deployment remains downstream. @@ -1771,16 +1878,16 @@ pub(crate) fn describe_intent(intent: &AggIntent) -> String { /// ([`construct_summary_agg`], so a nested aggregate gets its own /// independent enumeration instead of inheriting the parent's forced /// candidate), from this module's own [`realize_one`] (the representative -/// bound `SummaryNode` [`cse_preference`] needs for a +/// bound `OperatorNode` [`cse_preference`] needs for a /// [`CostModel::cse_share_decision`] comparison), and from /// [`crate::cost_model::DefaultCostModel::estimate_cost`] (the same /// representative-node need, for a [`Replacement::Rewrite`] candidate's own /// cost estimate). Every other caller goes through -/// [`SketchAlgorithmStrategy::replacements`] directly and decides for itself. +/// [`ASAPStrategies::replacements`] directly and decides for itself. pub(crate) fn realize_child( - root: &Rc, + root: &Rc, cost_model: &dyn CostModel, -) -> Result, RealizationError> { +) -> Result, RealizationError> { realize_child_with( root, CandidatePlanningInputs::with_default_accuracy(cost_model), @@ -1797,17 +1904,17 @@ pub(crate) fn realize_child( /// budget. A child whose declared target is `Exact` keeps it: an allocation /// never approximates something the caller declared exact. fn exact_topk_over_temporal_values( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, -) -> Result>, RealizationError> { - let QueryExpr::Aggregate { +) -> Result>, RealizationError> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names: _, filters, having: None, child, - } = root.as_ref() + }) = root.non_asap() else { return Ok(None); }; @@ -1817,19 +1924,19 @@ fn exact_topk_over_temporal_values( let [AggIntent::TopK { k, .. }] = measures.as_slice() else { return Ok(None); }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, child: input, .. - } = child.as_ref() + }) = child.non_asap() else { return Ok(None); }; - if !matches!(input.as_ref(), QueryExpr::TimeRange { .. }) { + if !matches!(input.non_asap(), Some(NonASAPOp::TimeRange { .. })) { return Ok(None); } let values = realize_child_with(child, planning_inputs, Some(&AccuracyTarget::Exact))?; - if matches!(values.expr, SummaryExpr::KeepPreAsap(_)) + if !values.contains_asap() || !values .guarantee .as_ref() @@ -1845,61 +1952,63 @@ fn exact_topk_over_temporal_values( ))? .clone(); let score = ranking_score_index(child, &values.schema)?; - let sorted = Rc::new(SummaryNode { - guarantee: values.guarantee.clone(), - schema: values.schema.clone(), - expr: SummaryExpr::ValueOperation { - child: values, - operation: ValueOperation::Sort { - keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + let guarantee = values.guarantee.clone(); + let schema = values.schema.clone(); + let sorted = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Sort { + keys: vec![SortKey { + expr: ScalarExpr::Column(score), ascending: false, nulls_first: false, }], partition_by: partition_by.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - }); - let node = Rc::new(SummaryNode { - guarantee: sorted.guarantee.clone(), - schema: sorted.schema.clone(), - expr: SummaryExpr::ValueOperation { - child: sorted, - operation: ValueOperation::Limit { - n: *k, + child: values, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let node = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Limit { + n: Some(*k), offset: 0, partition_by, - }, - timing: ExecutionTiming::QueryTime, - }, - }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + child: sorted, + }), + schema, + ) + .with_guarantee(guarantee), + ); + validate_default(&node, ExecutionTiming::QueryTime)?; Ok(Some(node)) } fn realize_temporal_average( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { +) -> Result>, RealizationError> { let Some(components) = crate::rewrite::temporal_average_components(root) else { return Ok(None); }; let mut node = realize_child_with(&components, planning_inputs, target)?; - let SummaryExpr::BinaryOp { operator, .. } = &mut Rc::make_mut(&mut node).expr else { + let Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) = + &mut Rc::make_mut(&mut node).operator + else { return Ok(None); }; operator.checked_finite_division = true; - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + validate_default(&node, ExecutionTiming::QueryTime)?; Ok(Some(node)) } pub(crate) fn realize_child_with( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { +) -> Result, RealizationError> { if let Some(node) = realize_temporal_average(root, planning_inputs, end_to_end_target)? { return Ok(node); } @@ -1913,29 +2022,29 @@ pub(crate) fn realize_child_with( Some(_) => Some(override_accuracy(declared, target)), } }); - match SketchAlgorithmStrategy::from_planning_inputs(planning_inputs) + match ASAPStrategies::from_planning_inputs(planning_inputs) .propose_with(root, overridden.as_ref(), None) .candidates .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. }) => Ok(node), Some(ReplacementSubDAG { - replacement: Replacement::Rewrite(_) | Replacement::ExactComposition(_), + replacement: Replacement::ExactComposition(_), .. }) => { - unreachable!("SketchAlgorithmStrategy never returns a Rewrite/composition candidate") + unreachable!("ASAPStrategies never returns a composition candidate") } // No candidate at all: `root` isn't `bindable_intent` shape (or its // intent has no realization `realizations_for_intent` can't // produce — never happens, that match is exhaustive), or every // candidate was accuracy-illegal — either way the same conservative - // fallback `SketchAlgorithmStrategy::matches` uses: keep the + // fallback `ASAPStrategies::matches` uses: keep the // pre-ASAP sub-DAG, executed exactly. - None => keep_pre_asap(root), + None => retain_exact(root), } } @@ -1944,29 +2053,24 @@ pub(crate) fn realize_child_with( /// accelerated, return `None` so the caller keeps the whole query exact; /// mixed raw/summary snapshots are never constructed. fn realize_binary( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { - let QueryExpr::BinaryOp { - op, +) -> Result>, RealizationError> { + let Some(NonASAPOp::BinaryOp { + operator, + return_bool, lhs, rhs, - vector_match, - } = root.as_ref() + }) = root.non_asap() else { return Ok(None); }; + let (op, vector_match) = (&operator.kind, &operator.vector_match); if !matches!(op, BinaryOpKind::Arithmetic(_)) || vector_match.is_some() { return Ok(None); } - let lhs_scalar = is_promql_scalar(lhs); - let rhs_scalar = is_promql_scalar(rhs); - if lhs_scalar && rhs_scalar { - return Ok(None); - } - let mut lhs_node = realize_binary_operand(lhs, planning_inputs, None)?; let mut rhs_node = realize_binary_operand(rhs, planning_inputs, None)?; @@ -2085,8 +2189,8 @@ fn realize_binary( return Ok(None); } - let lhs_accelerated = lhs_scalar || !matches!(lhs_node.expr, SummaryExpr::KeepPreAsap(_)); - let rhs_accelerated = rhs_scalar || !matches!(rhs_node.expr, SummaryExpr::KeepPreAsap(_)); + let lhs_accelerated = lhs_node.contains_asap(); + let rhs_accelerated = rhs_node.contains_asap(); if !lhs_accelerated || !rhs_accelerated { return Ok(None); } @@ -2133,47 +2237,106 @@ fn realize_binary( return Ok(None); } - Ok(Some(Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: ExecutionTiming::QueryTime, - lhs: lhs_node, - rhs: rhs_node, - operator: asap_types::post_asap::BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: op.clone(), - vector_match: vector_match.clone(), - }, - }, - schema: lift(&root.output_schema()?), + Ok(Some(Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: op.clone(), + vector_match: vector_match.clone(), + }, + return_bool: *return_bool, + lhs: lhs_node, + rhs: rhs_node, + }), + root.schema.clone(), + ) // Exact arithmetic does not erase approximation error. Until the // accuracy algebra has an operator-specific rule (and any value-range // evidence needed by multiplication/division), unknown stays unknown. - guarantee, - }))) + .with_guarantee(guarantee), + ))) +} + +/// Rebuild the summary chain above a per-series `Rate` accumulator with its +/// `FinalizeExactAccumulator` placed at `timing`. `strict` additionally +/// requires the fixed-window shape (a `PerEntity` Rate over a `TimeRange`); +/// `None` when no such boundary exists (strict only). +fn retime_rate_finalize( + node: &Rc, + timing: ExecutionTiming, + strict: bool, +) -> Option> { + let is_rate_boundary = |child: &OperatorNode| match &child.operator { + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), + reduction, + child: source, + .. + }) => { + !strict + || (matches!(reduction, Reduction::PerEntity) + && matches!(source.non_asap(), Some(NonASAPOp::TimeRange { .. }))) + } + _ => false, + }; + let rebuilt = |operator: Operator, timing: Option| { + let mut copy = node.as_ref().clone(); + copy.operator = operator; + copy.timing = timing; + Rc::new(copy) + }; + match &node.operator { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_rate_boundary(child) => { + Some(rebuilt(node.operator.clone(), Some(timing))) + } + Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { child } + | ASAPOp::SummaryAgg { child, .. } + | ASAPOp::SummaryEstimate { + summary_input: child, + .. + }, + ) => { + let placed = match retime_rate_finalize(child, timing, strict) { + Some(placed) => placed, + None if strict => return None, + None => return Some(Rc::clone(node)), + }; + let operator = node.operator.map_children(|_| Rc::clone(&placed)); + Some(rebuilt(operator, node.timing)) + } + _ if strict => None, + _ => Some(Rc::clone(node)), + } } /// Put an explicit read boundary between maintained exact state and a /// query-time value consumer. Approximate summaries must already carry a /// `SummaryEstimate`, so they deliberately do not pass this predicate. pub fn finalize_query_candidate( - node: Rc, - logical_output: &QueryExpr, -) -> Result, RealizationError> { - finalize_exact_accumulator_at(node, logical_output, ExecutionTiming::QueryTime) -} - -fn finalize_exact_accumulator_at( - node: Rc, - logical_output: &QueryExpr, - timing: ExecutionTiming, -) -> Result, RealizationError> { + node: Rc, + logical_output: &OperatorNode, +) -> Result, RealizationError> { + finalize_exact_accumulator(node, logical_output, ExecutionTiming::QueryTime) +} + +/// The read boundary's placement is fixed here, where the candidate's +/// semantics decide it (a fresh query-time summary over this evaluation's +/// finalized values vs. finalized values feeding maintenance); the lifecycle +/// timing pass honors it. +fn finalize_exact_accumulator( + node: Rc, + logical_output: &OperatorNode, + placement: ExecutionTiming, +) -> Result, RealizationError> { let is_exact_state = matches!( - node.expr, - SummaryExpr::SummaryAgg { + node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(..), .. - } + }) ); if !is_exact_state { return Ok(node); @@ -2182,41 +2345,36 @@ fn finalize_exact_accumulator_at( // boundary produces the logical operator's ordinary values. Preserve the // canonical pre-ASAP output types instead of leaking ExactAggregate into // query-time operators that follow this node. - let schema = lift(&logical_output.output_schema()?); + let schema = logical_output.schema.clone(); let guarantee = node.guarantee.clone(); - Ok(Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: node, - operation: ValueOperation::FinalizeExactAccumulator, - timing, - }, - schema, - guarantee, - })) + Ok(Rc::new( + OperatorNode::with_schema( + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: node }), + schema, + ) + .with_guarantee(guarantee) + .with_timing(Some(placement)), + )) } -fn is_supported_exact_binary(root: &QueryExpr) -> bool { +fn is_supported_exact_binary(root: &OperatorNode) -> bool { matches!( - root, - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(_), - vector_match: None, + root.non_asap(), + Some(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(_), + vector_match: None, + .. + }, .. - } - ) -} - -fn is_promql_scalar(expr: &QueryExpr) -> bool { - matches!( - expr, - QueryExpr::PromqlScalarBridge(_) | QueryExpr::Literal(_) + }) ) } /// Quantile operands inherit one workload target. A temporal mean is exact /// on its checked finite domain and needs no approximation budget. -fn shared_quantile_target(lhs: &QueryExpr, rhs: &QueryExpr) -> Option { - let quantile_target = |expr: &QueryExpr| match bindable_intent(expr) { +fn shared_quantile_target(lhs: &OperatorNode, rhs: &OperatorNode) -> Option { + let quantile_target = |expr: &OperatorNode| match bindable_intent(expr) { Some(AggIntent::Quantile { accuracy, q, .. }) if q.is_finite() && (0.0..=1.0).contains(q) => { @@ -2254,18 +2412,18 @@ fn ddsketch_ratio_operand_target(target: &AccuracyTarget) -> Option Option { - let SummaryExpr::SummaryEstimate { +fn ddsketch_quantile_alpha(node: &OperatorNode) -> Option { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query: PostAsapSketchStatistic::Quantile { .. }, - } = &node.expr + }) = &node.operator else { return None; }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = &summary_input.operator else { return None; }; @@ -2275,7 +2433,7 @@ fn ddsketch_quantile_alpha(node: &SummaryNode) -> Option { } } -fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { +fn has_missing_accuracy_evidence(node: &OperatorNode) -> bool { node.guarantee .as_ref() .is_none_or(ResultGuarantee::has_unknown) @@ -2284,10 +2442,10 @@ fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { /// A direct ratio has an operator-specific DDSketch proof, so it must select /// DDSketch rather than the cost model's generally preferred KLL candidate. fn realize_ddsketch_quantile_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: &AccuracyTarget, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let intent = bindable_intent(operand).and_then(|intent| match intent { AggIntent::Quantile { .. } => Some(override_accuracy(intent, target)), _ => None, @@ -2306,17 +2464,10 @@ fn realize_ddsketch_quantile_operand( } fn realize_binary_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { - if is_promql_scalar(operand) { - return Ok(Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::clone(operand)), - schema: Schema::lifted(Vec::new(), None), - guarantee: Some(ResultGuarantee::exact("PromQL scalar")), - })); - } +) -> Result, RealizationError> { realize_child_with(operand, planning_inputs, end_to_end_target) } @@ -2334,43 +2485,77 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { out } -/// Wrap an unrewritten pre-ASAP sub-DAG, lifting its schema with every column -/// `FieldDataType::Plain`. `pub` so a caller can fall back to this -/// explicitly — e.g. when `SketchAlgorithmStrategy::replacements()` returns no -/// candidate for a target, or a deployment wants to force a node its own -/// runtime can't actually implement — through the same fallback this -/// crate's own dispatch uses, without duplicating the schema-lift logic. -pub fn keep_pre_asap(expr: &Rc) -> Result, RealizationError> { - keep_pre_asap_rc(Rc::clone(expr)) -} - -fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationError> { - let schema = expr.output_schema()?; - Ok(Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(expr), - schema: lift(&schema), - // A kept pre-ASAP sub-DAG is executed exactly by the runtime - // (`Realization::PassThrough`'s contract) — zero error. - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), - })) +/// Keep an unrewritten pre-ASAP sub-DAG as it is. There is no wrapper node: +/// the sub-DAG itself is the plan, carrying an exact guarantee. The same +/// `Rc` is returned when the node already has a guarantee; otherwise a copy +/// with `guarantee = exact("RetainedExact")` — only for a sub-DAG with no +/// ASAP operator (a sub-DAG containing one keeps whatever its construction +/// established). `pub` so a caller can fall back to this explicitly — e.g. +/// when `ASAPStrategies::replacements()` returns no candidate for a +/// target, or a deployment wants to force a node its own runtime can't +/// actually implement — through the same fallback this crate's own dispatch +/// uses. +pub fn retain_exact(expr: &Rc) -> Result, RealizationError> { + retain_exact_rc(Rc::clone(expr)) +} + +fn retain_exact_rc(expr: Rc) -> Result, RealizationError> { + if expr.guarantee.is_some() || expr.contains_asap() { + return Ok(expr); + } + // Keeping the same sub-DAG twice (e.g. one `Scan` read by an exact + // aggregate and by a sketch, or by two candidates) must yield one node: + // sharing is pointer identity. Memoize the kept copy per input node while + // both are alive; weak references keep the memo from extending lifetimes + // or matching a reused address. + type KeptMemo = HashMap<*const OperatorNode, (Weak, Weak)>; + thread_local! { + static KEPT: RefCell = RefCell::new(HashMap::new()); + } + let key = Rc::as_ptr(&expr); + if let Some(kept) = KEPT.with(|memo| { + memo.borrow().get(&key).and_then(|(input, kept)| { + input + .upgrade() + .filter(|input| Rc::ptr_eq(input, &expr)) + .and_then(|_| kept.upgrade()) + }) + }) { + return Ok(kept); + } + let kept = Rc::new( + expr.as_ref() + .clone() + // A kept pre-ASAP sub-DAG is executed exactly by the runtime + // (`Realization::PassThrough`'s contract) — zero error. + .with_guarantee(Some(ResultGuarantee::exact("RetainedExact"))), + ); + KEPT.with(|memo| { + let mut memo = memo.borrow_mut(); + if memo.len() > 4096 { + memo.retain(|_, (input, kept)| input.strong_count() > 0 && kept.strong_count() > 0); + } + memo.insert(key, (Rc::downgrade(&expr), Rc::downgrade(&kept))); + }); + Ok(kept) } -// ── Construction: turn one already-decided Realization into a SummaryNode ─ +// ── Construction: turn one already-decided Realization into an OperatorNode ─ -/// The bindable shape [`SketchAlgorithmStrategy`] targets: a single intent, no +/// The bindable shape [`ASAPStrategies`] targets: a single intent, no /// `HAVING`. A multi-intent node (SQL `SELECT SUM(a), AVG(b)`), or one with a /// `HAVING` predicate (the filter would need the estimate first), stays -/// logical. Unsupported logical parents still conservatively become one -/// [`SummaryExpr::KeepPreAsap`] sub-DAG. Composable query-time value -/// operators (`Project`, `Filter`, `Sort`, and `Limit`) are retained during final -/// DAG assembly so their independently planned children remain visible. -pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { - if let QueryExpr::Aggregate { +/// logical. Unsupported logical parents are conservatively kept as pre-ASAP +/// sub-DAGs ([`retain_exact`]). Relational operators are retained during +/// final DAG assembly so their independently planned children remain +/// visible. +pub fn bindable_intent(node: &OperatorNode) -> Option<&AggIntent> { + if let Some(NonASAPOp::Aggregate { measures, filters, having, .. - } = node + }) = node.non_asap() { if let ([intent], None) = (measures.as_slice(), having) { if !any_measure_filtered(filters) { @@ -2382,7 +2567,7 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { } /// `expr` must still be the [`bindable_intent`] shape for `realization` to -/// have any effect; anything else falls back to [`keep_pre_asap`]. +/// have any effect; anything else falls back to [`retain_exact`]. /// Only `expr`'s own top-level decision is forced — recursion into `expr`'s /// child goes back through [`realize_child`] (fresh candidate /// enumeration, not a forced pick), so choosing one candidate for a target @@ -2390,10 +2575,10 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// /// `pub(crate)`: `grouping::HydraGroupingStrategy` (issue #256) is the one /// caller outside this module — the same first-class, -/// one-candidate-at-a-time primitive [`SketchAlgorithmStrategy`] itself +/// one-candidate-at-a-time primitive [`ASAPStrategies`] itself /// calls once per candidate, reused rather than duplicated so a Hydra /// candidate gets exactly the same schema derivation/column -/// resolution/readout construction as every other candidate, patching only +/// resolution/evaluation construction as every other candidate, patching only /// the `grouping` field this axis owns. /// Construct a summary with every model explicit (issue #172). `intent` /// is `expr`'s own [`bindable_intent`], or a copy of it with an allocated @@ -2404,13 +2589,13 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// fail-closed answer for a composition with no sound rule or one that /// misses `intent`'s target. pub(crate) fn construct_summary_with( - expr: &QueryExpr, + expr: &OperatorNode, intent: &AggIntent, realization: Realization, planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let local_target = match allocation.as_ref() { Some(GuaranteeSource::BudgetAllocation { local_target, .. }) => Some(local_target), _ => accuracy_target(intent), @@ -2427,9 +2612,9 @@ pub(crate) fn construct_summary_with( }, other => other, }; - if let QueryExpr::Aggregate { + if let Some(NonASAPOp::Aggregate { reduction, child, .. - } = expr + }) = expr.non_asap() { // `bindable_intent` already established the shape: exactly one // intent, no HAVING. (Multi-intent nodes and HAVING stay logical.) @@ -2454,28 +2639,28 @@ pub(crate) fn construct_summary_with( } } } - keep_pre_asap_rc(Rc::new(expr.clone())) + retain_exact_rc(Rc::new(expr.clone())) } fn finish_weighted_topk( - candidate: Rc, - logical: &QueryExpr, + candidate: Rc, + logical: &OperatorNode, intent: &AggIntent, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let AggIntent::TopK { k, .. } = intent else { unreachable!() }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(groups), child, .. - } = logical + }) = logical.non_asap() else { return Err(RealizationError::PhysicalRealization( "TopK requires explicit grouping", )); }; - let schema = lift(&child.output_schema()?); + let schema = child.schema.clone(); let score = ranking_score_index(child, &schema)?; let cols = schema .fields @@ -2502,84 +2687,81 @@ fn finish_weighted_topk( } } }; - Ok(asap_types::pre_asap::query_expr::ProjectItem { + Ok(ProjectItem { alias: Some(field.name.clone()), - expr: QueryExpr::Column(source), + expr: ScalarExpr::Column(source), }) }) .collect::, _>>()?; let guarantee = candidate.guarantee.clone(); - let projected = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: candidate, - operation: ValueOperation::Project { + let projected = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Project { cols, qualifier: None, - }, - timing: ExecutionTiming::QueryTime, - }, - schema: schema.clone(), - guarantee: guarantee.clone(), - }); - let sorted = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: projected, - operation: ValueOperation::Sort { - keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + child: candidate, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let sorted = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Sort { + keys: vec![SortKey { + expr: ScalarExpr::Column(score), ascending: false, nulls_first: false, }], partition_by: groups.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema: schema.clone(), - guarantee: guarantee.clone(), - }); - let result = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: sorted, - operation: ValueOperation::Limit { - n: *k, + child: projected, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let result = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Limit { + n: Some(*k), offset: 0, partition_by: groups.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema, - guarantee, - }); - validate_execution_data_states_at(&result, ExecutionDataState::QUERY_ROWS)?; + child: sorted, + }), + schema, + ) + .with_guarantee(guarantee), + ); + validate_default(&result, ExecutionTiming::QueryTime)?; Ok(result) } -fn is_current_series_source(child: &QueryExpr) -> bool { - let source = match child { - QueryExpr::TimeRange { child, .. } => child.as_ref(), - source => source, +fn is_current_series_source(child: &OperatorNode) -> bool { + let source = match child.non_asap() { + Some(NonASAPOp::TimeRange { child, .. }) => child.as_ref(), + _ => child, }; - matches!(source, QueryExpr::Scan { + matches!(source.non_asap(), Some(NonASAPOp::Scan { source: asap_types::pre_asap::Source::TimeSeries { .. }, schema, .. - } if schema.has_promql_series_identity()) + }) if schema.has_promql_series_identity()) } -fn is_snapshot_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { +fn is_snapshot_weighted_topk(intent: &AggIntent, child: &OperatorNode) -> bool { matches!(intent, AggIntent::TopK { .. }) && (is_current_series_source(child) - || matches!(child, - QueryExpr::Aggregate { measures, child, .. } + || matches!(child.non_asap(), + Some(NonASAPOp::Aggregate { measures, child, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]) || (matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(child.non_asap(), Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]))))) } /// Translate an [`Realization`] into the `(family, needs a -/// SummaryEstimate readout)` pair [`construct_summary_agg`] needs, or `None` -/// for `PassThrough` (the caller falls back to [`keep_pre_asap`]). +/// SummaryEstimate evaluation)` pair [`construct_summary_agg`] needs, or `None` +/// for `PassThrough` (the caller falls back to [`retain_exact`]). /// -/// Every family's partial state needs a readout to recover a value, except +/// Every family's partial state needs a evaluation to recover a value, except /// `ExactAggregate` — its partial state *is* the value already, so no /// estimate step follows it. fn summary_family(realization: Realization) -> Option<(FieldDataType, bool)> { @@ -2603,7 +2785,7 @@ fn summary_family(realization: Realization) -> Option<(FieldDataType, bool)> { /// input value. Composite realizations can instead consume a larger /// logical sub-DAG and bind a different key or value. struct PhysicalSummaryInput { - child: Rc, + child: Rc, input: SummaryUpdate, } @@ -2614,7 +2796,7 @@ enum PhysicalSummaryInputRuleResult { } type PhysicalSummaryInputRule = - fn(&AggIntent, &FieldDataType, &Reduction, &Rc) -> PhysicalSummaryInputRuleResult; + fn(&AggIntent, &FieldDataType, &Reduction, &Rc) -> PhysicalSummaryInputRuleResult; /// Ordered physical-realization rules for realizations that consume more /// than the immediate logical input. New composite primitives add a rule here @@ -2631,7 +2813,7 @@ fn realize_value_frequency_summary_input( intent: &AggIntent, family: &FieldDataType, _reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { // Frequency counts hash sample values as items but add one per observation. // Using the sample as a weight would turn counts into sums and admit signed CMS updates. @@ -2642,11 +2824,7 @@ fn realize_value_frequency_summary_input( { return PhysicalSummaryInputRuleResult::NotApplicable; } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "value frequency input needs a valid schema", - ); - }; + let schema = &child.schema; // One item per observation is a single value stream. `summary_candidates` // already withholds UnivMon from a distinct-tuple count; refused here too // so the invariant does not rest on that table alone. @@ -2658,7 +2836,7 @@ fn realize_value_frequency_summary_input( PhysicalSummaryInputRuleResult::Realized(PhysicalSummaryInput { child: Rc::clone(child), input: SummaryUpdate { - item: Some(SummaryInputExpr::Column(summarised_column(intent, &schema))), + item: Some(SummaryInputExpr::Column(summarised_column(intent, schema))), weight: SummaryInputExpr::Constant(1.0), weight_domain: WeightDomain::NonNegative { proof: NonNegativeWeightProof::UnitCount, @@ -2671,7 +2849,7 @@ fn realize_physical_summary_input( intent: &AggIntent, family: &FieldDataType, reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> Result { for rule in PHYSICAL_SUMMARY_INPUT_RULES { match rule(intent, family, reduction, child) { @@ -2683,7 +2861,7 @@ fn realize_physical_summary_input( } } - let child_schema = child.output_schema()?; + let child_schema = &child.schema; if matches!(intent, AggIntent::TopK { .. }) { return Err(RealizationError::PhysicalRealization( "Top-K needs an explicit item identity and additive update input", @@ -2693,76 +2871,70 @@ fn realize_physical_summary_input( child: Rc::clone(child), input: SummaryUpdate { item: None, - weight: summarised_input(intent, &child_schema)?, + weight: summarised_input(intent, child_schema)?, weight_domain: WeightDomain::UnknownOrSigned, }, }) } /// Emit `SummaryAgg` (recursively binding the child), plus the -/// `SummaryEstimate` readout when `estimate` is set. +/// `SummaryEstimate` evaluation when `estimate` is set. // Retain the exact expression and schema while placing its value production -// on the update path. This is the initial layout for values feeding a summary; -// lifecycle timing is authoritative. Read-time consumers keep their original -// shared nodes. -fn maintenance_exact_values(node: Rc) -> Option> { - let expr = match &node.expr { +// on the update path (a node runs when its consumer runs, so beneath a +// maintained summary this value production is ingestion-time work). +// Read-time consumers keep their original shared nodes. +fn maintenance_exact_values(node: Rc) -> Option> { + let operator = match &node.operator { // These guards can fall back at read time, but cannot recover a parent // sketch after an invalid value has entered its maintained state. - SummaryExpr::BinaryOp { operator, .. } + Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) if operator.checked_finite_division || operator.checked_relative_division => { return None; } - SummaryExpr::BinaryOp { - lhs, rhs, operator, .. - } if operator.vector_match.is_none() - && matches!( - operator.kind, - asap_types::pre_asap::BinaryOpKind::Arithmetic(_) - ) + Operator::NonASAP(NonASAPOp::BinaryOp { + lhs, + rhs, + operator, + return_bool, + }) if operator.vector_match.is_none() + && matches!(operator.kind, BinaryOpKind::Arithmetic(_)) && node .guarantee .as_ref() .is_some_and(ResultGuarantee::is_exact) => { - SummaryExpr::BinaryOp { + Operator::NonASAP(NonASAPOp::BinaryOp { lhs: maintenance_exact_values(lhs.clone())?, rhs: maintenance_exact_values(rhs.clone())?, operator: operator.clone(), - timing: ExecutionTiming::IngestionTime, - } + return_bool: *return_bool, + }) } - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } if matches!( - child.expr, - SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(..), - .. - } - ) => + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(..), + .. + }) + ) => { - SummaryExpr::ValueOperation { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: child.clone(), - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } + }) } _ => return Some(node), }; - Some(Rc::new(SummaryNode { - expr, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - })) + Some(Rc::new( + OperatorNode::with_schema(operator, node.schema.clone()) + .with_guarantee(node.guarantee.clone()), + )) } #[allow(clippy::too_many_arguments)] fn construct_summary_agg( - node: &QueryExpr, + node: &OperatorNode, reduction: &Reduction, intent: &AggIntent, input: PhysicalSummaryInput, @@ -2771,7 +2943,7 @@ fn construct_summary_agg( planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { // The single canonical pre-ASAP derivation (per-series vs cross-series, // name overrides) already computes the row shape; binding only retypes // the summary state column. @@ -2781,7 +2953,7 @@ fn construct_summary_agg( FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap) ); - let snapshot_weighted = matches!(node, QueryExpr::Aggregate { child, .. } + let snapshot_weighted = matches!(node.non_asap(), Some(NonASAPOp::Aggregate { child, .. }) if is_snapshot_weighted_topk(intent, child)); let mut family = family; let score_population = if snapshot_weighted { @@ -2809,10 +2981,10 @@ fn construct_summary_agg( None }; let physical_reduction = if snapshot_weighted { - let QueryExpr::Aggregate { child, .. } = node else { + let Some(NonASAPOp::Aggregate { child, .. }) = node.non_asap() else { unreachable!() }; - let source = input.child.output_schema()?; + let source = &input.child.schema; let Reduction::Reduce(keys) = reduction else { return Err(RealizationError::PhysicalRealization( "TopK requires explicit partitions", @@ -2850,19 +3022,19 @@ fn construct_summary_agg( } else { reduction.clone() }; - let out_schema = node.output_schema()?; - let measures = match node { - QueryExpr::Aggregate { measures, .. } => measures.len(), + let out_schema = &node.schema; + let measures = match node.non_asap() { + Some(NonASAPOp::Aggregate { measures, .. }) => measures.len(), _ => 1, }; - let state_idx = summary_col_index(&out_schema, reduction, measures); + let state_idx = summary_col_index(out_schema, reduction, measures); - let readout_schema = if keyed_heap - && matches!(node, QueryExpr::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)) + let evaluation_schema = if keyed_heap + && matches!(node.non_asap(), Some(NonASAPOp::Aggregate { child, .. }) if is_snapshot_weighted_topk(intent, child)) { - keyed_heap_readout_schema(&input, node)? + keyed_heap_evaluation_schema(&input, node)? } else { - lift(&out_schema) + out_schema.clone() }; let summary_input = input.input; @@ -2879,15 +3051,15 @@ fn construct_summary_agg( }; } } - readout(intent, &summary_input, planning_inputs.cost) + evaluation(intent, &summary_input, planning_inputs.cost) }); - let mut state_schema = lift(&out_schema); + let mut state_schema = out_schema.clone(); if keyed_heap { let mut state = state_schema.fields[state_idx].clone(); state.dtype = family.clone(); let mut fields = if snapshot_weighted { - readout_schema.fields[..reduction.group_keys().map_or(0, |keys| keys.len())].to_vec() + evaluation_schema.fields[..reduction.group_keys().map_or(0, |keys| keys.len())].to_vec() } else { Vec::new() }; @@ -2895,6 +3067,7 @@ fn construct_summary_agg( state_schema = Schema::lifted(fields, None); } else if let Some(field) = state_schema.fields.get_mut(state_idx) { field.dtype = family.clone(); + field.nullable = false; if matches!(&family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) { // State identity is independent of which statistic reads it. @@ -2902,9 +3075,9 @@ fn construct_summary_agg( } else if let (AggIntent::Quantile { .. }, SummaryInputExpr::Column(col)) = (intent, &summary_input.weight) { - // The quantile is a readout parameter: name the state after the + // The quantile is a evaluation parameter: name the state after the // column it summarizes, not after the query's output column. - let child_schema = input.child.output_schema()?; + let child_schema = input.child.schema.clone(); if let Ok(i) = resolve_column_ref(col, &child_schema) { field.name = child_schema.fields[i].name.clone(); } @@ -2931,9 +3104,9 @@ fn construct_summary_agg( .ok_or(RealizationError::PhysicalRealization( "snapshot ranking requires a supported current-series population", ))?; - let SummaryExpr::ValueOperation { child, .. } = &population.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &population.operator else { return Err(RealizationError::PhysicalRealization( - "missing population readout", + "missing population evaluation", )); }; Rc::clone(child) @@ -2943,27 +3116,14 @@ fn construct_summary_agg( // initial layout; a retained summary's lifecycle moves it to ingestion. finalize_query_candidate(bound_child, &input.child)? } else { - let child = finalize_exact_accumulator_at( - bound_child, - &input.child, - ExecutionTiming::IngestionTime, - )?; - let child = maintenance_exact_values(child).unwrap_or(keep_pre_asap(&input.child)?); - // Maintenance arithmetic must satisfy the ingestion contract; e.g. a - // per-series sum over different selectors has no exact aligned - // layout, so this candidate fails closed and exact execution remains. - // Unlike checked division, it does not fall back to `keep_pre_asap`: - // that retains the range expression at ingestion time, where range - // functions cannot run (they need a query evaluation time). - if matches!(child.expr, SummaryExpr::BinaryOp { .. }) { - validate_execution_data_states_at(&child, ExecutionDataState::INGESTION_ROWS)?; - } - child + let child = + finalize_exact_accumulator(bound_child, &input.child, ExecutionTiming::IngestionTime)?; + maintenance_exact_values(child).unwrap_or(retain_exact(&input.child)?) }; // ── Guarantee (issue #172) ────────────────────────────────────────── // Derived *before* the node exists, so an illegal composition is never - // materialized: the local guarantee of this family's readout (or exact + // materialized: the local guarantee of this family's evaluation (or exact // accumulator) composed over the child's, under the operator this // family applies to the child's values. let local_target = match allocation.as_ref() { @@ -2976,7 +3136,7 @@ fn construct_summary_agg( local_target, ); let membership_query = if snapshot_weighted { - Some(readout(intent, &summary_input, planning_inputs.cost)) + Some(evaluation(intent, &summary_input, planning_inputs.cost)) } else { query.clone() }; @@ -3054,56 +3214,61 @@ fn construct_summary_agg( // a genuine empty-`by` reduction apart from a per-entity shape with no // grouping concept at all (issue #163). `construct_summary_agg` is the // single place that decides this; nothing downstream re-derives it. - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let agg = OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { child: bound_child, family, input: summary_input, reduction: physical_reduction, grouping: GroupingStrategy::default(), filter: None, - }, - schema: state_schema, + }), + state_schema, + ) + .with_guarantee( // Summary *state* carries no caller-visible guarantee; only a // finalized value does. An exact accumulator's state is its value. - guarantee: if estimate { None } else { guarantee.clone() }, - }); + if estimate { None } else { guarantee.clone() }, + ); + let agg = std::rc::Rc::new(agg); match query { - // The readout: downstream of the estimate the schema is the plain + // The evaluation: downstream of the estimate the schema is the plain // pre-ASAP row shape again (the summary-state type does not // propagate). - Some(query) => Ok(Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: agg, - query, - }, - schema: readout_schema, - guarantee, - })), + Some(query) => Ok(std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: agg, + query, + }), + evaluation_schema, + ) + .with_guarantee(guarantee), + )), None => Ok(agg), } } -// Heap readout rows contain the encoded item identity, subpopulation keys, +// Heap evaluation rows contain the encoded item identity, subpopulation keys, // and an estimated score. They never inherit the exact-value producer's schema. -fn keyed_heap_readout_schema( +fn keyed_heap_evaluation_schema( input: &PhysicalSummaryInput, - node: &QueryExpr, + node: &OperatorNode, ) -> Result { - let source = input.child.output_schema()?; + let source = &input.child.schema; let mut refs = Vec::new(); - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, child, .. - } = node + }) = node.non_asap() else { return Err(RealizationError::PhysicalRealization( - "heap readout requires an aggregate", + "heap evaluation requires an aggregate", )); }; if let Reduction::Reduce(groups) = reduction { if groups.is_without() { return Err(RealizationError::PhysicalRealization( - "heap readout requires explicit grouping", + "heap evaluation requires explicit grouping", )); } for index in groups.iter() { @@ -3161,7 +3326,7 @@ fn keyed_heap_readout_schema( .ok_or(RealizationError::PhysicalRealization( "heap item identity is missing", ))?, - &source, + source, &mut refs, )?; let mut fields = Vec::::new(); @@ -3197,7 +3362,7 @@ fn keyed_heap_readout_schema( } if fields.is_empty() { return Err(RealizationError::PhysicalRealization( - "heap readout has no identity columns", + "heap evaluation has no identity columns", )); } fields.push(Field::new( @@ -3208,7 +3373,7 @@ fn keyed_heap_readout_schema( Ok(Schema::lifted(fields, None)) } -fn ranking_score_index(logical: &QueryExpr, values: &Schema) -> Result { +fn ranking_score_index(logical: &OperatorNode, values: &Schema) -> Result { if is_current_series_source(logical) { return values .fields @@ -3221,11 +3386,11 @@ fn ranking_score_index(logical: &QueryExpr, values: &Schema) -> Result, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !matches!(family, FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) - || !matches!(child.as_ref(), QueryExpr::Aggregate { reduction: Reduction::PerEntity, measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) + || !matches!(child.non_asap(), Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "counter ranking needs a valid value schema", - ); - }; + let schema = &child.schema; if !schema.closed { return PhysicalSummaryInputRuleResult::Unsupported( "counter ranking needs the complete resolved series identity", @@ -3332,7 +3493,7 @@ fn realize_current_series_summary_input( intent: &AggIntent, family: &FieldDataType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !is_current_series_source(child) { return PhysicalSummaryInputRuleResult::NotApplicable; @@ -3359,11 +3520,7 @@ fn realize_current_series_summary_input( "snapshot ranking requires resolved partitions", ); } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "snapshot ranking requires a valid source schema", - ); - }; + let schema = &child.schema; let items = schema .fields .iter() @@ -3393,7 +3550,7 @@ fn realize_keyed_additive_summary_input( intent: &AggIntent, family: &FieldDataType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) { return PhysicalSummaryInputRuleResult::NotApplicable; @@ -3408,18 +3565,18 @@ fn realize_keyed_additive_summary_input( ) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, having: None, child: raw_child, .. - } = child.as_ref() + }) = child.non_asap() else { return PhysicalSummaryInputRuleResult::NotApplicable; }; let counter_input = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(raw_child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(raw_child.non_asap(), Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])); let weight = match measures.as_slice() { [AggIntent::Count { .. }] => SummaryInputExpr::Constant(1.0), @@ -3511,9 +3668,8 @@ fn realize_keyed_additive_summary_input( }) } -fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { - let schema = child.output_schema().ok()?; - let column = schema.fields.get(index)?; +fn schema_column_ref(child: &OperatorNode, index: usize) -> Option { + let column = child.schema.fields.get(index)?; Some(match &column.table { Some(table) => ColumnRef::Qualified { table: table.clone(), @@ -3535,7 +3691,7 @@ fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { fn compose_guarantee( family: &FieldDataType, query: Option<&PostAsapSketchStatistic>, - child: &SummaryNode, + child: &OperatorNode, intent: &AggIntent, accuracy: &dyn AccuracyModel, evidence: &dyn AccuracyEvidenceProvider, @@ -3685,8 +3841,8 @@ fn summarised_input( )) } -/// The `SummaryEstimate` readout for a summary-bound intent. -fn readout( +/// The `SummaryEstimate` evaluation for a summary-bound intent. +fn evaluation( intent: &AggIntent, input: &SummaryUpdate, cost_model: &dyn CostModel, @@ -3706,13 +3862,15 @@ fn readout( value: None, }, // Core doesn't know the shape of a deployment-specific `Extension` - // intent, so it can't build its readout either — delegate to the + // intent, so it can't build its evaluation either — delegate to the // same `CostModel` that decided (via `realize_extension`) this - // intent gets a summary realization at all. See `readout_extension`'s + // intent gets a summary realization at all. See `evaluation_extension`'s // doc for the invariant this depends on. AggIntent::Extension { ext_kind, payload } => match &input.weight { - SummaryInputExpr::Column(col) => cost_model.readout_extension(ext_kind, payload, col), - _ => unreachable!("extension readout requires one column"), + SummaryInputExpr::Column(col) => { + cost_model.evaluation_extension(ext_kind, payload, col) + } + _ => unreachable!("extension evaluation requires one column"), }, other => { unreachable!("no summary realization for {other:?} (realizations_for_intent)") @@ -3720,16 +3878,9 @@ fn readout( } } -/// Lift a pre-ASAP [`Schema`] to a [`Schema`] with every column -/// `FieldDataType::Plain` — shared by [`construct_summary_agg`] and -/// [`keep_pre_asap`], both in this module. -fn lift(schema: &Schema) -> Schema { - Schema::lifted(schema.fields.clone(), schema.time_index) -} - // ── SharedSubDAGStrategy ──────────────────────────────────────────────── -/// Wraps `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing +/// Wraps `asap_types::ir::cse::share_common_sub_dags`'s sharing /// decision as an explicit candidate pair, wherever a [`TargetSubDAG`] /// already has two or more consumers. /// @@ -3762,11 +3913,11 @@ impl ReplacementStrategy for SharedSubDAGStrategy { strategy: "SharedSubDAGStrategy", // The already-interned `Rc` itself: reusing it verbatim *is* // "build once and share" — no new node to construct. - replacement: Replacement::Rewrite(Rc::clone(target.root)), + replacement: Replacement::SubDAG(Rc::clone(target.root)), provenance: ReplacementProvenance::CseShare, rationale: format!( "build once and share: share_common_sub_dags already interned this \ - sub-DAG once and reused it across {count} consumers — one build can \ + sub_dag once and reused it across {count} consumers — one build can \ answer all of them instead of computing it {count} times" ), }, @@ -3775,11 +3926,11 @@ impl ReplacementStrategy for SharedSubDAGStrategy { // A structurally-identical but freshly-allocated `Rc`: same // value (`PartialEq`), deliberately *not* the same pointer, // representing "undo the sharing and recompute independently". - replacement: Replacement::Rewrite(Rc::new((**target.root).clone())), + replacement: Replacement::SubDAG(Rc::new((**target.root).clone())), provenance: ReplacementProvenance::CseRecompute, rationale: format!( "build independently: undo the sharing share_common_sub_dags found and \ - recompute this sub-DAG separately at each of its {count} consumers — \ + recompute this sub_dag separately at each of its {count} consumers — \ worth it only when independence outweighs the shared-maintenance cost, \ a CostModel's call (e.g. CostModel::cse_share_decision) and not this \ strategy's" @@ -3798,14 +3949,14 @@ impl ReplacementStrategy for SharedSubDAGStrategy { /// A generous, documented backstop against a hypothetically ill-behaved /// future [`ReplacementStrategy`] (see the module docs' "Termination" /// section) — not a bound either shipped strategy could ever approach. -/// [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] both converge in +/// [`ASAPStrategies`] and [`SharedSubDAGStrategy`] both converge in /// exactly 2 passes over a fixed target set, regardless of workload size. pub const MAX_SEARCH_ITERATIONS: usize = 1_000; // ── TargetSubDAGCandidates ────────────────────────────────────────────── /// Candidates for one distinct [`TargetSubDAG`] (its -/// own `target` `Rc`, keyed by pointer identity in +/// own `target` `Rc`, keyed by pointer identity in /// [`CandidateLogicalASAPDAGs`]'s internal map — never re-derived by value) plus every /// [`ReplacementSubDAG`] alternative any registered [`ReplacementStrategy`] /// proposed for it. @@ -3818,7 +3969,7 @@ pub const MAX_SEARCH_ITERATIONS: usize = 1_000; #[derive(Debug, Clone)] pub struct TargetSubDAGCandidates { /// The target sub-DAG this group is for. - pub target: Rc, + pub target: Rc, /// How many operator-child positions across the whole workload /// reference this exact `Rc` — see [`discover_targets`]. pub consumer_count: usize, @@ -3835,7 +3986,7 @@ pub struct TargetSubDAGCandidates { } impl TargetSubDAGCandidates { - fn new(target: Rc, consumer_count: usize) -> Self { + fn new(target: Rc, consumer_count: usize) -> Self { Self { target, consumer_count, @@ -3852,10 +4003,12 @@ impl TargetSubDAGCandidates { fn add_candidate(&mut self, candidate: ReplacementSubDAG) -> bool { let is_duplicate = self.candidates.iter().any(|existing| { match (&existing.replacement, &candidate.replacement) { - (Replacement::Rewrite(existing_rc), Replacement::Rewrite(rc)) => { + (Replacement::SubDAG(existing_rc), Replacement::SubDAG(rc)) + if is_logical_rewrite(existing_rc) && is_logical_rewrite(rc) => + { is_duplicate_rewrite(existing_rc, rc, &self.target) } - (Replacement::Summary(existing_node), Replacement::Summary(node)) => { + (Replacement::SubDAG(existing_node), Replacement::SubDAG(node)) => { is_duplicate_summary(existing_node, node) } ( @@ -3876,11 +4029,11 @@ impl TargetSubDAGCandidates { } } -/// Are `existing` and `candidate` the same [`Replacement::Rewrite`] -/// candidate for a group targeting `target`? +/// Are `existing` and `candidate` the same logical-rewrite +/// [`Replacement::SubDAG`] candidate for a group targeting `target`? /// -/// Structural (`QueryExpr`) value equality alone is *not* enough here: this -/// module's one shipped multi-candidate `Replacement::Rewrite` source, +/// Structural (`OperatorNode`) value equality alone is *not* enough here: +/// this module's one shipped multi-candidate logical-rewrite source, /// [`SharedSubDAGStrategy`], deliberately returns **two** candidates that /// are value-equal to each other (`build once and share` vs. `build /// independently` — see that strategy's own doc) but represent genuinely @@ -3898,7 +4051,7 @@ impl TargetSubDAGCandidates { /// So: two candidates whose "is this the target's own `Rc`?" bit disagrees /// are never duplicates of each other, full stop. Only when that bit /// *agrees* does this fall through to the real dedup discipline — -/// [`structural_hash`] as a candidate-narrowing filter, `QueryExpr`'s +/// [`structural_hash`] as a candidate-narrowing filter, `OperatorNode`'s /// derived `PartialEq` as the actual decision — protecting against the /// (currently hypothetical, since neither shipped strategy causes it) /// case of the exact same alternative being proposed twice. A fresh @@ -3907,9 +4060,9 @@ impl TargetSubDAGCandidates { /// wider traversal to amortize the cache across the way `InternTable`'s own /// use of `structural_hash` does. fn is_duplicate_rewrite( - existing: &Rc, - candidate: &Rc, - target: &Rc, + existing: &Rc, + candidate: &Rc, + target: &Rc, ) -> bool { let existing_is_target = Rc::ptr_eq(existing, target); let candidate_is_target = Rc::ptr_eq(candidate, target); @@ -3921,16 +4074,16 @@ fn is_duplicate_rewrite( && existing == candidate } -/// Are `existing` and `candidate` the same [`Replacement::Summary`] -/// candidate? +/// Are `existing` and `candidate` the same bound-summary +/// [`Replacement::SubDAG`] candidate? /// -/// [`SummaryNode`] derives neither `PartialEq` nor `Hash` (it embeds -/// `SketchParams`/`f64`-bearing accuracy targets deep inside `SummaryExpr`, -/// the same reason `QueryExpr` can't derive `Hash` either — see -/// [`structural_hash`]'s own doc). Per this module's inherited "hash is a -/// filter, `PartialEq` is the decision, no exceptions" rule, there is no -/// real equality check to back a dedup *decision* here — and skipping the -/// check is the only choice that rule permits: never merging two candidates +/// A bound summary embeds `SketchParams`/`f64`-bearing accuracy targets and +/// guarantees, so value equality is not a dedup decision this module is +/// willing to make (see [`structural_hash`]'s own doc on `f64` hashing). +/// Per this module's inherited "hash is a filter, `PartialEq` is the +/// decision, no exceptions" rule, there is no real equality check to back a +/// dedup *decision* here — and skipping the check is the only choice that +/// rule permits: never merging two candidates /// is harmless (at worst, a redundant entry in a group's candidate list), /// while comparing by some proxy this module can't actually verify (e.g. /// `Debug` text, or `ReplacementSubDAG::rationale` — documented elsewhere in @@ -3939,7 +4092,7 @@ fn is_duplicate_rewrite( /// shipped today already return a structurally distinct candidate for every /// entry of one `replacements()` call, so this is future-proofing against a /// hypothetical repeat call, not a gap either strategy's own tests exercise. -fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { +fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { false } @@ -3948,34 +4101,34 @@ fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc` whose -/// group holds its alternatives. +/// caller can still map a `Root`'s `Id` back to the `Rc` whose +/// group holds its alternatives. Memos are keyed by `*const OperatorNode`. pub struct CandidateLogicalASAPDAGs { /// The workload's roots, after the one `share_common_sub_dags` pass /// [`search_workload_with`] runs up front — the same post-CSE roots /// every `TargetSubDAG` in `groups` was discovered from. - pub roots: Vec<(Id, Rc)>, - groups: HashMap<*const QueryExpr, TargetSubDAGCandidates>, + pub roots: Vec<(Id, Rc)>, + groups: HashMap<*const OperatorNode, TargetSubDAGCandidates>, /// Discovery order — stable iteration for [`CandidateLogicalASAPDAGs::target_subdag_candidates`]/ /// [`CandidateLogicalASAPDAGs::cost_sorted`], since `HashMap` iteration order isn't. - order: Vec<*const QueryExpr>, + order: Vec<*const OperatorNode>, /// Composition proofs are computed with the search model, then retained /// through costing and DAG assembly so no later default can replace it. composition_plans: Vec, } struct PreparedComposition { - target: *const QueryExpr, + target: *const OperatorNode, operation: ExactComposition, - child: Rc, - plan: Rc, + child: Rc, + plan: Rc, } impl CandidateLogicalASAPDAGs { fn prepare_compositions( &mut self, accuracy: &dyn AccuracyModel, - targets: &HashMap<*const QueryExpr, Vec>, + targets: &HashMap<*const OperatorNode, Vec>, ) { self.composition_plans.clear(); for group in self.groups.values() { @@ -3990,14 +4143,16 @@ impl CandidateLogicalASAPDAGs { .into_iter() .flat_map(|g| &g.candidates) .filter_map(|c| match &c.replacement { - Replacement::Summary(child) if operation.accepts_child(child) => { + Replacement::SubDAG(child) + if !is_logical_rewrite(child) && operation.accepts_child(child) => + { Some(Rc::clone(child)) } _ => None, }) .collect(), OperationPlacement::Maintenance => { - keep_pre_asap(&operation.child_target).into_iter().collect() + retain_exact(&operation.child_target).into_iter().collect() } }; for child in children { @@ -4035,11 +4190,11 @@ impl CandidateLogicalASAPDAGs { /// never a silently truncated inventory presented as exhaustive. #[derive(Debug)] pub struct CandidateDAGInventory { - pub candidates: Vec)>>, + pub candidates: Vec)>>, pub rejected_assemblies: Vec, } -type CandidateDAGChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); +type CandidateDAGChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); impl CandidateLogicalASAPDAGs { pub fn enumerate_candidate_dags( @@ -4074,7 +4229,7 @@ impl CandidateLogicalASAPDAGs { fn enumerate_candidate_roots( &self, - roots: &[(Id, Rc)], + roots: &[(Id, Rc)], expansion_limit: usize, ) -> Result, RealizationError> { let mut reachable = Vec::new(); @@ -4090,8 +4245,10 @@ impl CandidateLogicalASAPDAGs { cursor += 1; if let Some(group) = self.groups.get(&ptr) { for candidate in &group.candidates { - if let Replacement::Rewrite(rewritten) = &candidate.replacement { - walk(rewritten, &mut reachable, &mut nodes, &mut counts); + if let Replacement::SubDAG(rewritten) = &candidate.replacement { + if is_logical_rewrite(rewritten) { + walk(rewritten, &mut reachable, &mut nodes, &mut counts); + } } } } @@ -4185,77 +4342,12 @@ impl CandidateLogicalASAPDAGs { .collect::, _>>(); match roots { Ok(roots) => { - let roots = asap_types::post_asap::share_common_summary_sub_dags(roots); + let roots = share_common_sub_dags(roots); use std::hash::{Hash, Hasher}; let mut hash = std::collections::hash_map::DefaultHasher::new(); - let mut pending = roots - .iter() - .map(|(_, node)| node.as_ref()) - .collect::>(); - while let Some(node) = pending.pop() { - std::mem::discriminant(&node.expr).hash(&mut hash); - let raw = match &node.expr { - SummaryExpr::KeepPreAsap(raw) => Some(raw.as_ref()), - _ => None, - }; - let operation = match &node.expr { - SummaryExpr::ValueOperation { - timing, operation, .. - } => serde_json::json!((timing, operation)), - SummaryExpr::BinaryOp { - timing, operator, .. - } => serde_json::json!((timing, operator)), - SummaryExpr::SummaryMerge { timing, .. } => serde_json::json!(timing), - _ => serde_json::Value::Null, - }; - let mut value = - serde_json::to_value((&node.schema, &node.guarantee, raw, operation)) - .map_err(|_| { - RealizationError::PhysicalRealization( - "candidate identity serialization failed", - ) - })?; - fn normalize(value: &mut serde_json::Value) { - match value { - serde_json::Value::Number(number) - if number.as_f64() == Some(0.0) => - { - *value = serde_json::json!(0); - } - serde_json::Value::Array(values) => { - values.iter_mut().for_each(normalize) - } - serde_json::Value::Object(values) => { - values.values_mut().for_each(normalize) - } - _ => {} - } - } - normalize(&mut value); - value.sort_all_objects(); - value.to_string().hash(&mut hash); - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - pending.extend([lhs.as_ref(), rhs.as_ref()]) - } - SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::SummarySubtract { left, right } => { - pending.extend([left.as_ref(), right.as_ref()]) - } - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryAgg { child, .. } => pending.push(child.as_ref()), - SummaryExpr::SummaryJoin { outer, inner, .. } => { - pending.extend([outer.as_ref(), inner.as_ref()]) - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - pending.push(summary_input.as_ref()) - } - SummaryExpr::SummaryMerge { children, .. } => { - pending.extend(children.iter().map(|child| child.as_ref())) - } - } + let mut cache = HashCache::new(); + for (_, node) in &roots { + structural_hash(node, &mut cache).hash(&mut hash); } let bucket = seen.entry(hash.finish()).or_default(); if !bucket @@ -4281,25 +4373,25 @@ impl CandidateLogicalASAPDAGs { /// Lifecycle-aware whole-subplan costs keyed by target and candidate identity. #[derive(Default, Clone)] pub(crate) struct CandidateCostOverrides { - costs: HashMap<(*const QueryExpr, *const ReplacementSubDAG), Cost>, - raw_costs: HashMap<*const QueryExpr, Cost>, + costs: HashMap<(*const OperatorNode, *const ReplacementSubDAG), Cost>, + raw_costs: HashMap<*const OperatorNode, Cost>, /// Targets for which the caller requested an atomic raw-vs-summary /// decision. Other memo groups continue through ordinary CSE selection. - finalized_targets: HashSet<*const QueryExpr>, + finalized_targets: HashSet<*const OperatorNode>, } impl CandidateCostOverrides { - pub(crate) fn finalize_target(&mut self, target: &Rc) { + pub(crate) fn finalize_target(&mut self, target: &Rc) { self.finalized_targets.insert(Rc::as_ptr(target)); } - fn finalizes(&self, target: &Rc) -> bool { + fn finalizes(&self, target: &Rc) -> bool { self.finalized_targets.contains(&Rc::as_ptr(target)) } pub(crate) fn insert( &mut self, - target: &Rc, + target: &Rc, candidate: &ReplacementSubDAG, cost: Cost, ) { @@ -4307,17 +4399,17 @@ impl CandidateCostOverrides { .insert((Rc::as_ptr(target), candidate as *const _), cost); } - fn get(&self, target: &Rc, candidate: &ReplacementSubDAG) -> Option { + fn get(&self, target: &Rc, candidate: &ReplacementSubDAG) -> Option { self.costs .get(&(Rc::as_ptr(target), candidate as *const _)) .copied() } - pub(crate) fn insert_raw(&mut self, target: &Rc, cost: Cost) { + pub(crate) fn insert_raw(&mut self, target: &Rc, cost: Cost) { self.raw_costs.insert(Rc::as_ptr(target), cost); } - fn raw(&self, target: &Rc) -> Option { + fn raw(&self, target: &Rc) -> Option { self.raw_costs.get(&Rc::as_ptr(target)).copied() } } @@ -4334,7 +4426,7 @@ impl CandidateLogicalASAPDAGs { } /// Whether no targets were discovered at all (an empty workload, or one - /// with no `QueryExpr` nodes reachable from any root — never true for a + /// with no `OperatorNode`s reachable from any root — never true for a /// non-empty `roots`, since every root is itself a target). pub fn is_empty(&self) -> bool { self.groups.is_empty() @@ -4343,7 +4435,10 @@ impl CandidateLogicalASAPDAGs { /// The candidate set for `target`, if `target`'s own `Rc` is a discovered /// `TargetSubDAG` (i.e. `Rc::ptr_eq` to some node reachable from /// `roots`). - pub fn candidates_for_target(&self, target: &Rc) -> Option<&TargetSubDAGCandidates> { + pub fn candidates_for_target( + &self, + target: &Rc, + ) -> Option<&TargetSubDAGCandidates> { self.groups.get(&Rc::as_ptr(target)) } @@ -4465,17 +4560,17 @@ impl CandidateLogicalASAPDAGs { /// half of issue #287. Looked up by `Rc` pointer identity, the same /// currency [`CandidateLogicalASAPDAGs::candidates_for_target`]/[`GlobalSelection::for_target`] already /// use. -/// Holds an owned `Rc` clone alongside each profile (not just its -/// raw pointer) so this map keeps every node it describes alive for as long -/// as the map itself lives — a `RecurrenceProfileMap` is safe to outlive the -/// `CandidateLogicalASAPDAGs` it was built from. Without this, a raw `*const QueryExpr` key +/// Holds an owned `Rc` clone alongside each profile (not just +/// its raw pointer) so this map keeps every node it describes alive for as +/// long as the map itself lives — a `RecurrenceProfileMap` is safe to outlive +/// the `CandidateLogicalASAPDAGs` it was built from. Without this, a raw `*const OperatorNode` key /// could, after the originating `CandidateLogicalASAPDAGs` (the only other owner of those /// `Rc`s) is dropped, collide with an unrelated, later allocation that /// happens to reuse the same freed address — silently returning a stale /// profile for the wrong node (issue #287 review, bug 4). #[derive(Debug, Clone)] pub struct RecurrenceProfileMap { - profiles: HashMap<*const QueryExpr, (Rc, RecurrenceProfile)>, + profiles: HashMap<*const OperatorNode, (Rc, RecurrenceProfile)>, } impl RecurrenceProfileMap { @@ -4484,7 +4579,7 @@ impl RecurrenceProfileMap { /// in the [`CandidateLogicalASAPDAGs`] this map was built from (or carried no /// recurring/one-shot/update-rate metadata at all) — always a valid, /// "no metadata" answer, never a panic. - pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { + pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { self.profiles .get(&Rc::as_ptr(target)) .map(|(_, profile)| *profile) @@ -4504,7 +4599,7 @@ impl CandidateLogicalASAPDAGs { /// `self.roots[i]` — the same order [`search_workload`]/ /// [`search_workload_with`] were originally called with (post-CSE /// dedup preserves both root count and order — see - /// `asap_types::pre_asap::cse::share_common_sub_dags`'s own + /// `asap_types::ir::cse::share_common_sub_dags`'s own /// `.map(...).collect()` body). This keeps `Id` fully opaque (no `Eq`/ /// `Hash`/`Clone` bound needed on it at all — issue #287's "keep /// caller/query identifiers opaque" requirement) at the cost of the @@ -4581,12 +4676,12 @@ impl CandidateLogicalASAPDAGs { } } - let mut rates: HashMap<*const QueryExpr, f64> = HashMap::new(); - let mut one_shot_counts: HashMap<*const QueryExpr, usize> = HashMap::new(); + let mut rates: HashMap<*const OperatorNode, f64> = HashMap::new(); + let mut one_shot_counts: HashMap<*const OperatorNode, usize> = HashMap::new(); // Sites actually reached by at least one root's own recurrence tag // during the walk below — see this method's own "Unreachable // sites" doc. - let mut reached: HashSet<*const QueryExpr> = HashSet::new(); + let mut reached: HashSet<*const OperatorNode> = HashSet::new(); for ((_, root), recurrence) in self.roots.iter().zip(root_recurrence) { let recurrence = *recurrence; @@ -4596,7 +4691,7 @@ impl CandidateLogicalASAPDAGs { // recomputed occurrence is evaluated twice as well; stopping // expansion after the first pointer visit undercounts exactly // the effective-consumer rate recurrence-aware costing needs. - let mut queue: VecDeque<(*const QueryExpr, usize)> = VecDeque::new(); + let mut queue: VecDeque<(*const OperatorNode, usize)> = VecDeque::new(); queue.push_back((root_ptr, 1)); while let Some((ptr, path_count)) = queue.pop_front() { @@ -4740,7 +4835,7 @@ impl CandidateLogicalASAPDAGs { &self, workload: &QueryWorkload, root_workload_entries: &[usize], - ) -> Result>, RecurrenceError> { + ) -> Result>, RecurrenceError> { let entry_count = workload.entries().count(); if root_workload_entries.len() != self.roots.len() { return Err(RecurrenceError::RootCountMismatch { @@ -4748,7 +4843,7 @@ impl CandidateLogicalASAPDAGs { got: root_workload_entries.len(), }); } - let mut bindings: HashMap<*const QueryExpr, HashSet> = HashMap::new(); + let mut bindings: HashMap<*const OperatorNode, HashSet> = HashMap::new(); for ((_, root), &entry_index) in self.roots.iter().zip(root_workload_entries) { if entry_index >= entry_count { return Err(RecurrenceError::InvalidWorkloadEntry { @@ -4790,12 +4885,12 @@ impl CandidateLogicalASAPDAGs { /// child always has `edge_count >= 1` in practice, but this keeps the /// helper correct regardless). fn contribute( - ptr: *const QueryExpr, + ptr: *const OperatorNode, times: usize, recurrence: RootRecurrence, - rates: &mut HashMap<*const QueryExpr, f64>, - one_shot_counts: &mut HashMap<*const QueryExpr, usize>, - reached: &mut HashSet<*const QueryExpr>, + rates: &mut HashMap<*const OperatorNode, f64>, + one_shot_counts: &mut HashMap<*const OperatorNode, usize>, + reached: &mut HashSet<*const OperatorNode>, ) { if times == 0 { return; @@ -4816,7 +4911,7 @@ fn contribute( /// [`CandidateLogicalASAPDAGs::cost_sorted`]. #[derive(Debug)] pub struct RankedTargetSubDAGCandidates<'a> { - pub target: &'a Rc, + pub target: &'a Rc, pub consumer_count: usize, pub candidates: Vec<&'a ReplacementSubDAG>, /// `costs[i]` is `candidates[i]`'s own grouping-state cost when available, @@ -4864,7 +4959,7 @@ fn rank_group<'a>( // estimate, compare N independent states with the shared grid directly. let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); let has_hydra = ranked.iter().any(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return false; }; summary_grouping(node).is_some_and(|grouping| { @@ -4896,7 +4991,7 @@ fn rank_group<'a>( return ranked; } - // Shape 3: `SketchAlgorithmStrategy`'s sketch-family candidates (every + // Shape 3: `ASAPStrategies`'s sketch-family candidates (every // candidate is a `Summary` that realizes a `SketchAlgorithm`) — rank via // `CostModel::rank_candidates`, the same hook `realizations_for_intent` // itself consults. @@ -4904,16 +4999,16 @@ fn rank_group<'a>( let kinds: Option> = ranked .iter() .map(|c| match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }) .collect(); if let Some(kinds) = kinds { let order = crate::cost_model::validated_candidate_ranking(cost_model, intent, &kinds); ranked.sort_by_key(|c| { let kind = match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }; kind.and_then(|k| order.iter().position(|o| *o == k)) .unwrap_or(usize::MAX) @@ -4970,21 +5065,21 @@ fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> }) } -/// [`cse_preference`] only needs one representative bound [`SummaryNode`] +/// [`cse_preference`] only needs one representative bound [`OperatorNode`] /// for `target` (to build a [`CseCandidate`] for /// [`CostModel::cse_share_decision`]), not the full ranked candidate list -/// [`SketchAlgorithmStrategy::replacements`] returns — so this just reuses +/// [`ASAPStrategies::replacements`] returns — so this just reuses /// [`realize_child`], the same rank-and-take-first helper /// `construct_summary_agg`'s own recursion and /// [`crate::cost_model::DefaultCostModel::estimate_cost`] already use, /// wrapped to swallow the (here, uninteresting) error into `None`. -fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option> { +fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option> { realize_child(target, cost_model).ok() } -/// The `SketchAlgorithm` a bound [`Replacement::Summary`] candidate ultimately +/// The `SketchAlgorithm` a bound [`Replacement::SubDAG`] candidate ultimately /// realizes, if any (`None` for an `ExactAggregate`/pass-through -/// `Summary` — nothing to rank against another `SketchAlgorithm`). +/// sub-DAG — nothing to rank against another `SketchAlgorithm`). /// /// Mirrors this module's own `#[cfg(test)]`-only `summary_family_algorithm` /// helper (in the test module below), which does the identical @@ -4993,23 +5088,27 @@ fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_kind_of(summary_input), - SummaryExpr::SummaryAgg { +fn sketch_kind_of(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + sketch_kind_of(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } => Some(kind.algorithm().clone()), + }) => Some(kind.algorithm().clone()), _ => None, } } /// The grouping strategy used by a bound summary candidate, unwrapping its -/// readout node when necessary. -fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => summary_grouping(summary_input), - SummaryExpr::SummaryAgg { grouping, .. } => Some(grouping), +/// evaluation node when necessary. +fn summary_grouping(node: &OperatorNode) -> Option<&GroupingStrategy> { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + summary_grouping(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { grouping, .. }) => Some(grouping), _ => None, } } @@ -5034,7 +5133,7 @@ fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { #[derive(Debug)] pub struct TargetSubDAGSelection<'a> { /// The target sub-DAG this selection is for. - pub target: &'a Rc, + pub target: &'a Rc, /// [`TargetSubDAGCandidates::consumer_count`] — how many operator-child positions /// directly reference `target`, ignoring every ancestor's own choice. pub consumer_count: usize, @@ -5065,18 +5164,18 @@ pub struct TargetSubDAGSelection<'a> { #[derive(Debug)] pub struct CompositionDecision<'a> { /// The exact child/operation pair validated by the search accuracy model. - pub plan: Rc, + pub plan: Rc, /// The child target the composed operator consumes. - pub child_target: &'a Rc, + pub child_target: &'a Rc, /// For a read-time operation: the child's own candidate committed alongside - /// (the summary readout the operator folds). `None` for an update-path + /// (the summary evaluation the operator folds). `None` for an update-path /// transform, whose input is raw update data — its cost is charged to /// the maintained summary *above* it instead. pub child_candidate: Option<&'a ReplacementSubDAG>, /// The composed plan's recurring rate — `read_operation_plan_cost_rate` /// or `maintenance_operation_plan_cost_rate`. pub cost_rate: CostRate, - /// `raw_recompute_cost_rate` — the `KeepPreAsap` baseline it beat. + /// `raw_recompute_cost_rate` — the kept-sub-DAG baseline it beat. pub baseline_rate: CostRate, /// The statistics (and their provenance) both rates were computed from. pub inputs: ExactCompositionCostInputs, @@ -5087,12 +5186,13 @@ pub struct CompositionDecision<'a> { /// [`CandidateLogicalASAPDAGs::cost_sorted`] use. #[derive(Debug)] pub struct GlobalSelection<'a> { - order: Vec<*const QueryExpr>, - groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'a>>, + order: Vec<*const OperatorNode>, + groups: HashMap<*const OperatorNode, TargetSubDAGSelection<'a>>, /// [`Self::assemble_selected_dag`]'s memo — one bound node per target for the /// life of this selection, so two parents composing over one shared - /// child get the *same* `Rc`. - assembled_nodes: RefCell>>, + /// child get the *same* `Rc` (a kept pre-ASAP sub-DAG + /// shared by two parents stays one `Rc` the same way). + assembled_nodes: RefCell>>, } fn normalize_cross_input_equi_predicate( @@ -5100,15 +5200,17 @@ fn normalize_cross_input_equi_predicate( left_width: usize, total_width: usize, ) -> Option { - let QueryExpr::Compare { + let ScalarExpr::Compare { left, op: asap_types::pre_asap::CompareOpKind::Eq, right, - } = pred.0.as_ref() + semantics, + } = &pred.0 else { return None; }; - let (QueryExpr::Column(left_id), QueryExpr::Column(right_id)) = (left.as_ref(), right.as_ref()) + let (ScalarExpr::Column(left_id), ScalarExpr::Column(right_id)) = + (left.as_ref(), right.as_ref()) else { return None; }; @@ -5121,20 +5223,12 @@ fn normalize_cross_input_equi_predicate( } else { return None; }; - Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left_id)), + Some(Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(left_id)), op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right_id)), - }))) -} - -fn relational_join_guarantee( - left: Option<&ResultGuarantee>, - right: Option<&ResultGuarantee>, -) -> Option { - left.zip(right) - .filter(|(left, right)| left.is_exact() && right.is_exact()) - .map(|_| ResultGuarantee::exact("RelationalJoin over exact inputs")) + right: Box::new(ScalarExpr::Column(right_id)), + semantics: *semantics, + })) } impl<'a> GlobalSelection<'a> { @@ -5146,47 +5240,52 @@ impl<'a> GlobalSelection<'a> { /// The selection for `target`, if `target`'s own `Rc` is a discovered /// site (i.e. `Rc::ptr_eq` to some node reachable from the workload's /// roots). - pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { + pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { self.groups.get(&Rc::as_ptr(target)) } /// Link this selection's per-site decisions into one data_state-validated /// post-ASAP DAG rooted at `target` — the one place a committed - /// composition's child *reference* becomes an actual `Rc` + /// composition's child *reference* becomes an actual `Rc` /// edge (issue #171). `None` if `target` is not a discovered site. /// /// Per site: a [`Replacement::ExactComposition`] uses its validated /// operation/child plan, retaining the search model's guarantee; - /// a [`Replacement::Summary`] is + /// a bound-summary [`Replacement::SubDAG`] is /// re-linked so its `SummaryAgg` child is the child target's own /// DAG assembly whenever that is phase-legal beneath maintenance /// (so a child that chose an `ValueOperationAtIngestionTime` actually ends up under - /// the summary); a [`Replacement::Rewrite`] or an unmatched site stays - /// the conservative `KeepPreAsap`. Memoized by target identity, so a - /// shared inner summary is one `Rc` no matter how many roots reach it. + /// the summary); a logical-rewrite [`Replacement::SubDAG`] is kept + /// as it is (exact); an unmatched site keeps its own operator with each + /// child assembled independently ([`Self::assemble_residual`]). + /// Memoized by target identity, so a shared inner summary is one `Rc` + /// no matter how many roots reach it. pub fn assemble_selected_dag( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { if !self.groups.contains_key(&Rc::as_ptr(target)) { return Ok(None); } self.assemble_target(target).map(Some) } - /// Assemble a complete query result, including an exact-state readout when + /// Assemble a complete query result, including an exact-state evaluation when /// needed. `assemble_selected_dag` also serves internal state frontiers; /// callers exposing query results must use this boundary instead. pub fn assemble_selected_query( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { self.assemble_selected_dag(target)? .map(|node| finalize_query_candidate(node, target)) .transpose() } - fn assemble_target(&self, target: &Rc) -> Result, RealizationError> { + fn assemble_target( + &self, + target: &Rc, + ) -> Result, RealizationError> { let ptr = Rc::as_ptr(target); if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); @@ -5198,10 +5297,12 @@ impl<'a> GlobalSelection<'a> { .groups .get(&ptr) .and_then(|sel| sel.chosen) - .is_some_and(|candidate| matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { child, .. } - if !matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if contains_aggregate(raw))))); + .is_some_and(|candidate| { + matches!(&candidate.replacement, + Replacement::SubDAG(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) + if child.contains_asap() || !contains_aggregate(child))) + }); let node = if query_time_nested_sum(target) && !selected_composed_summary { self.assemble_residual(target)? } else { @@ -5212,8 +5313,10 @@ impl<'a> GlobalSelection<'a> { .map(|c| &c.replacement) { None => self.assemble_residual(target)?, - Some(Replacement::Rewrite(rewritten)) => keep_pre_asap(rewritten)?, - Some(Replacement::Summary(node)) => self.relink_summary(node, target)?, + Some(Replacement::SubDAG(node)) if node.contains_asap() => { + self.relink_summary(node, target)? + } + Some(Replacement::SubDAG(kept)) => retain_exact(kept)?, Some(Replacement::ExactComposition(_)) => Rc::clone( &self.groups[&ptr] .composition @@ -5229,130 +5332,124 @@ impl<'a> GlobalSelection<'a> { Ok(node) } - /// Preserve composable query-time value operators in post-ASAP form even - /// when the operator itself has no summary realization. Its child is - /// assembled independently, so a selected summary remains visible - /// beneath `Project`/`Filter`/`Sort`/`Limit` instead of being swallowed by - /// one opaque `KeepPreAsap` sub-DAG. + /// Keep `target`'s own operator and assemble each child independently, + /// so a selected summary remains visible beneath a relational operator + /// that has no summary realization of its own instead of being + /// swallowed by one opaque kept sub-DAG. Every child that is a + /// discovered target is assembled (and finalized to query-time values); + /// any other child is kept as it is. The guarantee is composed from the + /// assembled children: all exact → exact; exactly one child → that + /// child's guarantee; otherwise unknown. An inner `Join` first has its + /// cross-input equi-predicate normalized; any other join is kept whole. fn assemble_residual( &self, - target: &Rc, - ) -> Result, RealizationError> { - if let QueryExpr::Join { + target: &Rc, + ) -> Result, RealizationError> { + if target.children().is_empty() { + // A leaf has nothing to assemble beneath it: keep it as it is. + return retain_exact(target); + } + let mut operator = target.operator.clone(); + if let Operator::NonASAP(NonASAPOp::Join { left, right, kind, pred, - } = target.as_ref() + }) = &mut operator { - let left_width = left.output_schema()?.fields.len(); - let total_width = left_width + right.output_schema()?.fields.len(); - let normalized_pred = matches!(kind, asap_types::pre_asap::JoinKind::Inner) + let left_width = left.schema.fields.len(); + let total_width = left_width + right.schema.fields.len(); + let normalized_pred = matches!(kind, JoinKind::Inner) .then(|| normalize_cross_input_equi_predicate(pred, left_width, total_width)) .flatten(); - let Some(pred) = normalized_pred else { - return keep_pre_asap(target); + let Some(normalized) = normalized_pred else { + return retain_exact(target); }; - let left = finalize_query_candidate(self.assemble_target(left)?, left)?; - let right = finalize_query_candidate(self.assemble_target(right)?, right)?; - let guarantee = - relational_join_guarantee(left.guarantee.as_ref(), right.guarantee.as_ref()); - let node = Rc::new(SummaryNode { - expr: SummaryExpr::RelationalJoin { - left, - right, - kind: kind.clone(), - pred, - pruning: None, - }, - schema: lift(&target.output_schema()?), - guarantee, - }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; - return Ok(node); + *pred = normalized; } - let (child_target, operation) = match target.as_ref() { - QueryExpr::Project { - cols, - qualifier, - child, - } => ( - child, - ValueOperation::Project { - cols: cols.clone(), - qualifier: qualifier.clone(), - }, - ), - QueryExpr::Filter { pred, child } => { - (child, ValueOperation::Filter { pred: pred.clone() }) + let mut failure = None; + let mut children = Vec::new(); + let operator = operator.map_children(|child| { + if failure.is_some() { + return Rc::clone(child); } - QueryExpr::Sort { - keys, - partition_by, - child, - } => ( - child, - ValueOperation::Sort { - keys: keys.clone(), - partition_by: partition_by.clone(), - }, - ), - QueryExpr::Limit { n, offset, child } => ( - child, - ValueOperation::Limit { - n: *n, - offset: *offset, - partition_by: match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => partition_by.clone(), - _ => Default::default(), - }, - }, - ), - QueryExpr::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } if query_time_nested_sum(target) => ( - child, - ValueOperation::Exact(ExactOperation::Aggregate { - reduction: reduction.clone(), - measures: measures.clone(), - output_names: output_names.clone(), - filters: filters.clone(), - having: having.clone(), - }), - ), - _ => return keep_pre_asap(target), - }; - let child = finalize_query_candidate(self.assemble_target(child_target)?, child_target)?; - let guarantee = child.guarantee.clone(); - let node = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child, - operation, - timing: ExecutionTiming::QueryTime, - }, - schema: lift(&target.output_schema()?), - guarantee, + let assembled = if self.groups.contains_key(&Rc::as_ptr(child)) { + self.assemble_target(child) + .and_then(|node| finalize_query_candidate(node, child)) + } else { + Ok(Rc::clone(child)) + }; + match assembled { + Ok(node) => { + children.push(Rc::clone(&node)); + node + } + Err(error) => { + failure = Some(error); + Rc::clone(child) + } + } + }); + if let Some(error) = failure { + return Err(error); + } + // An operator that computes new values from its input rows has no + // sound accuracy composition over an approximate input (e.g. `max` + // over a quantile evaluation's rank error). Without a selected + // composition such a node stays an exact pre-ASAP sub-DAG; only the + // read-time nested SUM keeps its assembled children. + let computes_values = matches!( + target.non_asap(), + Some( + NonASAPOp::Aggregate { .. } + | NonASAPOp::BinaryOp { .. } + | NonASAPOp::SQLWindowFunc { .. } + ) + ) && !query_time_nested_sum(target); + let approximate_input = children.iter().any(|child| { + !child + .guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + if computes_values && approximate_input { + return retain_exact(target); + } + let guarantee = match children.as_slice() { + [child] => child.guarantee.clone(), + children + if children.iter().all(|child| { + child + .guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }) => + { + Some(ResultGuarantee::exact(format!( + "{} over exact inputs", + target.operator.kind_name() + ))) + } + _ => None, + }; + let node = Rc::new( + OperatorNode::with_schema(operator, target.schema.clone()).with_guarantee(guarantee), + ); + validate_default(&node, ExecutionTiming::QueryTime)?; Ok(node) } - /// Re-link a bound `Summary` candidate's `SummaryAgg` child to the + /// Re-link a bound summary candidate's `SummaryAgg` child to the /// child target's own DAG assembly when that is legal beneath /// maintenance; otherwise keep the candidate exactly as constructed. fn relink_summary( &self, - node: &Rc, - target: &Rc, - ) -> Result, RealizationError> { - let QueryExpr::Aggregate { + node: &Rc, + target: &Rc, + ) -> Result, RealizationError> { + let Some(NonASAPOp::Aggregate { child: pre_child, .. - } = target.as_ref() + }) = target.non_asap() else { return Ok(Rc::clone(node)); }; @@ -5377,16 +5474,16 @@ impl<'a> GlobalSelection<'a> { /// A mergeable outer SUM over a relationally wrapped aggregate is a read-time /// reduction of the inner summary values. Maintaining the outer SUM directly -/// would hide that inner temporal aggregate inside `KeepPreAsap` and lose its -/// independently selected summary. -fn query_time_nested_sum(target: &QueryExpr) -> bool { - let QueryExpr::Aggregate { +/// would hide that inner temporal aggregate inside one kept sub-DAG and lose +/// its independently selected summary. +fn query_time_nested_sum(target: &OperatorNode) -> bool { + let Some(NonASAPOp::Aggregate { measures, filters, having: None, child, .. - } = target + }) = target.non_asap() else { return false; }; @@ -5395,13 +5492,15 @@ fn query_time_nested_sum(target: &QueryExpr) -> bool { && contains_aggregate(child) } -fn contains_aggregate(expr: &QueryExpr) -> bool { - match expr { - QueryExpr::Aggregate { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => contains_aggregate(child), +fn contains_aggregate(expr: &OperatorNode) -> bool { + match expr.non_asap() { + Some(NonASAPOp::Aggregate { .. }) => true, + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => contains_aggregate(child), _ => false, } } @@ -5409,50 +5508,53 @@ fn contains_aggregate(expr: &QueryExpr) -> bool { /// Rebuild `node` (a `SummaryAgg`, possibly under a `SummaryEstimate`) with /// `new_child` as the `SummaryAgg`'s child, if the result still validates /// as maintained state; otherwise return `node` unchanged. -fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { +fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => { + }) => { let inner = relink_agg_child(summary_input, new_child); if Rc::ptr_eq(&inner, summary_input) { return Rc::clone(node); } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: inner, - query: query.clone(), - }, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - }) + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: inner, + query: query.clone(), + }), + node.schema.clone(), + ) + .with_guarantee(node.guarantee.clone()), + ) } - SummaryExpr::SummaryAgg { + Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, grouping, filter, - } => { + }) => { if Rc::ptr_eq(child, new_child) { return Rc::clone(node); } - let rebuilt = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(new_child), - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: filter.clone(), - }, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - }); - match validate_execution_data_states_at(&rebuilt, ExecutionDataState::INGESTION_SUMMARY) - { + let rebuilt = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(new_child), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: filter.clone(), + }), + node.schema.clone(), + ) + .with_guarantee(node.guarantee.clone()), + ); + match validate_default(&rebuilt, ExecutionTiming::IngestionTime) { Ok(_) => rebuilt, Err(_) => Rc::clone(node), } @@ -5461,13 +5563,15 @@ fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc) -> Option<&Rc> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => maintained_summary(summary_input), - SummaryExpr::SummaryAgg { .. } => Some(node), +fn maintained_summary(node: &Rc) -> Option<&Rc> { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + maintained_summary(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => Some(node), _ => None, } } @@ -5484,10 +5588,10 @@ struct CompositionContext { /// child target ptr → the child's candidate an ancestor's composition /// already committed to (a later parent must compose with the *same* /// one, and the child's own selection is forced to it). - committed_child: HashMap<*const QueryExpr, *const ReplacementSubDAG>, + committed_child: HashMap<*const OperatorNode, *const ReplacementSubDAG>, /// site ptr → the maintained `SummaryAgg` directly above it, when its - /// parent chose a bound `Summary` — what an `ValueOperationAtIngestionTime` here feeds. - maintaining_parent: HashMap<*const QueryExpr, Rc>, + /// parent chose a bound summary — what an `ValueOperationAtIngestionTime` here feeds. + maintaining_parent: HashMap<*const OperatorNode, Rc>, } /// One eligible composed alternative at a site, before the cheapest wins. @@ -5500,10 +5604,10 @@ struct CompositionOption<'a> { /// composed-plan rate is *known* and beats the raw-recompute baseline — /// costed against each compatible child candidate already in `CandidateLogicalASAPDAGs` /// (or the one an earlier parent committed). Unknown statistics yield no -/// option at all: the conservative `KeepPreAsap` path stays. +/// option at all: the conservative kept-sub-DAG path stays. fn composition_options<'a>( group: &'a TargetSubDAGCandidates, - groups: &'a HashMap<*const QueryExpr, TargetSubDAGCandidates>, + groups: &'a HashMap<*const OperatorNode, TargetSubDAGCandidates>, effective: usize, cost_model: &dyn CostModel, context: &CompositionContext, @@ -5522,7 +5626,7 @@ fn composition_options<'a>( continue; }; let already_committed = context.committed_child.get(&child_ptr).copied(); - let cost = |summary: &SummaryNode, shared: bool| { + let cost = |summary: &OperatorNode, shared: bool| { let request = ExactCompositionCostRequest { target: &group.target, composition, @@ -5558,10 +5662,10 @@ fn composition_options<'a>( if !is_automatically_selectable(child_candidate, cost_model) { continue; } - let Replacement::Summary(summary) = &child_candidate.replacement else { + let Replacement::SubDAG(summary) = &child_candidate.replacement else { continue; }; - if !composition.accepts_child(summary) { + if is_logical_rewrite(summary) || !composition.accepts_child(summary) { continue; } let Some(prepared) = plans.iter().find(|p| { @@ -5670,8 +5774,8 @@ impl CandidateLogicalASAPDAGs { let topo = topological_order(&self.order, &dag); let mut effective_uses = dag.external_root_uses.clone(); - let mut chosen_share: HashMap<*const QueryExpr, ShareDecision> = HashMap::new(); - let mut groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'_>> = HashMap::new(); + let mut chosen_share: HashMap<*const OperatorNode, ShareDecision> = HashMap::new(); + let mut groups: HashMap<*const OperatorNode, TargetSubDAGSelection<'_>> = HashMap::new(); let mut context = CompositionContext::default(); for ptr in &topo { @@ -5898,8 +6002,8 @@ impl CandidateLogicalASAPDAGs { // Record the maintained summary this site's bound candidate // builds, for a child that may compose an `ValueOperationAtIngestionTime` // beneath it. - if let (Some(Replacement::Summary(node)), QueryExpr::Aggregate { child, .. }) = - (chosen.map(|c| &c.replacement), group.target.as_ref()) + if let (Some(Replacement::SubDAG(node)), Some(NonASAPOp::Aggregate { child, .. })) = + (chosen.map(|c| &c.replacement), group.target.non_asap()) { if let Some(summary) = maintained_summary(node) { context @@ -5911,7 +6015,7 @@ impl CandidateLogicalASAPDAGs { let outgoing_multiplier = multiplier(*ptr, &effective_uses, &chosen_share); match chosen { Some(ReplacementSubDAG { - replacement: Replacement::Rewrite(source), + replacement: Replacement::SubDAG(source), provenance: ReplacementProvenance::AccuracyReconciliation, .. }) => { @@ -5924,8 +6028,10 @@ impl CandidateLogicalASAPDAGs { } _ => { let selected_rewrite = match chosen.map(|candidate| &candidate.replacement) { - Some(Replacement::Rewrite(rewrite)) => rewrite, - Some(Replacement::Summary(_) | Replacement::ExactComposition(_)) | None => { + Some(Replacement::SubDAG(rewrite)) if is_logical_rewrite(rewrite) => { + rewrite + } + Some(Replacement::SubDAG(_) | Replacement::ExactComposition(_)) | None => { &group.target } }; @@ -5991,9 +6097,9 @@ fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn C /// ancestor sits anywhere on the path from a root to a site — see the /// module docs' "Whole-plan (cross-group) selection" section. fn multiplier( - parent_ptr: *const QueryExpr, - effective_uses: &HashMap<*const QueryExpr, usize>, - chosen_share: &HashMap<*const QueryExpr, ShareDecision>, + parent_ptr: *const OperatorNode, + effective_uses: &HashMap<*const OperatorNode, usize>, + chosen_share: &HashMap<*const OperatorNode, ShareDecision>, ) -> usize { let effective = *effective_uses.get(&parent_ptr).expect( "topological_order guarantees a parent is processed (and its effective_consumer_count \ @@ -6017,7 +6123,7 @@ fn cse_candidate_pair( for candidate in &group.candidates { match candidate.provenance { ReplacementProvenance::CseShare => { - let Replacement::Rewrite(rc) = &candidate.replacement else { + let Replacement::SubDAG(rc) = &candidate.replacement else { return None; }; if !Rc::ptr_eq(rc, &group.target) || share.replace(candidate).is_some() { @@ -6025,7 +6131,7 @@ fn cse_candidate_pair( } } ReplacementProvenance::CseRecompute => { - let Replacement::Rewrite(rc) = &candidate.replacement else { + let Replacement::SubDAG(rc) = &candidate.replacement else { return None; }; if Rc::ptr_eq(rc, &group.target) @@ -6103,26 +6209,25 @@ fn pick_shared_sub_dag_candidate( /// The parent/child structure [`CandidateLogicalASAPDAGs::global_selection`]'s DP walks — /// built separately from [`discover_targets`]'s own `order`/`nodes`/`counts` /// maps (which only track *aggregate* reference counts, not per-parent -/// breakdown or direction) rather than extending that already-reviewed, -/// already-tested pass. Same "small duplicated traversal over reshaping -/// proven code" call as [`is_shared_subtree_group`]. +/// breakdown or direction). Selection needs per-parent edge counts to +/// distinguish shared producers from repeated uses within one consumer. struct ReferenceDAG { /// child ptr -> `(parent ptr, edge count from that one parent)`, for /// every direct operator-child edge in the relational-skeleton scope /// [`walk_children`] itself uses (an edge count above 1 happens when /// one parent references the same child from two different fields, /// e.g. a `Join`'s `left`/`right` both being the same `Rc`). - parents_of: HashMap<*const QueryExpr, Vec<(*const QueryExpr, usize)>>, + parents_of: HashMap<*const OperatorNode, Vec<(*const OperatorNode, usize)>>, /// parent ptr -> every distinct child ptr it directly references — the /// reverse of `parents_of`, for [`topological_order`]'s Kahn's-algorithm /// traversal. - children_of: HashMap<*const QueryExpr, Vec<*const QueryExpr>>, + children_of: HashMap<*const OperatorNode, Vec<*const OperatorNode>>, /// How many of the workload's own `roots` point directly at each node — /// a node's "external" use. Nothing inside the DAG decides this (it /// isn't a reference from another discovered site), so it's never /// subject to any ancestor's Share/Recompute choice — it's the base /// case [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence starts from. - external_root_uses: HashMap<*const QueryExpr, usize>, + external_root_uses: HashMap<*const OperatorNode, usize>, } /// Build an ordering DAG containing every edge that could be selected: @@ -6145,7 +6250,10 @@ fn reference_dag(space: &CandidateLogicalASAPDAGs) -> ReferenceDAG { let group = &space.groups[ptr]; record_possible_edges(*ptr, &group.target, &mut dag); for candidate in &group.candidates { - if let Replacement::Rewrite(rewrite) = &candidate.replacement { + if let Replacement::SubDAG(rewrite) = &candidate.replacement { + if !is_logical_rewrite(rewrite) { + continue; + } if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut dag); } else { @@ -6161,8 +6269,8 @@ fn reference_dag(space: &CandidateLogicalASAPDAGs) -> ReferenceDAG { /// [`ReferenceDAG`]'s fields), retaining the greatest multiplicity seen /// when the target and alternative rewrites expose the same edge. fn add_edge( - parent_ptr: *const QueryExpr, - child_ptr: *const QueryExpr, + parent_ptr: *const OperatorNode, + child_ptr: *const OperatorNode, edge_count: usize, dag: &mut ReferenceDAG, ) { @@ -6177,7 +6285,11 @@ fn add_edge( } } -fn record_possible_edges(parent_ptr: *const QueryExpr, node: &QueryExpr, dag: &mut ReferenceDAG) { +fn record_possible_edges( + parent_ptr: *const OperatorNode, + node: &OperatorNode, + dag: &mut ReferenceDAG, +) { for (child_ptr, edge_count) in direct_child_counts(node) { add_edge(parent_ptr, child_ptr, edge_count, dag); } @@ -6185,8 +6297,8 @@ fn record_possible_edges(parent_ptr: *const QueryExpr, node: &QueryExpr, dag: &m /// Direct relational-skeleton children and their edge multiplicities. /// `Concat` is transparent, matching [`walk_children`]'s site scope. -fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { - fn push(children: &mut Vec<(*const QueryExpr, usize)>, child: &Rc) { +fn direct_child_counts(node: &OperatorNode) -> Vec<(*const OperatorNode, usize)> { + fn push(children: &mut Vec<(*const OperatorNode, usize)>, child: &Rc) { let ptr = Rc::as_ptr(child); match children.iter_mut().find(|(existing, _)| *existing == ptr) { Some((_, count)) => *count += 1, @@ -6194,57 +6306,19 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { } } - fn collect(node: &QueryExpr, children: &mut Vec<(*const QueryExpr, usize)>) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { - push(children, c); - } - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => { - push(children, child); - } - Concat { - children: concat_children, - .. - } => { - for c in concat_children { - collect(c, children); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - push(children, left); - push(children, right); - } - BinaryOp { lhs, rhs, .. } => { - push(children, lhs); - push(children, rhs); + fn collect(node: &OperatorNode, children: &mut Vec<(*const OperatorNode, usize)>) { + if let Some(NonASAPOp::Concat { + children: concat_children, + .. + }) = node.non_asap() + { + for c in concat_children { + collect(c, children); } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} + return; + } + for child in node.children() { + push(children, child); } } @@ -6260,14 +6334,17 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { /// different root paths can have a parent that's discovered *after* it (see /// this function's own test for a worked diamond example), which is exactly /// backwards for [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence. -fn topological_order(order: &[*const QueryExpr], dag: &ReferenceDAG) -> Vec<*const QueryExpr> { - let mut in_degree: HashMap<*const QueryExpr, usize> = HashMap::new(); +fn topological_order( + order: &[*const OperatorNode], + dag: &ReferenceDAG, +) -> Vec<*const OperatorNode> { + let mut in_degree: HashMap<*const OperatorNode, usize> = HashMap::new(); for ptr in order { let degree = dag.parents_of.get(ptr).map(Vec::len).unwrap_or(0); in_degree.insert(*ptr, degree); } - let mut queue: VecDeque<*const QueryExpr> = order + let mut queue: VecDeque<*const OperatorNode> = order .iter() .copied() .filter(|ptr| in_degree[ptr] == 0) @@ -6291,9 +6368,9 @@ fn topological_order(order: &[*const QueryExpr], dag: &ReferenceDAG) -> Vec<*con assert_eq!( topo.len(), order.len(), - "topological_order: the discovered-site reference DAG has a cycle — every QueryExpr \ - node is built from Rc children, which can't form one, so this indicates a bug in \ - reference_dag rather than a real cyclic workload", + "topological_order: the discovered-site reference dag has a cycle — every \ + OperatorNode is built from Rc children, which can't form one, so this indicates a bug \ + in reference_dag rather than a real cyclic workload", ); topo } @@ -6319,13 +6396,13 @@ fn topological_order(order: &[*const QueryExpr], dag: &ReferenceDAG) -> Vec<*con /// the target itself) exactly like [`SharedSubDAGStrategy`], so it belongs /// in this list rather than being derived per-workload the way /// [`RollupStrategy`] is. Rewriting `avg` into `sum`/`count` upfront is what -/// lets [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] see a +/// lets [`ASAPStrategies`] and [`SharedSubDAGStrategy`] see a /// mergeable accumulator to sketch or share at all — see that module's own /// doc comment for why a bare `avg` node otherwise never becomes a /// [`ReplacementStrategy`] target for anything. pub fn default_strategies() -> Vec> { vec![ - Box::new(SketchAlgorithmStrategy::default_cost_model()), + Box::new(ASAPStrategies::default_cost_model()), Box::new(HydraGroupingStrategy::default_cost_model()), Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::AvgToSumOverCountStrategy), @@ -6333,14 +6410,14 @@ pub fn default_strategies() -> Vec> { ] } -/// Like [`default_strategies`], but [`SketchAlgorithmStrategy`] ranks/binds via +/// Like [`default_strategies`], but [`ASAPStrategies`] ranks/binds via /// `cost_model` instead of the built-in [`DefaultCostModel`] — the same -/// customization point [`SketchAlgorithmStrategy::new`] itself offers. +/// customization point [`ASAPStrategies::new`] itself offers. pub fn default_strategies_with<'a>( cost_model: &'a dyn CostModel, ) -> Vec> { vec![ - Box::new(SketchAlgorithmStrategy::new(cost_model)), + Box::new(ASAPStrategies::new(cost_model)), Box::new(HydraGroupingStrategy::new(cost_model)), Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::SemanticEquivalentRewriteStrategy), @@ -6350,23 +6427,21 @@ pub fn default_strategies_with<'a>( /// Default context-free strategies with both deployment costing and typed /// planning-time accuracy evidence. This is the production counterpart of -/// constructing [`SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence`] and +/// constructing [`ASAPStrategies::new_with_planning_inputs_and_evidence`] and /// [`HydraGroupingStrategy::new_with_planning_inputs_and_evidence`] separately. pub fn default_strategies_with_evidence<'a>( cost_model: &'a dyn CostModel, evidence: &'a dyn AccuracyEvidenceProvider, ) -> Vec> { vec![ + Box::new(ASAPStrategies::new_with_planning_inputs_and_evidence( + cost_model, + &DEFAULT_ACCURACY_MODEL, + &DEFAULT_ALLOCATOR, + evidence, + )), Box::new( - SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - cost_model, - &DEFAULT_ACCURACY_MODEL, - &DEFAULT_ALLOCATOR, - evidence, - ), - ), - Box::new( - HydraGroupingStrategy::new_with_planning_inputs_and_evidence( + HydraGroupingStrategy::new_with_planning_inputs_and_evidence( cost_model, &DEFAULT_ACCURACY_MODEL, &DEFAULT_ALLOCATOR, @@ -6384,12 +6459,12 @@ pub fn default_strategies_with_evidence<'a>( /// Search a whole workload's pre-ASAP roots for every candidate replacement /// [`default_strategies`] can find, deduped into a [`CandidateLogicalASAPDAGs`]. Candidate /// *generation* uses the built-in [`DefaultCostModel`] (via -/// [`default_strategies`], the same way [`SketchAlgorithmStrategy::default_cost_model`] +/// [`default_strategies`], the same way [`ASAPStrategies::default_cost_model`] /// does); call [`CandidateLogicalASAPDAGs::cost_sorted`] on the result for the final /// `sorted_by(cost_model)` step. Use [`search_workload_with`] to plug in a /// custom strategy set (e.g. built via [`default_strategies_with`] for a /// deployment-specific [`CostModel`]). -pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalASAPDAGs { +pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalASAPDAGs { search_workload_with(roots, &default_strategies()) } @@ -6409,10 +6484,10 @@ pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalA /// section). Deduping candidate plans this way needs no /// [`CostModel`] at all — that only enters at two well-defined points: each /// [`ReplacementStrategy`] in `strategies` may already carry its own (e.g. -/// [`SketchAlgorithmStrategy::new`]'s), and [`CandidateLogicalASAPDAGs::cost_sorted`]'s final +/// [`ASAPStrategies::new`]'s), and [`CandidateLogicalASAPDAGs::cost_sorted`]'s final /// ranking step takes one explicitly. pub fn search_workload_with<'s, Id>( - roots: Vec<(Id, Rc)>, + roots: Vec<(Id, Rc)>, strategies: &[Box], ) -> CandidateLogicalASAPDAGs { let mut space = search_cse_workload_with(cse_workload(roots), strategies); @@ -6423,7 +6498,7 @@ pub fn search_workload_with<'s, Id>( /// [`search_workload_with`] plus a per-root end-to-end `AccuracyTarget` /// (issue #172) — the workload's `QueryRequirements.accuracy`, threaded /// alongside each root. After the search, every root that carries a target -/// has its group's bound [`Replacement::Summary`] candidates checked with +/// has its group's bound-summary [`Replacement::SubDAG`] candidates checked with /// `accuracy_model`'s [`AccuracyModel::satisfies`]: a candidate whose /// guarantee is fully known and misses the target is moved from /// [`TargetSubDAGCandidates::candidates`] to [`TargetSubDAGCandidates::rejected`] *before* @@ -6431,16 +6506,16 @@ pub fn search_workload_with<'s, Id>( /// group. A constructible candidate with unknown accuracy remains visible for /// downstream review under an approximate target, but default whole-plan /// selection does not commit it. An exact target cannot accept an unknown -/// approximate summary. A `KeepPreAsap` candidate is +/// approximate summary. A kept pre-ASAP candidate is /// exact and always survives — the raw/pre-ASAP alternative is what an -/// unsatisfiable root keeps. Logical [`Replacement::Rewrite`] candidates -/// are not bound values and are left alone; the targets *inside* a rewrite -/// are their own groups. +/// unsatisfiable root keeps. Logical-rewrite [`Replacement::SubDAG`] +/// candidates are not bound values and are left alone; the targets *inside* +/// a rewrite are their own groups. /// /// Precedence against per-node `AggIntent.accuracy` is documented in /// [`crate::accuracy`]'s module docs. pub fn search_workload_with_targets<'s, Id>( - roots: Vec<(Id, Rc, Option)>, + roots: Vec<(Id, Rc, Option)>, strategies: &[Box], accuracy_model: &dyn AccuracyModel, ) -> CandidateLogicalASAPDAGs { @@ -6454,7 +6529,7 @@ pub fn search_workload_with_targets<'s, Id>( .collect(); let mut space = search_cse_workload_with(cse_workload(roots), strategies); // `cse_workload` preserves root order, so targets zip by position. - let root_ptrs: Vec<(*const QueryExpr, AccuracyTarget)> = space + let root_ptrs: Vec<(*const OperatorNode, AccuracyTarget)> = space .roots .iter() .zip(targets) @@ -6496,11 +6571,11 @@ pub fn search_workload_with_targets<'s, Id>( .candidates .drain(..) .partition(|candidate| match &candidate.replacement { - Replacement::Summary(node) => node.guarantee.as_ref().map_or_else( + Replacement::SubDAG(node) if is_logical_rewrite(node) => true, + Replacement::SubDAG(node) => node.guarantee.as_ref().map_or_else( || !matches!(target, AccuracyTarget::Exact), |g| accuracy_model.satisfies(&g.optimistic_floor(), &target), ), - Replacement::Rewrite(_) => true, // A composition's guarantee depends on the concrete child; // prepare_compositions checks those pairs after all roots. Replacement::ExactComposition(_) => true, @@ -6508,7 +6583,7 @@ pub fn search_workload_with_targets<'s, Id>( group.candidates = legal; group.rejected.extend(illegal.into_iter().map(|candidate| { let (metric, bound, failure_probability) = match &candidate.replacement { - Replacement::Summary(node) => node + Replacement::SubDAG(node) => node .guarantee .as_ref() .map(|g| { @@ -6523,7 +6598,6 @@ pub fn search_workload_with_targets<'s, Id>( None, None, )), - Replacement::Rewrite(_) => unreachable!("rewrites are never rejected here"), Replacement::ExactComposition(_) => ( asap_types::post_asap::ErrorMetric::AbsoluteValue, None, @@ -6548,22 +6622,22 @@ pub fn search_workload_with_targets<'s, Id>( /// The strictest accuracy among `siblings` that read the same summary input /// as `root` — same child, grouping and filters, and the same intent apart -/// from its accuracy (and a quantile's rank, a readout parameter) — when +/// from its accuracy (and a quantile's rank, a evaluation parameter) — when /// stricter than `root`'s own. One summary sized for the strictest consumer /// serves every sibling: #509's summary-capability rule. fn strictest_sibling_accuracy( - root: &QueryExpr, - siblings: &[Rc], + root: &OperatorNode, + siblings: &[Rc], ) -> Option { fn approximate(intent: &AggIntent) -> Option<&AccuracyTarget> { accuracy_target(intent).filter(|accuracy| !matches!(accuracy, AccuracyTarget::Exact)) } - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, filters, child, .. - } = root + }) = root.non_asap() else { return None; }; @@ -6571,12 +6645,12 @@ fn strictest_sibling_accuracy( let own = accuracy_budget(approximate(intent)?); let (mut eps, mut delta) = own; for sibling in siblings { - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: sibling_reduction, filters: sibling_filters, child: sibling_child, .. - } = sibling.as_ref() + }) = sibling.non_asap() else { continue; }; @@ -6614,48 +6688,45 @@ fn strictest_sibling_accuracy( } } -fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { - // `share_common_sub_dags` wants owned `QueryExpr`s, not already-`Rc` - // roots — the same `Rc::try_unwrap`-with-clone-fallback pattern - // `asap_types::pre_asap::cse::intern_child` itself uses to recover an - // owned node without cloning in the common (uniquely-owned) case. - let owned_roots: Vec<(Id, QueryExpr)> = roots - .into_iter() - .map(|(id, rc)| { - let expr = Rc::try_unwrap(rc).unwrap_or_else(|shared| (*shared).clone()); - (id, expr) - }) - .collect(); - share_common_sub_dags(owned_roots) +fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { + share_common_sub_dags(roots) } fn search_cse_workload_with<'s, Id>( - cse_roots: Vec<(Id, Rc)>, + cse_roots: Vec<(Id, Rc)>, strategies: &[Box], ) -> CandidateLogicalASAPDAGs { + for (_, root) in &cse_roots { + assert!( + !root.contains_asap(), + "search_workload: a workload root already contains an ASAP operator \ + ({}); replacement search takes the front end's pre-ASAP DAG only", + root.operator.kind_name() + ); + } let mut order = Vec::new(); let mut nodes = HashMap::new(); - let mut counts: HashMap<*const QueryExpr, usize> = HashMap::new(); + let mut counts: HashMap<*const OperatorNode, usize> = HashMap::new(); discover_targets(&cse_roots, &mut order, &mut nodes, &mut counts); - let siblings: Vec> = order + let siblings: Vec> = order .iter() .filter_map(|ptr| { let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Aggregate { .. }).then(|| Rc::clone(node)) + matches!(node.non_asap(), Some(NonASAPOp::Aggregate { .. })).then(|| Rc::clone(node)) }) .collect(); let rollup_strategy = RollupStrategy::new(&siblings); let accuracy_reconciliation_strategy = AccuracyReconciliationStrategy::new(&siblings); - let limits: Vec> = order + let limits: Vec> = order .iter() .filter_map(|ptr| { let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Limit { .. }).then(|| Rc::clone(node)) + matches!(node.non_asap(), Some(NonASAPOp::Limit { .. })).then(|| Rc::clone(node)) }) .collect(); let topk_reuse_strategy = TopKLimitReuseStrategy::new(&limits); - let mut groups: HashMap<*const QueryExpr, TargetSubDAGCandidates> = HashMap::new(); + let mut groups: HashMap<*const OperatorNode, TargetSubDAGCandidates> = HashMap::new(); for ptr in &order { groups.insert( *ptr, @@ -6678,7 +6749,7 @@ fn search_cse_workload_with<'s, Id>( "search_workload: fixpoint search did not converge within {MAX_SEARCH_ITERATIONS} \ rounds — a registered ReplacementStrategy's Replacement::Rewrite candidates keep \ exposing new, never-before-seen descendant structure every round. \ - SketchAlgorithmStrategy/SharedSubDAGStrategy never do this (see replacement.rs's \ + ASAPStrategies/SharedSubDAGStrategy never do this (see replacement.rs's \ module docs' \"Termination\" section); check any custom strategies passed to \ search_workload_with.", ); @@ -6741,8 +6812,10 @@ fn search_cse_workload_with<'s, Id>( } for candidate in &proposed { - if let Replacement::Rewrite(rc) = &candidate.replacement { - discover_new_descendant_targets(rc, &mut order, &mut nodes, &mut counts); + if let Replacement::SubDAG(rc) = &candidate.replacement { + if is_logical_rewrite(rc) { + discover_new_descendant_targets(rc, &mut order, &mut nodes, &mut counts); + } } } @@ -6782,10 +6855,11 @@ fn search_cse_workload_with<'s, Id>( /// ancestor is recomputed. We only do this when an ordinary repeated group /// proves that `SharedSubDAGStrategy` is part of this search's strategy set. fn add_effective_count_cse_candidates( - order: &[*const QueryExpr], - groups: &mut HashMap<*const QueryExpr, TargetSubDAGCandidates>, + order: &[*const OperatorNode], + groups: &mut HashMap<*const OperatorNode, TargetSubDAGCandidates>, ) { - let mut possible_children: HashMap<*const QueryExpr, Vec<*const QueryExpr>> = HashMap::new(); + let mut possible_children: HashMap<*const OperatorNode, Vec<*const OperatorNode>> = + HashMap::new(); for ptr in order { let group = &groups[ptr]; let children = possible_children.entry(*ptr).or_default(); @@ -6795,7 +6869,10 @@ fn add_effective_count_cse_candidates( } } for candidate in &group.candidates { - if let Replacement::Rewrite(rewrite) = &candidate.replacement { + if let Replacement::SubDAG(rewrite) = &candidate.replacement { + if !is_logical_rewrite(rewrite) { + continue; + } for (child, _) in direct_child_counts(rewrite) { if !children.contains(&child) { children.push(child); @@ -6853,10 +6930,10 @@ fn add_effective_count_cse_candidates( /// `Rc` and its real `consumer_count` — see the module docs' "Where /// `TargetSubDAG` discovery comes from" section for the full rationale. fn discover_targets( - roots: &[(Id, Rc)], - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + roots: &[(Id, Rc)], + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { for (_, root) in roots { walk(root, order, nodes, counts); @@ -6865,17 +6942,17 @@ fn discover_targets( /// Scan `candidate`'s **children** (deliberately never `candidate`'s own /// top-level pointer — see the module docs' "Termination" section: a -/// [`Replacement::Rewrite`]'s value is an alternative *for* the target that +/// logical rewrite's value is an alternative *for* the target that /// proposed it, never a new target of its own) for any `Rc` not already /// known, appending each to `order`/`nodes`/`counts` so /// [`search_workload_with`]'s next round processes it. A no-op when every /// child is already known — the case both shipped strategies always produce /// (see that section). fn discover_new_descendant_targets( - candidate: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + candidate: &Rc, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { walk_children(candidate, order, nodes, counts); } @@ -6883,10 +6960,10 @@ fn discover_new_descendant_targets( /// Visit `node`: count this occurrence, and — the first time this exact /// `Rc` is seen — record it as a target and recurse into its children. fn walk( - node: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + node: &Rc, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { let ptr = Rc::as_ptr(node); let already_visited = counts.contains_key(&ptr); @@ -6898,61 +6975,25 @@ fn walk( } } -/// `node`'s own **relational-skeleton** operator children — the same scope -/// `asap_types::pre_asap::cse::share_common_sub_dags`/`rebuild_children` -/// itself uses (see that module's "Algorithm" section) and -/// `tests::count_consumers` mirrors for its own fixtures. Exhaustive over -/// every `QueryExpr` variant: a new variant fails to compile here until this -/// match is extended too. +/// `node`'s own operator children ([`OperatorNode::children`]: operator +/// inputs plus the operator nodes its scalar expressions read), the same +/// scope `asap_types::ir::cse::share_common_sub_dags` itself uses and +/// `tests::count_consumers` mirrors for its own fixtures. `Concat` is +/// transparent: its branches are walked in place of it. fn walk_children( - node: &QueryExpr, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, + node: &OperatorNode, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, ) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, order, nodes, counts), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => walk(child, order, nodes, counts), - Concat { children, .. } => { - for c in children { - walk_children(c, order, nodes, counts); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - walk(left, order, nodes, counts); - walk(right, order, nodes, counts); - } - BinaryOp { lhs, rhs, .. } => { - walk(lhs, order, nodes, counts); - walk(rhs, order, nodes, counts); - } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} + if let Some(NonASAPOp::Concat { children, .. }) = node.non_asap() { + for c in children { + walk_children(c, order, nodes, counts); + } + return; + } + for child in node.children() { + walk(child, order, nodes, counts); } } @@ -6961,17 +7002,19 @@ mod tests { use super::*; use crate::accuracy::PropagationStats; use crate::cost_model::Cost; - use crate::test_support::lower_promql; + use crate::test_support::{agg, agg_per_entity, lower_promql, metric_scan, timed}; + use asap_types::ir::operator_properties::{Reduction as ReductionTy, Source}; + use asap_types::ir::TimeRangeKind; use asap_types::pre_asap::agg_intent::{ agg_is_exact, default_cardinality, default_quantile, MathFunc, TimeFunc, }; - use asap_types::pre_asap::query_expr::{Reduction as ReductionTy, Source}; use asap_types::pre_asap::schema::{DataType, Field, Schema as SchemaTy}; + use asap_types::types::AccuracyTarget; use std::collections::HashMap; // Candidate shape without execution timing: what is computed, not where. - fn timing_free_shape(node: &Rc) -> serde_json::Value { + fn timing_free_shape(node: &Rc) -> serde_json::Value { fn strip(value: &mut serde_json::Value) { match value { serde_json::Value::Object(fields) => { @@ -6982,9 +7025,10 @@ mod tests { _ => {} } } - let mut shape = - serde_json::to_value(asap_types::post_asap::compile_post_asap_dag(node).unwrap()) - .unwrap(); + let mut shape = serde_json::to_value( + asap_types::ir::physical_export::compile_physical_asap_dag(&timed(node)).unwrap(), + ) + .unwrap(); strip(&mut shape); shape } @@ -6996,7 +7040,7 @@ mod tests { ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), ] { - let root = Rc::new(lower_promql(query, accuracy)); + let root = lower_promql(query, accuracy); let inventory = search_workload(vec![(0usize, root)]) .enumerate_candidate_dags(4096) .unwrap(); @@ -7011,37 +7055,32 @@ mod tests { } } - // Grouped Sum over Rate readouts stays a summary state in the inventory, + // Grouped Sum over Rate evaluations stays a summary state in the inventory, // so lifecycle assignment can place it in precompute or at query time. #[test] fn grouped_rate_sum_inventory_keeps_sum_state_for_lifecycle_placement() { - let root = Rc::new(lower_promql( - "sum by(job)(rate(m[1m]))", - AccuracyTarget::Exact, - )); + let root = lower_promql("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact); let inventory = search_workload(vec![(0usize, root)]) .enumerate_candidate_dags(4096) .unwrap(); - let is_exact = |node: &SummaryNode, kind: ExactKind| { - matches!(&node.expr, SummaryExpr::SummaryAgg { + let is_exact = |node: &OperatorNode, kind: ExactKind| { + matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. - } if *k == kind) + }) if *k == kind) }; assert!(inventory.candidates.iter().any(|forest| { - let SummaryExpr::ValueOperation { child: sum, .. } = &forest[0].1.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: sum }) = &forest[0].1.operator else { return false; }; - let SummaryExpr::SummaryAgg { child: rate, .. } = &sum.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child: rate, .. }) = &sum.operator else { return false; }; is_exact(sum, ExactKind::Sum) - && matches!(&rate.expr, SummaryExpr::ValueOperation { - child, operation: ValueOperation::FinalizeExactAccumulator, .. - } if is_exact(child, ExactKind::Rate)) + && matches!(&rate.operator, Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_exact(child, ExactKind::Rate)) })); } - // Every exposed query result has a readout; internal accumulator frontiers stay states. + // Every exposed query result has a evaluation; internal accumulator frontiers stay states. #[test] fn query_candidate_roots_do_not_leak_exact_accumulator_state() { for query in [ @@ -7049,13 +7088,13 @@ mod tests { "sum by(job)(m)", "sum_over_time(m[1m])", ] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact)); + let root = lower_promql(query, AccuracyTarget::Exact); let space = search_workload(vec![(0usize, root.clone())]); let inventory = space.enumerate_candidate_dags(4096).unwrap(); assert!(!inventory.candidates.is_empty()); - let strategy = SketchAlgorithmStrategy::new(&DefaultCostModel); + let strategy = ASAPStrategies::new(&DefaultCostModel); for candidate in strategy.propose(&TargetSubDAG::new(&root)).candidates { - if let Replacement::Summary(node) = candidate.replacement { + if let Replacement::SubDAG(node) = candidate.replacement { let output = finalize_query_candidate(node, &root).unwrap(); assert!( output @@ -7092,7 +7131,7 @@ mod tests { #[test] fn unpriced_inventory_retains_quantile_families_and_raw_execution() { - let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); + let query = agg(vec![2], default_quantile(0.9), metric_scan(&["job"])); let space = search_workload(vec![(0usize, query)]); let inventory = space.enumerate_candidate_dags(4096).unwrap(); let roots = inventory @@ -7105,7 +7144,7 @@ mod tests { assert!(inventory .candidates .iter() - .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); + .any(|forest| !forest[0].1.contains_asap())); } // Independent roots must not require materializing their Cartesian product. @@ -7115,11 +7154,11 @@ mod tests { .map(|id| { ( id, - Rc::new(agg( + agg( vec![2], default_quantile((id + 1) as f64 / 25.0), metric_scan(&["job"]), - )), + ), ) }) .collect(); @@ -7141,7 +7180,7 @@ mod tests { assert!(inventory .candidates .iter() - .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); + .any(|forest| !forest[0].1.contains_asap())); } assert!(space.enumerate_candidate_dags_for_root(&24, 4096).is_err()); assert!(space.enumerate_candidate_dags_for_root(&0, 0).is_err()); @@ -7154,11 +7193,11 @@ mod tests { .map(|id| { ( id, - Rc::new(agg( + agg( vec![2], default_quantile(0.5 + id as f64 * 0.4), metric_scan(&["job"]), - )), + ), ) }) .collect(); @@ -7180,30 +7219,31 @@ mod tests { #[test] fn inventory_budget_never_returns_a_silent_partial_search() { - let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); + let query = agg(vec![2], default_quantile(0.9), metric_scan(&["job"])); let space = search_workload(vec![(0usize, query)]); assert!(space.enumerate_candidate_dags(0).is_err()); } fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left)), + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(left)), op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right)), - })) + right: Box::new(ScalarExpr::Column(right)), + semantics: asap_types::ir::ExprSemantics::Sql, + }) } // Finite samples can overflow a sum although their native average is finite. #[test] fn temporal_average_requires_finite_division_guard() { - let root = Rc::new(lower_promql("avg_over_time(a[5m])", AccuracyTarget::Exact)); + let root = lower_promql("avg_over_time(a[5m])", AccuracyTarget::Exact); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let operator = candidates .iter() .find_map(|c| match &c.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::BinaryOp { operator, .. } => Some(operator), + Replacement::SubDAG(node) => match &node.operator { + Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) => Some(operator), _ => None, }, _ => None, @@ -7221,20 +7261,20 @@ mod tests { // Approximate requests also admit exact temporal ranking candidates. #[test] fn approximate_temporal_topk_admits_exact_maintained_values() { - let root = Rc::new(lower_promql( + let root = lower_promql( "topk by(job)(1,count_over_time(a[5m]))", AccuracyTarget::EpsilonDelta { epsilon: 0.01, delta: 0.01, }, - )); + ); let planning_inputs = CandidatePlanningInputs::with_default_accuracy(&crate::cost_model::DefaultCostModel); let node = exact_topk_over_temporal_values(&root, planning_inputs) .unwrap() .expect("exact ranking is legal for an approximate request"); assert!(node.guarantee.as_ref().unwrap().is_exact()); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); + crate::test_support::time_and_export(&node).unwrap(); } // Exact Top-K consumes the Planner's maintained temporal values. @@ -7244,7 +7284,7 @@ mod tests { "topk(5, sum_over_time(a[5m]))", "topk by(job)(5, count_over_time(a[5m]))", ] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact)); + let root = lower_promql(query, AccuracyTarget::Exact); let planning_inputs = CandidatePlanningInputs::with_default_accuracy( &crate::cost_model::DefaultCostModel, ); @@ -7252,29 +7292,21 @@ mod tests { .unwrap() .expect("exact Top-K candidate"); assert!(node.guarantee.as_ref().unwrap().is_exact()); - let SummaryExpr::ValueOperation { + let Operator::NonASAP(NonASAPOp::Limit { child: sorted, - operation: - ValueOperation::Limit { - n, - offset, - partition_by, - }, - .. - } = &node.expr + n, + offset, + partition_by, + }) = &node.operator else { panic!("temporal TopK must compose Sort and Limit"); }; - assert_eq!((*n, *offset), (5, 0)); - let SummaryExpr::ValueOperation { - operation: - ValueOperation::Sort { - keys, - partition_by: sort_groups, - }, + assert_eq!((*n, *offset), (Some(5), 0)); + let Operator::NonASAP(NonASAPOp::Sort { + keys, + partition_by: sort_groups, child: values, - .. - } = &sorted.expr + }) = &sorted.operator else { panic!("Limit must consume sorted temporal values"); }; @@ -7286,7 +7318,7 @@ mod tests { assert_eq!(keys.len(), 1); assert!(!keys[0].ascending); assert_eq!(node.schema, values.schema); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); + crate::test_support::time_and_export(&node).unwrap(); } } @@ -7297,7 +7329,7 @@ mod tests { impl AccuracyEvidenceProvider for Domain { fn quantile_input_domain( &self, - _: &QueryExpr, + _: &OperatorNode, ) -> Option { Some(crate::accuracy::QuantileInputDomain { lower: 1.0, @@ -7319,7 +7351,7 @@ mod tests { "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", "quantile_over_time(0.5,a[5m]) / avg_over_time(a[5m])", ] { - let root = Rc::new(lower_promql(query, target.clone())); + let root = lower_promql(query, target.clone()); let node = realize_binary(&root, inputs, Some(&target)) .unwrap() .expect("bounded ratio candidate"); @@ -7334,10 +7366,10 @@ mod tests { epsilon: 0.01, delta: 0.01, }; - let root = Rc::new(lower_promql( + let root = lower_promql( "quantile_over_time(0.5,a[5m]) / quantile_over_time(0.9,a[5m])", target.clone(), - )); + ); let planning_inputs = CandidatePlanningInputs::with_default_accuracy(&crate::cost_model::DefaultCostModel); let candidate = realize_binary(&root, planning_inputs, Some(&target)) @@ -7345,10 +7377,10 @@ mod tests { .expect("direct quantile ratio candidate"); assert!(candidate.guarantee.is_none()); - let other = Rc::new(lower_promql( + let other = lower_promql( "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", target.clone(), - )); + ); assert!(realize_binary(&other, planning_inputs, Some(&target)) .unwrap() .is_none()); @@ -7367,11 +7399,65 @@ mod tests { #[test] fn relational_join_is_exact_only_when_both_inputs_are_exact() { - let exact = ResultGuarantee::exact("test exact input"); - assert!(relational_join_guarantee(Some(&exact), Some(&exact)) - .is_some_and(|guarantee| guarantee.is_exact())); - assert!(relational_join_guarantee(Some(&exact), None).is_none()); - assert!(relational_join_guarantee(None, Some(&exact)).is_none()); + // `relational_join_guarantee` folded into assembly's generic + // "keep the operator, assemble its children" branch: an assembled + // inner equi-`Join` is exact exactly when both assembled inputs are. + let join = |left_intent: AggIntent, right_intent: AggIntent| { + let left = agg(vec![2], left_intent, metric_scan(&["job"])); + let right = agg( + vec![2], + right_intent, + crate::test_support::scan("n", metric_scan(&["job"]).schema.clone()), + ); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: asap_types::ir::operator_properties::JoinKind::Inner, + pred: equi_pred(0, 2), + left, + right, + })) + .unwrap() + }; + let is_exact = |node: &OperatorNode| { + node.guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }; + for (root, both_exact_expected) in [ + ( + join(AggIntent::Sum { col: None }, AggIntent::Sum { col: None }), + true, + ), + ( + join(AggIntent::Sum { col: None }, quantile_eps_intent(0.5, 0.05)), + false, + ), + ] { + let space = search_workload(vec![(0usize, Rc::clone(&root))]); + let assembled = space + .global_selection(&DefaultCostModel) + .assemble_selected_query(&space.roots[0].1) + .unwrap() + .unwrap(); + let Some(NonASAPOp::Join { left, right, .. }) = assembled.non_asap() else { + panic!("the join is kept and its inputs assembled: {assembled:?}"); + }; + assert_eq!( + is_exact(&assembled), + is_exact(left) && is_exact(right), + "join guarantee must be exact iff both inputs are exact" + ); + if both_exact_expected { + assert!(is_exact(&assembled), "exact inputs give an exact join"); + } + } + } + + fn quantile_eps_intent(q: f64, e: f64) -> AggIntent { + AggIntent::Quantile { + col: None, + q, + accuracy: AccuracyTarget::Epsilon(e), + } } fn eps(e: f64) -> AccuracyTarget { @@ -7921,68 +8007,44 @@ mod tests { ); } - // ── SketchAlgorithmStrategy / SharedSubDAGStrategy fixtures ─────────── - - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: SchemaTy::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: ReductionTy::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } + // ── ASAPStrategies / SharedSubDAGStrategy fixtures ─────────── - // ── SketchAlgorithmStrategy ───────────────────────────────────────────── + // ── ASAPStrategies ───────────────────────────────────────────── #[test] fn matches_a_bindable_aggregate() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - assert!(SketchAlgorithmStrategy::default_cost_model().matches(&target)); + assert!(ASAPStrategies::default_cost_model().matches(&target)); } #[test] fn does_not_match_a_multi_intent_or_having_aggregate() { - let strategy = SketchAlgorithmStrategy::default_cost_model(); - - let multi = Rc::new(QueryExpr::Aggregate { - reduction: ReductionTy::by(vec![2]), - measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }); + let strategy = ASAPStrategies::default_cost_model(); + + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: ReductionTy::by(vec![2]), + measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&multi); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); - let mut having_q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut having_q { - *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), - ))); - } - let having_q = Rc::new(having_q); + let having_q = crate::test_support::aggregate( + ReductionTy::by(vec![2]), + vec![default_quantile(0.99)], + vec![], + Some(asap_types::ir::Predicate(ScalarExpr::Literal( + asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true), + ))), + metric_scan(&["job"]), + ); let target = TargetSubDAG::new(&having_q); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); @@ -7990,10 +8052,10 @@ mod tests { #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); - assert!(!SketchAlgorithmStrategy::default_cost_model().matches(&target)); - assert!(SketchAlgorithmStrategy::default_cost_model() + assert!(!ASAPStrategies::default_cost_model().matches(&target)); + assert!(ASAPStrategies::default_cost_model() .replacements(&target) .is_empty()); } @@ -8001,11 +8063,11 @@ mod tests { #[test] fn approximate_quantile_enumerates_every_summary_candidate() { // Quantile's candidate list is [Kll, DDSketch] (summary_candidates) — - // every entry must come back as its own bound SummaryNode candidate, + // every entry must come back as its own bound summary candidate, // not just Kll (the CostModel-ranked head realizations_for_intent commits to). - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); assert_eq!( replacements.len(), 2, @@ -8015,8 +8077,8 @@ mod tests { let kinds: Vec = replacements .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDAG(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8031,14 +8093,14 @@ mod tests { #[test] fn cardinality_epsilon_delta_keeps_unknown_accuracy_candidates() { - let q = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let q = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); let kinds: Vec = replacements .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDAG(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8053,7 +8115,7 @@ mod tests { ] ); - let q = Rc::new(agg( + let q = agg( vec![2], AggIntent::Cardinality { cols: vec![], @@ -8063,13 +8125,13 @@ mod tests { }, }, metric_scan(&["job"]), - )); - let kinds: Vec<_> = SketchAlgorithmStrategy::default_cost_model() + ); + let kinds: Vec<_> = ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&q)) .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDAG(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8094,35 +8156,28 @@ mod tests { q: 0.99, accuracy: AccuracyTarget::Exact, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); assert!(matches!( &replacements[0].replacement, - Replacement::Summary(node) if matches!( - node.expr, - asap_types::post_asap::SummaryExpr::KeepPreAsap(_) - ) + Replacement::SubDAG(node) if !node.contains_asap() )); assert!(replacements[0].rationale.contains("only realization")); } #[test] fn exact_mergeable_intent_yields_exactly_one_accumulator_candidate() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let replacements = ASAPStrategies::default_cost_model().replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); assert!(matches!( &replacements[0].replacement, - Replacement::Summary(node) if matches!( - node.expr, - asap_types::post_asap::SummaryExpr::SummaryAgg { .. } + Replacement::SubDAG(node) if matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { .. }) ) )); } @@ -8149,15 +8204,15 @@ mod tests { #[test] fn custom_cost_model_still_enumerates_every_candidate_not_just_its_own_pick() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); let custom = PreferDDSketch; - let replacements = SketchAlgorithmStrategy::new(&custom).replacements(&target); + let replacements = ASAPStrategies::new(&custom).replacements(&target); let kinds: Vec = replacements .iter() .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { + Replacement::SubDAG(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { panic!("expected a Summary replacement") } }) @@ -8181,9 +8236,9 @@ mod tests { // so this test injects `RankAdditiveModel` to admit the composition // and keep exercising the per-node enumeration property it is about. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); + let outer = agg(vec![], default_quantile(0.99), inner); let target = TargetSubDAG::new(&outer); - let replacements = SketchAlgorithmStrategy::new_with_planning_inputs( + let replacements = ASAPStrategies::new_with_planning_inputs( &DefaultCostModel, &RankAdditiveModel, &EqualSplitAllocator, @@ -8191,9 +8246,9 @@ mod tests { .replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); - assert!(replacements - .iter() - .all(|candidate| { matches!(candidate.replacement, Replacement::Summary(_)) })); + assert!(replacements.iter().all(|candidate| { + matches!(&candidate.replacement, Replacement::SubDAG(n) if n.contains_asap()) + })); // The inner target is still independently enumerated and ranked — // a custom cost model that prefers DDSketch for it is honored, and // nothing about the outer target's choice reaches it. @@ -8201,7 +8256,7 @@ mod tests { vec![("q", Rc::clone(&outer))], &default_strategies_with(&PreferDDSketchViaCostModel), ); - let QueryExpr::Aggregate { child, .. } = space.roots[0].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = space.roots[0].1.non_asap() else { unreachable!() }; let inner_group = space @@ -8211,7 +8266,7 @@ mod tests { .candidates .iter() .filter_map(|c| match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), + Replacement::SubDAG(node) => sketch_kind_of(node), _ => None, }) .collect(); @@ -8225,12 +8280,12 @@ mod tests { /// The `FieldDataType`'s committed `SketchAlgorithm`, from the top /// `SummaryAgg` reachable under a (possibly `SummaryEstimate`-wrapped) /// bound root. - fn summary_family_algorithm(node: &SummaryNode) -> SketchAlgorithm { - match &node.expr { - asap_types::post_asap::SummaryExpr::SummaryEstimate { summary_input, .. } => { + fn summary_family_algorithm(node: &OperatorNode) -> SketchAlgorithm { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { summary_family_algorithm(summary_input) } - asap_types::post_asap::SummaryExpr::SummaryAgg { family, .. } => match family { + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => match family { asap_types::post_asap::FieldDataType::Sketch(kind, _) => kind.algorithm().clone(), other => panic!("expected a Sketch family, got {other:?}"), }, @@ -8242,11 +8297,7 @@ mod tests { #[test] fn does_not_match_a_single_consumer_target() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); assert_eq!(target.consumer_count, 1); assert!(!SharedSubDAGStrategy.matches(&target)); @@ -8255,11 +8306,7 @@ mod tests { #[test] fn two_or_more_consumers_yields_the_share_vs_independent_pair() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::with_consumer_count(&q, 2); assert!(SharedSubDAGStrategy.matches(&target)); @@ -8267,7 +8314,7 @@ mod tests { assert_eq!(replacements.len(), 2, "{replacements:?}"); let shared = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; assert!( @@ -8277,7 +8324,7 @@ mod tests { assert!(replacements[0].rationale.contains("build once and share")); let independent = match &replacements[1].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; assert!( @@ -8293,11 +8340,7 @@ mod tests { #[test] fn three_consumers_are_reported_verbatim_in_both_rationales() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::with_consumer_count(&q, 3); let replacements = SharedSubDAGStrategy.replacements(&target); assert!(replacements[0].rationale.contains('3')); @@ -8307,13 +8350,13 @@ mod tests { /// Builds realistic multi-consumer `TargetSubDAG`s the same way this /// module's own [`discover_targets`]/`walk` does: dedup by `Rc::as_ptr`, /// walking only the relational-skeleton operator children - /// `asap_types::pre_asap::cse::share_common_sub_dags` itself scopes to, + /// `asap_types::ir::cse::share_common_sub_dags` itself scopes to, /// so a shared node nested below another shared node is only ever /// counted at the highest (maximal) point sharing starts. Test-only: /// this module deliberately does not ship a workload-wide discovery /// pass of its own (see the module docs' "Non-goals"). - fn count_consumers(roots: &[Rc]) -> HashMap<*const QueryExpr, usize> { - fn walk(node: &Rc, counts: &mut HashMap<*const QueryExpr, usize>) { + fn count_consumers(roots: &[Rc]) -> HashMap<*const OperatorNode, usize> { + fn walk(node: &Rc, counts: &mut HashMap<*const OperatorNode, usize>) { let ptr = Rc::as_ptr(node); let already_visited = counts.contains_key(&ptr); *counts.entry(ptr).or_insert(0) += 1; @@ -8321,50 +8364,15 @@ mod tests { walk_children(node, counts); } } - fn walk_children(node: &QueryExpr, counts: &mut HashMap<*const QueryExpr, usize>) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, counts), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => walk(child, counts), - Concat { children, .. } => { - for c in children { - walk_children(c, counts); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - walk(left, counts); - walk(right, counts); - } - BinaryOp { lhs, rhs, .. } => { - walk(lhs, counts); - walk(rhs, counts); + fn walk_children(node: &OperatorNode, counts: &mut HashMap<*const OperatorNode, usize>) { + if let Some(NonASAPOp::Concat { children, .. }) = node.non_asap() { + for c in children { + walk_children(c, counts); } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} + return; + } + for child in node.children() { + walk(child, counts); } } @@ -8382,13 +8390,13 @@ mod tests { // Sum aggregate over the same scan, built independently at each root. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let shared = asap_types::pre_asap::cse::share_common_sub_dags(vec![("a", a), ("b", b)]); + let shared = asap_types::ir::cse::share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; assert!(Rc::ptr_eq(ra, rb), "fixture sanity: the two roots merged"); - let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); + let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); let counts = count_consumers(&roots); let count = counts[&Rc::as_ptr(&roots[0])]; assert_eq!(count, 2); @@ -8416,7 +8424,7 @@ mod tests { delta: 0.01, }, }; - let root = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let root = agg(vec![2], intent, metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); // One group for the Aggregate, one for its Scan child. @@ -8424,7 +8432,7 @@ mod tests { let agg_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("an Aggregate group must be discovered"); assert_eq!(agg_group.consumer_count, 1); assert_eq!( @@ -8436,24 +8444,26 @@ mod tests { assert!(agg_group .candidates .iter() - .all(|c| matches!(c.replacement, Replacement::Summary(_)))); + .all(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap()))); assert_eq!( agg_group .candidates .iter() .filter(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return false; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = + &node.operator + else { return false; }; matches!( - &summary_input.expr, - SummaryExpr::SummaryAgg { + &summary_input.operator, + Operator::ASAP(ASAPOp::SummaryAgg { grouping: GroupingStrategy::SharedMultiSubpopulation { .. }, .. - } + }) ) }) .count(), @@ -8477,7 +8487,7 @@ mod tests { let scan_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Scan { .. }))) .expect("a Scan group must be discovered"); assert_eq!(scan_group.consumer_count, 1); assert!( @@ -8488,20 +8498,20 @@ mod tests { #[test] fn cardinality_group_keeps_all_four_candidates() { - let root = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let root = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let agg_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); assert_eq!(agg_group.candidates.len(), 4); assert!(agg_group.candidates.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.is_none() + Replacement::SubDAG(node) if node.guarantee.is_none() && candidate.has_missing_accuracy_evidence() ))); - let root = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let root = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let targeted = search_workload_with_targets( vec![( "q", @@ -8522,7 +8532,7 @@ mod tests { .iter() .any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.is_none() + Replacement::SubDAG(node) if node.guarantee.is_none() && candidate.has_missing_accuracy_evidence() ))); assert!(!targeted @@ -8535,7 +8545,7 @@ mod tests { let exact_target = search_workload_with_targets( vec![( "q", - Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))), + agg(vec![2], default_cardinality(), metric_scan(&["job"])), Some(AccuracyTarget::Exact), )], &default_strategies(), @@ -8554,11 +8564,11 @@ mod tests { // Two independently-built, structurally identical Sum aggregates: // share_common_sub_dags (run inside search_workload) collapses them // onto one Rc with consumer_count 2, so this single group should - // carry SketchAlgorithmStrategy's one ExactAggregate candidate *and* + // carry ASAPStrategies's one ExactAggregate candidate *and* // SharedSubDAGStrategy's share-vs-recompute pair. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); // roots[0] and roots[1] must have merged onto the same Rc. assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); @@ -8572,15 +8582,17 @@ mod tests { group.candidates ); + // Old `Replacement::Summary` ↔ a `Subtree` containing an ASAP node; + // old `Replacement::Rewrite` ↔ a pure pre-ASAP `Subtree`. let summary_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Summary(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap())) .count(); let rewrite_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if !n.contains_asap())) .count(); assert_eq!(summary_count, 1); assert_eq!(rewrite_count, 2); @@ -8589,11 +8601,11 @@ mod tests { // (the "false-positive dedup" failure mode `is_duplicate_rewrite` // exists to prevent). let one_is_the_target = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target)), + |c| matches!(&c.replacement, Replacement::SubDAG(rc) if Rc::ptr_eq(rc, &group.target)), + ); + let one_is_not = group.candidates.iter().any( + |c| matches!(&c.replacement, Replacement::SubDAG(rc) if !Rc::ptr_eq(rc, &group.target)), ); - let one_is_not = group.candidates.iter().any(|c| { - matches!(&c.replacement, Replacement::Rewrite(rc) if !Rc::ptr_eq(rc, &group.target)) - }); assert!(one_is_the_target && one_is_not); } @@ -8604,27 +8616,27 @@ mod tests { // walking the whole DAG, not just root-level pointer identity // (a naive whole-root-only consumer-count pass would miss this; // this module's discover_targets must not). + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let shared = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); // Different predicates so the two Filter *parents* stay distinct // (don't themselves merge under CSE) — only their shared `child` // should collapse onto one `Rc`. - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::clone(&shared), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::clone(&shared), - }; + let root_a = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: Rc::clone(&shared), + })) + .unwrap(); + let root_b = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: Rc::clone(&shared), + })) + .unwrap(); - let space = search_workload(vec![("a", Rc::new(root_a)), ("b", Rc::new(root_b))]); + let space = search_workload(vec![("a", root_a), ("b", root_b)]); assert_eq!( space.len(), 4, @@ -8638,17 +8650,17 @@ mod tests { // pointer as) the pre-search `shared` variable. Recover it from the // post-CSE root's own `child` field instead of the stale `shared` // handle. - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: post_cse_shared_a, .. - } = space.roots[0].1.as_ref() + }) = space.roots[0].1.non_asap() else { panic!("expected a Filter root"); }; - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: post_cse_shared_b, .. - } = space.roots[1].1.as_ref() + }) = space.roots[1].1.non_asap() else { panic!("expected a Filter root"); }; @@ -8674,14 +8686,10 @@ mod tests { #[test] fn add_candidate_rejects_a_true_rewrite_duplicate() { // SharedSubDAGStrategy's `Replacement::Rewrite` candidates are - // real `QueryExpr` values with `PartialEq`, so `add_candidate` can + // real `OperatorNode` values with `PartialEq`, so `add_candidate` can // (and must) actually reject a genuine repeat — unlike the // `Replacement::Summary` case (see the test below). - let root = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let root = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 2); let target = TargetSubDAG::with_consumer_count(&root, 2); let mut inserted = 0; @@ -8712,8 +8720,8 @@ mod tests { #[test] fn add_candidate_never_dedups_summary_candidates() { - // Documented, deliberate consequence of `SummaryNode` deriving no - // `PartialEq` (see `is_duplicate_summary`'s own doc): re-proposing + // Documented, deliberate consequence of `is_duplicate_summary` + // refusing value equality on `f64`-bearing summaries: re-proposing // the same `Replacement::Summary` candidates DOES grow the group — // this module refuses to guess at an equality check it can't back // with a real `PartialEq`. `search_workload_with` never actually @@ -8721,9 +8729,9 @@ mod tests { // module docs' "Termination" section), so this test exists to pin // the documented behavior, not to endorse calling `replacements` // twice for the same target. - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 1); - let strategy = SketchAlgorithmStrategy::default_cost_model(); + let strategy = ASAPStrategies::default_cost_model(); let target = TargetSubDAG::new(&root); for candidate in strategy.replacements(&target) { group.add_candidate(candidate); @@ -8742,11 +8750,7 @@ mod tests { #[test] fn is_duplicate_rewrite_never_merges_share_with_recompute() { - let target = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let target = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let share = Rc::clone(&target); let recompute = Rc::new((*target).clone()); assert!(!Rc::ptr_eq(&share, &recompute)); @@ -8760,11 +8764,7 @@ mod tests { #[test] fn is_duplicate_rewrite_catches_a_real_repeat() { - let target = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let target = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let first_recompute = Rc::new((*target).clone()); let second_recompute = Rc::new((*target).clone()); assert!(!Rc::ptr_eq(&first_recompute, &second_recompute)); @@ -8785,7 +8785,7 @@ mod tests { let mut roots = Vec::new(); let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); for i in 0..20 { - roots.push((i, Rc::new(shared.clone()))); + roots.push((i, Rc::new((*shared).clone()))); } let space = search_workload(roots); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); @@ -8798,18 +8798,18 @@ mod tests { .unwrap(); assert!(matches!( &ranked_group.candidates[0].replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target) + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, &group.target) )); let rewrites: Vec<&ReplacementSubDAG> = ranked_group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if !n.contains_asap())) .copied() .collect(); assert_eq!(rewrites.len(), 2); let first_shares_target = match &rewrites[0].replacement { - Replacement::Rewrite(rc) => Rc::ptr_eq(rc, &group.target), - Replacement::Summary(_) | Replacement::ExactComposition(_) => false, + Replacement::SubDAG(rc) => Rc::ptr_eq(rc, &group.target), + Replacement::ExactComposition(_) => false, }; assert!( first_shares_target, @@ -8835,17 +8835,17 @@ mod tests { } } - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let ranked = space.cost_sorted(&PreferDDSketch); let agg_group = ranked .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); assert_eq!(agg_group.candidates.len(), 2); let first_kind = match &agg_group.candidates[0].replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }; assert_eq!(first_kind, Some(SketchAlgorithm::DDSketch)); } @@ -8863,7 +8863,7 @@ mod tests { candidates.to_vec() } - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { + fn estimated_subpopulation_count(&self, _target: &OperatorNode) -> Option { Some(self.0) } } @@ -8876,19 +8876,15 @@ mod tests { delta: 0.01, }, }; - let root = Rc::new(agg( - vec![2, 3], - intent, - metric_scan(&["tenant_id", "endpoint"]), - )); + let root = agg(vec![2, 3], intent, metric_scan(&["tenant_id", "endpoint"])); let strategies = default_strategies_with(&model); let space = search_workload_with(vec![("tenant_endpoint_count", root)], &strategies); let ranked = space.cost_sorted(&model); let aggregate = ranked .iter() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("aggregate group"); - let Replacement::Summary(node) = &aggregate.candidates[0].replacement else { + let Replacement::SubDAG(node) = &aggregate.candidates[0].replacement else { panic!("grouping candidate must be a summary") }; summary_grouping(node) @@ -8912,12 +8908,12 @@ mod tests { /// and target produces, not some other (or stale) number. #[test] fn cost_sorted_pairs_each_candidate_with_its_own_estimate_cost() { - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let ranked = space.cost_sorted(&DefaultCostModel); let agg_group = ranked .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); assert_eq!( agg_group.costs.len(), @@ -8939,7 +8935,7 @@ mod tests { // ── global_selection (issue #271) ─────────────────────────────────── - /// A `CostModel` with a constant, `sub_dag`-independent recompute cost + /// A `CostModel` with a constant, `sub-DAG`-independent recompute cost /// and shared-maintenance cost, chosen (40 recompute-per-use, 100 /// maintenance) so that a `SharedSubDAGStrategy` group's /// `cse_share_decision` flips exactly between a consumer count of 2 @@ -8996,7 +8992,7 @@ mod tests { } } - let aggregate = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let aggregate = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("left", Rc::clone(&aggregate)), ("right", aggregate)]); let root = &space.roots[0].1; assert!(cse_candidate_pair(space.candidates_for_target(root).unwrap()).is_some()); @@ -9011,7 +9007,7 @@ mod tests { // must equal the group's own raw consumer_count, and its `chosen` // candidate must be cost_sorted's top pick, for both the sketch // group and its child Scan. - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let ranked = space.cost_sorted(&DefaultCostModel); @@ -9038,12 +9034,12 @@ mod tests { // A bare Scan: no registered strategy has an opinion on it, so it // gets a group with an empty candidate list (see TargetSubDAGCandidates's own // doc) — global_selection must not invent a candidate for it. - let root = Rc::new(metric_scan(&["job"])); + let root = metric_scan(&["job"]); let space = search_workload(vec![("q", root)]); let selected = space.global_selection(&DefaultCostModel); let scan_group = selected .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Scan { .. }))) .unwrap(); assert!(scan_group.chosen.is_none()); assert_eq!(scan_group.effective_consumer_count, 1); @@ -9051,7 +9047,7 @@ mod tests { #[test] fn global_selection_falls_back_to_local_ranking_for_sketch_family_groups() { - // SketchAlgorithmStrategy groups have no cross-group-aware cost hook + // ASAPStrategies groups have no cross-group-aware cost hook // (rank_candidates takes no consumer_count) — global_selection must // still return cost_sorted's own top pick for them (documented in // the module docs' "Whole-plan (cross-group) selection" section), @@ -9076,16 +9072,16 @@ mod tests { } } - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let space = search_workload(vec![("q", root)]); let selected = space.global_selection(&PreferDDSketch); let agg_group = selected .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .unwrap(); let kind = match &agg_group.chosen.unwrap().replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, }; assert_eq!(kind, Some(SketchAlgorithm::DDSketch)); @@ -9109,24 +9105,29 @@ mod tests { #[test] fn mixed_rewrite_group_keeps_and_selects_its_explicit_cse_pair() { - let target = Rc::new(metric_scan(&["job"])); + let target = metric_scan(&["job"]); let mut group = TargetSubDAGCandidates::new(Rc::clone(&target), 2); group.candidates = vec![ ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::clone(&target)), + replacement: Replacement::SubDAG(Rc::clone(&target)), provenance: ReplacementProvenance::CseShare, rationale: "share".into(), }, ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new(target.as_ref().clone())), + replacement: Replacement::SubDAG(Rc::new(target.as_ref().clone())), provenance: ReplacementProvenance::CseRecompute, rationale: "recompute".into(), }, ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::CurrentTimestamp)), + replacement: Replacement::SubDAG( + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlVectorFromScalar(ScalarExpr::EvalTimestamp), + )) + .unwrap(), + ), provenance: ReplacementProvenance::LogicalRewrite, rationale: "different rewrite strategy".into(), }, @@ -9166,7 +9167,7 @@ mod tests { // `a` and `c` are both non-`Aggregate` nodes (`Filter`/`Dedup`) so // neither is bindable — each group is a *clean* two-candidate // SharedSubDAGStrategy share-vs-recompute pair, with no - // SketchAlgorithmStrategy `Summary` candidate mixed in to complicate + // ASAPStrategies `Summary` candidate mixed in to complicate // ranking (see `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // for what a *mixed*-shape group looks like — deliberately avoided // here to isolate the SharedSubDAGStrategy-only interaction). @@ -9182,23 +9183,25 @@ mod tests { // which flips its own decision to Share. Only global_selection, // which folds `a`'s decision into `c`'s effective_consumer_count // before deciding `c`, gets this right. + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let c = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), + let c = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap() }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(c()), + let a = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: c(), + })) + .unwrap() }; - let space = search_workload(vec![ - ("root1", Rc::new(a())), - ("root2", Rc::new(a())), - ("root3", Rc::new(c())), - ]); + let space = search_workload(vec![("root1", a()), ("root2", a()), ("root3", c())]); // Fixture sanity: root1/root2 merged onto one shared `a`, and `c` // (root1/root2's shared child, and root3 itself) merged onto one @@ -9206,7 +9209,7 @@ mod tests { // (non-mixed) two-candidate SharedSubDAGStrategy pairs. assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_via_a, .. } = a_rc.as_ref() else { + let Some(NonASAPOp::Filter { child: c_via_a, .. }) = a_rc.non_asap() else { panic!("expected root1/root2 to still be a Filter"); }; assert!(Rc::ptr_eq(c_via_a, &space.roots[2].1)); @@ -9240,7 +9243,7 @@ mod tests { .unwrap(); let c_top_shares = matches!( &c_ranked.candidates[0].replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, c_via_a) + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, c_via_a) ); assert!( !c_top_shares, @@ -9261,7 +9264,7 @@ mod tests { ); let a_shares = matches!( &a_selected.chosen.unwrap().replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, a_rc) + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, a_rc) ); assert!( !a_shares, @@ -9274,7 +9277,7 @@ mod tests { ); let c_shares = matches!( &c_selected.chosen.unwrap().replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, c_via_a) + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, c_via_a) ); assert!( c_shares, @@ -9312,10 +9315,12 @@ mod tests { } } - let shared = Rc::new(QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), - }); + let shared = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap(); let space = search_workload(vec![ ("left", Rc::clone(&shared)), ("right", Rc::clone(&shared)), @@ -9328,20 +9333,26 @@ mod tests { #[test] fn effective_repetition_materializes_a_cse_choice_for_a_single_edge_child() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let c = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), + let c = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap() }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(c()), + let a = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: c(), + })) + .unwrap() }; - let space = search_workload(vec![("root1", Rc::new(a())), ("root2", Rc::new(a()))]); + let space = search_workload(vec![("root1", a()), ("root2", a())]); let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_rc, .. } = a_rc.as_ref() else { + let Some(NonASAPOp::Filter { child: c_rc, .. }) = a_rc.non_asap() else { panic!("expected Filter root"); }; @@ -9356,8 +9367,8 @@ mod tests { #[test] fn shared_ancestor_keeps_a_single_use_cse_descendant_selected() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; struct AlwaysShare; impl CostModel for AlwaysShare { @@ -9378,22 +9389,25 @@ mod tests { } } - let child = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), + let child = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap() }; - let parent = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(child()), + let parent = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: child(), + })) + .unwrap() }; - let space = search_workload(vec![ - ("root1", Rc::new(parent())), - ("root2", Rc::new(parent())), - ]); + let space = search_workload(vec![("root1", parent()), ("root2", parent())]); let parent_rc = &space.roots[0].1; - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: child_rc, .. - } = parent_rc.as_ref() + }) = parent_rc.non_asap() else { panic!("expected Filter root"); }; @@ -9417,38 +9431,44 @@ mod tests { #[test] fn global_selection_propagates_uses_through_the_selected_rewrite() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; struct ReplaceFilterChild; impl ReplacementStrategy for ReplaceFilterChild { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - matches!(target.root.as_ref(), QueryExpr::Filter { .. }) + matches!(target.root.non_asap(), Some(NonASAPOp::Filter { .. })) } fn replacements(&self, _target: &TargetSubDAG<'_>) -> Vec { vec![ReplacementSubDAG { strategy: "ReplaceFilterChild", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["replacement"])), - })), + replacement: Replacement::SubDAG( + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["replacement"]), + }, + )) + .unwrap(), + ), provenance: ReplacementProvenance::LogicalRewrite, rationale: "replace the Filter and its input".into(), }] } } - let original_child = Rc::new(metric_scan(&["original"])); - let root = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), + let original_child = metric_scan(&["original"]); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), child: Rc::clone(&original_child), - }); + })) + .unwrap(); let strategies: Vec> = vec![Box::new(ReplaceFilterChild)]; let space = search_workload_with(vec![("q", root)], &strategies); let root = &space.roots[0].1; let selected = space.global_selection(&DefaultCostModel); - let Replacement::Rewrite(rewrite) = &selected + let Replacement::SubDAG(rewrite) = &selected .for_target(root) .unwrap() .chosen @@ -9457,17 +9477,17 @@ mod tests { else { panic!("expected logical rewrite"); }; - let QueryExpr::Dedup { + let Some(NonASAPOp::Dedup { child: replacement_child, .. - } = rewrite.as_ref() + }) = rewrite.non_asap() else { panic!("expected Dedup rewrite"); }; - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: original_child, .. - } = root.as_ref() + }) = root.non_asap() else { panic!("expected Filter root"); }; @@ -9503,11 +9523,11 @@ mod tests { fn candidate_cost(&self, _: &ReplacementSubDAG, _: &TargetSubDAG<'_>) -> Option { Some(Cost(1.0)) } - fn summary_support_evidence(&self, _: &SummaryNode) -> Option { + fn summary_support_evidence(&self, _: &OperatorNode) -> Option { Some(false) } } - let root = Rc::new(lower_promql("sum_over_time(a[1m])", AccuracyTarget::Exact)); + let root = lower_promql("sum_over_time(a[1m])", AccuracyTarget::Exact); let space = search_workload(vec![("q", root)]); let selected = space.global_selection(&Unsupported); assert!(selected @@ -9520,18 +9540,15 @@ mod tests { // Composable temporal/grouped Sum must be executable as one producer. #[test] fn grouped_temporal_sum_has_one_summary_producer_candidate() { - let root = Rc::new(lower_promql( - "sum by(job)(sum_over_time(a[1m]))", - AccuracyTarget::Exact, - )); + let root = lower_promql("sum by(job)(sum_over_time(a[1m]))", AccuracyTarget::Exact); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); assert!(candidates .iter() .any(|candidate| matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))))); + Replacement::SubDAG(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())))); struct PreferComposed; impl CostModel for PreferComposed { fn rank_candidates( @@ -9548,9 +9565,9 @@ mod tests { ) -> Option { Some(Cost( if matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))) + Replacement::SubDAG(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())) { 1.0 } else { @@ -9562,9 +9579,9 @@ mod tests { let space = search_workload(vec![("q", root.clone())]); let selected = space.global_selection(&PreferComposed); let node = selected.assemble_target(&space.roots[0].1).unwrap(); - assert!(matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))); + assert!(matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())); } // Mixed candidate ranking must honor explicit costs, not legacy estimates. @@ -9593,10 +9610,7 @@ mod tests { )) } } - let root = Rc::new(lower_promql( - "sum by(job)(sum_over_time(a[1m]))", - AccuracyTarget::Exact, - )); + let root = lower_promql("sum by(job)(sum_over_time(a[1m]))", AccuracyTarget::Exact); let space = search_workload(vec![("q", root)]); let selection = space.global_selection(&ExplicitCosts); let selected = selection @@ -9632,16 +9646,8 @@ mod tests { } } - let a = Rc::new(agg( - vec![2], - AggIntent::Avg { col: None }, - metric_scan(&["job"]), - )); - let b = Rc::new(agg( - vec![2], - AggIntent::Avg { col: None }, - metric_scan(&["job"]), - )); + let a = agg(vec![2], AggIntent::Avg { col: None }, metric_scan(&["job"])); + let b = agg(vec![2], AggIntent::Avg { col: None }, metric_scan(&["job"])); let space = search_workload(vec![("a", a), ("b", b)]); let root = &space.roots[0].1; let selected = space.global_selection(&PreferLogicalRewrite); @@ -9657,26 +9663,30 @@ mod tests { #[test] fn topological_order_puts_a_later_discovered_parent_before_its_child() { - // Mirrors nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered's + // Mirrors nested_shared_sub-DAG_below_an_unshared_parent_is_still_discovered's // diamond fixture: discover_targets's own `order` visits root_b (a // parent of `shared`) *after* `shared` itself, because `shared` was // already fully walked via root_a first. A naive "process // discover_targets's own order" DP would see root_b's child edge // after already processing `shared` — topological_order must not // make that mistake. + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::new(shared.clone()), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::new(shared), - }; - let roots = vec![("a", Rc::new(root_a)), ("b", Rc::new(root_b))]; + let root_a = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: shared.clone(), + })) + .unwrap(); + let root_b = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: shared, + })) + .unwrap(); + let roots = vec![("a", root_a), ("b", root_b)]; let mut order = Vec::new(); let mut nodes = HashMap::new(); @@ -9702,10 +9712,10 @@ mod tests { // Discovery-order sanity: root_b comes after the shared child in // discover_targets's own order (the exact non-topological case this // test exists to cover). - let QueryExpr::Filter { + let Some(NonASAPOp::Filter { child: shared_via_a, .. - } = space.roots[0].1.as_ref() + }) = space.roots[0].1.non_asap() else { panic!("expected a Filter root"); }; @@ -9738,7 +9748,7 @@ mod tests { // module docs — so this always converges in exactly 2 passes). let a = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let b = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); assert!(!space.is_empty()); } @@ -9764,19 +9774,23 @@ mod tests { fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { let n = self.next.get(); self.next.set(n + 1); + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let fresh_inner_layer = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(n)))), - child: Rc::clone(target.root), - }; - let outer_wrapper = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(fresh_inner_layer), - }; + let fresh_inner_layer = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(n))), + child: Rc::clone(target.root), + })) + .unwrap(); + let outer_wrapper = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: fresh_inner_layer, + })) + .unwrap(); vec![ReplacementSubDAG { strategy: "AlwaysGrowingStrategy", - replacement: Replacement::Rewrite(Rc::new(outer_wrapper)), + replacement: Replacement::SubDAG(outer_wrapper), provenance: ReplacementProvenance::LogicalRewrite, rationale: format!("pathological candidate #{n}"), }] @@ -9786,13 +9800,13 @@ mod tests { #[test] #[should_panic(expected = "did not converge")] fn a_pathologically_growing_strategy_trips_the_iteration_cap() { - let root = Rc::new(metric_scan(&["job"])); + let root = metric_scan(&["job"]); let strategies: Vec> = vec![Box::new(AlwaysGrowingStrategy { next: std::cell::Cell::new(0), })]; let _ = search_workload_with(vec![("q", root)], &strategies); } - // ── realize_child / keep_pre_asap: end-to-end single-target realization ── + // ── realize_child / retain_exact: end-to-end single-target realization ── // // Moved from the former `bind.rs` (issue #251): `bind.rs`'s own // workload-wide orchestration (`implement_workload`/ @@ -9805,17 +9819,6 @@ mod tests { // pattern by hand since `realize_child` is `pub(crate)`), these tests // call `realize_child` directly. - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: ReductionTy::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - fn field<'a>(schema: &'a Schema, name: &str) -> &'a Field { schema .fields @@ -9825,29 +9828,29 @@ mod tests { } fn realize_first( - expr: &QueryExpr, + expr: &OperatorNode, cost_model: &dyn CostModel, - ) -> Result, RealizationError> { + ) -> Result, RealizationError> { realize_child(&Rc::new(expr.clone()), cost_model) } - fn realize(expr: &QueryExpr) -> Result, RealizationError> { + fn realize(expr: &OperatorNode) -> Result, RealizationError> { realize_first(expr, &DefaultCostModel) } #[test] fn quantile_realizes_kll_wrapped_in_estimate() { // quantile by (job) (m) at ε=0.01 → Estimate(Quantile) over - // SummaryAgg(Kll{k:269}) over KeepPreAsap(Scan). job = col 2. + // SummaryAgg(Kll{k:269}) over the kept Scan. job = col 2. let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, PostAsapSketchStatistic::Quantile { q } if *q == 0.99)); // Estimate edge: plain row shape — group key + Float64 answer. @@ -9860,15 +9863,15 @@ mod tests { FieldDataType::Plain(DataType::Utf8) ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -9888,8 +9891,9 @@ mod tests { GroupingStrategy::default() ) ); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(ref e) - if matches!(**e, QueryExpr::Scan { .. }))); + // The kept pre-ASAP leaf is the Scan node itself (no wrapper). + assert!(matches!(child.non_asap(), Some(NonASAPOp::Scan { .. }))); + assert!(!child.contains_asap()); } /// A deployment-supplied [`CostModel`] can override the default KLL @@ -9919,11 +9923,15 @@ mod tests { // Default: KLL (see `quantile_realizes_kll_wrapped_in_estimate` above). let default_root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &default_root.expr else { - panic!("expected SummaryEstimate root, got {:?}", default_root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &default_root.operator + else { + panic!( + "expected SummaryEstimate root, got {:?}", + default_root.operator + ); }; - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert!(matches!( family, @@ -9932,11 +9940,15 @@ mod tests { // With `PreferDDSketchViaCostModel`: DDSketch instead, same query. let custom_root = realize_first(&q, &PreferDDSketchViaCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &custom_root.expr else { - panic!("expected SummaryEstimate root, got {:?}", custom_root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &custom_root.operator + else { + panic!( + "expected SummaryEstimate root, got {:?}", + custom_root.operator + ); }; - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -9953,8 +9965,8 @@ mod tests { /// A deployment-supplied `CostModel` can realize an `AggIntent::Extension` /// intent as a real sketch instead of the default `PassThrough` (issue /// #150) — `realizations_for_intent` must consult `realize_extension` - /// for the `Extension` arm, and `readout` must consult - /// `readout_extension` to build its `SketchStatistic` without panicking. + /// for the `Extension` arm, and `evaluation` must consult + /// `evaluation_extension` to build its `SketchStatistic` without panicking. struct FrequencyCostModel; impl CostModel for FrequencyCostModel { @@ -9980,7 +9992,7 @@ mod tests { } } - fn readout_extension( + fn evaluation_extension( &self, ext_kind: &str, payload: &serde_json::Value, @@ -10006,7 +10018,7 @@ mod tests { }; let q = agg(vec![], intent, metric_scan(&[])); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] @@ -10018,12 +10030,12 @@ mod tests { let q = agg(vec![], intent, metric_scan(&[])); let root = realize_first(&q, &FrequencyCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!( query, @@ -10031,8 +10043,8 @@ mod tests { if k == "item" && v == "checkout" )); - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -10053,10 +10065,10 @@ mod tests { fn exact_sum_realizes_accumulator_without_estimate() { let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { family, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &root.operator else { panic!( "expected bare SummaryAgg (no estimate), got {:?}", - root.expr + root.operator ); }; assert_eq!( @@ -10076,14 +10088,16 @@ mod tests { use std::time::Duration; let q = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["job"])), - }, + child: metric_scan(&["job"]), + })) + .unwrap(), ); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { family, .. } = &root.expr else { - panic!("expected SummaryAgg, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &root.operator else { + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, @@ -10114,17 +10128,19 @@ mod tests { use std::time::Duration; let q = agg_per_entity( default_quantile(0.99), - QueryExpr::TimeRange { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: Duration::from_secs(10), - child: Rc::new(metric_scan(&["job"])), - }, + child: metric_scan(&["job"]), + })) + .unwrap(), ); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { reduction, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!(reduction, &ReductionTy::PerEntity); } @@ -10142,11 +10158,11 @@ mod tests { }; let q = agg(vec![], intent, metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { reduction, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!(reduction, &ReductionTy::by(vec![])); } @@ -10158,39 +10174,41 @@ mod tests { // accumulator. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); - let root = realize(&outer).unwrap(); + // Timing is not stored during realization: time the candidate under + // the default lifecycle assignment to read the maintenance boundary. + let root = timed(&realize(&outer).unwrap()); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { child, family, .. } = &summary_input.expr else { - panic!("expected outer SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, .. }) = &summary_input.operator + else { + panic!( + "expected outer SummaryAgg, got {:?}", + summary_input.operator + ); }; assert!(matches!( family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll )); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } = &child.expr - else { - panic!("expected explicit maintenance readout"); + assert_eq!(child.timing, Some(ExecutionTiming::IngestionTime)); + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) = &child.operator else { + panic!("expected explicit maintenance evaluation"); }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: inner_family, child: leaf, .. - } = &child.expr + }) = &child.operator else { - panic!("expected inner SummaryAgg, got {:?}", child.expr); + panic!("expected inner SummaryAgg, got {:?}", child.operator); }; assert_eq!( inner_family, &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); - assert!(matches!(leaf.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!leaf.contains_asap()); } /// Issue #115: the summary is built over the intent's own input columns. @@ -10225,13 +10243,15 @@ mod tests { } } - /// The update expression of the first `SummaryAgg` in the DAG. - fn find_summary_input(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryAgg { input, .. } if input.item.is_none() => { + /// The update expression of the first `SummaryAgg` in the tree. + fn find_summary_input(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) if input.item.is_none() => { Some(input.weight.clone()) } - SummaryExpr::SummaryEstimate { summary_input, .. } => find_summary_input(summary_input), + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + find_summary_input(summary_input) + } _ => None, } } @@ -10252,59 +10272,62 @@ mod tests { ] { let q = agg(vec![2], intent.clone(), metric_scan(&["job"])); let root = realize(&q).unwrap(); + // Kept pass-through: the pre-ASAP node itself, not a wrapper. assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(ref e) if **e == q), - "expected KeepPreAsap passthrough for {intent:?}" + !root.contains_asap() && root.operator == q.operator && root.schema == q.schema, + "expected kept pre-ASAP passthrough for {intent:?}" ); } } #[test] fn logical_parent_subsumes_bindable_child() { - // Filter over a bindable quantile: `KeepPreAsap` has no post-ASAP - // children, so the conservative fallback keeps the whole sub-DAG + // Filter over a bindable quantile: a kept non-ASAP sub-DAG has no + // summary children, so the conservative fallback keeps the whole sub-DAG // logical. + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; - use asap_types::pre_asap::query_expr::Predicate; - let q = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(0.5))), - })), - child: Rc::new(agg(vec![], default_quantile(0.99), metric_scan(&[]))), - }; + right: Box::new(ScalarExpr::Literal(ScalarValue::Float64(0.5))), + semantics: asap_types::ir::ExprSemantics::Promql, + }), + child: agg(vec![], default_quantile(0.99), metric_scan(&[])), + })) + .unwrap(); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(ref e) if **e == q)); + assert!( + !root.contains_asap() && root.operator == q.operator && root.schema == q.schema, + "expected the whole Filter sub_dag kept pre-ASAP" + ); } #[test] fn having_and_multi_intent_stay_logical() { + use asap_types::ir::Predicate; use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let mut q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { - *having = Some(Predicate(Rc::new(QueryExpr::Literal( - ScalarValue::Boolean(true), - )))); - } - assert!(matches!( - realize(&q).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); - - let multi = QueryExpr::Aggregate { - reduction: ReductionTy::by(vec![2]), - measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }; - assert!(matches!( - realize(&multi).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); + let q = crate::test_support::aggregate( + ReductionTy::by(vec![2]), + vec![default_quantile(0.99)], + vec![], + Some(Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true)))), + metric_scan(&["job"]), + ); + assert!(!realize(&q).unwrap().contains_asap()); + + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: ReductionTy::by(vec![2]), + measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); + assert!(!realize(&multi).unwrap().contains_asap()); } // No binding rule applies a per-measure `FILTER` (#466), so the @@ -10312,18 +10335,16 @@ mod tests { #[test] fn filtered_measure_stays_logical() { use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; let mut q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { filters, .. } = &mut q { - *filters = vec![Some(Predicate(Rc::new(QueryExpr::Literal( - ScalarValue::Boolean(true), + if let Operator::NonASAP(NonASAPOp::Aggregate { filters, .. }) = + &mut Rc::make_mut(&mut q).operator + { + *filters = vec![Some(Predicate(ScalarExpr::Literal(ScalarValue::Boolean( + true, ))))]; } assert!(bindable_intent(&q).is_none()); - assert!(matches!( - realize(&q).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); + assert!(!realize(&q).unwrap().contains_asap()); } #[test] @@ -10337,7 +10358,7 @@ mod tests { metric_scan(&["job"]), ); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] @@ -10349,20 +10370,20 @@ mod tests { }, metric_scan(&["job"]), ); - let root = Rc::new(agg( + let root = agg( vec![], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); + ); let proposals = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); assert!(!proposals.is_empty()); assert!(proposals.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.as_ref().is_some_and(|g| + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|g| g.bound.evaluate().is_none() && g.failure_probability.evaluate().is_none()) ))); @@ -10402,7 +10423,7 @@ mod tests { }, metric_scan(&["job"]), ); - let q = Rc::new(agg( + let q = agg( vec![], AggIntent::TopK { k: 5, @@ -10412,8 +10433,8 @@ mod tests { }, }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10423,7 +10444,7 @@ mod tests { assert!(!replacements.is_empty()); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|g| g.metric == ErrorMetric::TopKMembership && g.failure_probability.evaluate() == Some(0.005)) @@ -10439,15 +10460,15 @@ mod tests { }, metric_scan(&["service"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10457,26 +10478,26 @@ mod tests { let node = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => { + Replacement::SubDAG(node) if candidate.rationale.contains("CmsWithHeap") => { Some(node) } _ => None, }) .expect("CmsWithHeap candidate"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &node.expr + }) = &node.operator else { - panic!("expected Top-K readout") + panic!("expected Top-K evaluation") }; assert!(matches!(query, PostAsapSketchStatistic::TopK { k: 10 })); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, .. - } = &summary_input.expr + }) = &summary_input.operator else { panic!("expected fused summary aggregation") }; @@ -10496,7 +10517,7 @@ mod tests { FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap )); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } #[test] @@ -10506,15 +10527,15 @@ mod tests { AggIntent::Sum { col: None }, metric_scan(&["service"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10531,7 +10552,7 @@ mod tests { let node = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if candidate.rationale.contains("CountSketchWithHeap") => { Some(node) @@ -10539,15 +10560,16 @@ mod tests { _ => None, }) .expect("CountSketchWithHeap candidate"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &node.expr + }) = &node.operator else { - panic!("expected Top-K readout") + panic!("expected Top-K evaluation") }; assert!(matches!(query, PostAsapSketchStatistic::TopK { k: 5 })); - let SummaryExpr::SummaryAgg { child, input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, input, .. }) = &summary_input.operator + else { panic!("expected fused summary aggregation") }; assert!(matches!( @@ -10558,31 +10580,37 @@ mod tests { input.weight, SummaryInputExpr::Column(ColumnRef::SampleValue) ); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } #[test] fn temporal_per_entity_topk_uses_series_identity_and_sample_value() { - let inner = QueryExpr::Aggregate { - reduction: ReductionTy::PerEntity, - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: std::time::Duration::from_secs(60), - child: Rc::new(metric_scan(&["service"])), - }), - }; - let outer = Rc::new(agg( + let inner = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: ReductionTy::PerEntity, + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + range: std::time::Duration::from_secs(60), + child: metric_scan(&["service"]), + }, + )) + .unwrap(), + })) + .unwrap(); + let outer = agg( vec![2], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10590,13 +10618,14 @@ mod tests { ); let candidates = strategy.replacements(&TargetSubDAG::new(&outer)); let input = candidates.iter().find_map(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return None; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator + else { return None; }; - let SummaryExpr::SummaryAgg { input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &summary_input.operator else { return None; }; input.item.is_some().then_some(input) @@ -10625,15 +10654,15 @@ mod tests { }, metric_scan(&["service", "region"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10644,10 +10673,10 @@ mod tests { .replacements(&TargetSubDAG::new(&outer)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { - match &summary_input.expr { - SummaryExpr::SummaryAgg { input, .. } => Some(input.clone()), + Replacement::SubDAG(node) => match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) => Some(input.clone()), _ => None, } } @@ -10678,15 +10707,15 @@ mod tests { ); // The inner aggregate outputs its grouping keys first, so column 2 is // `region`. Each region is a separate Top-K subpopulation. - let outer = Rc::new(agg( + let outer = agg( vec![2], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -10696,12 +10725,12 @@ mod tests { .replacements(&TargetSubDAG::new(&outer)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { - match &summary_input.expr { - SummaryExpr::SummaryAgg { + Replacement::SubDAG(node) => match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, reduction, .. - } => Some((input.clone(), reduction.clone())), + }) => Some((input.clone(), reduction.clone())), _ => None, } } @@ -10726,7 +10755,7 @@ mod tests { fn sql_reducer_resolves_named_input_column() { // SUM(bytes) over a tabular scan: `col` resolves positionally to the // named column, not the PromQL sample value. - let scan = QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "t".into(), }, @@ -10740,11 +10769,12 @@ mod tests { unique_keys: vec![], closed: true, }, - }; + })) + .unwrap(); let q = agg(vec![0], AggIntent::Sum { col: Some(1) }, scan); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { input, .. } = &root.expr else { - panic!("expected SummaryAgg, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &root.operator else { + panic!("expected SummaryAgg, got {:?}", root.operator); }; let SummaryInputExpr::Column(col) = &input.weight else { panic!("expected observation column") @@ -10809,10 +10839,12 @@ mod tests { } } - fn summary_child(node: &SummaryNode) -> &Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => summary_child(summary_input), - SummaryExpr::SummaryAgg { child, .. } => child, + fn summary_child(node: &OperatorNode) -> &Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + summary_child(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => child, other => panic!("expected a SummaryAgg, got {other:?}"), } } @@ -10823,9 +10855,8 @@ mod tests { // registered rule, so every outer sketch candidate is refused with a // typed reason and the raw/pre-ASAP alternative is what remains. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); - let proposals = - SketchAlgorithmStrategy::default_cost_model().propose(&TargetSubDAG::new(&outer)); + let outer = agg(vec![], default_quantile(0.99), inner); + let proposals = ASAPStrategies::default_cost_model().propose(&TargetSubDAG::new(&outer)); assert!( proposals.candidates.is_empty(), "no outer sketch may be proposed over an approximate child without a rule: {:?}", @@ -10848,7 +10879,7 @@ mod tests { } // Fallback keeps the whole sub-DAG pre-ASAP — executed exactly. let realized = realize_child(&outer, &DefaultCostModel).unwrap(); - assert!(matches!(realized.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!realized.contains_asap()); assert!(realized .guarantee .as_ref() @@ -10856,9 +10887,8 @@ mod tests { // Cross-metric: a quantile over a cardinality estimate. let inner = agg(vec![2], default_cardinality(), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); - let proposals = - SketchAlgorithmStrategy::default_cost_model().propose(&TargetSubDAG::new(&outer)); + let outer = agg(vec![], default_quantile(0.99), inner); + let proposals = ASAPStrategies::default_cost_model().propose(&TargetSubDAG::new(&outer)); assert!(proposals.candidates.is_empty()); assert!(proposals.rejected.iter().all(|r| matches!( &r.error, @@ -10870,14 +10900,14 @@ mod tests { #[test] fn exact_child_contributes_zero_error() { // quantile(0.9, sum by (job) (m)): KLL over an exact Sum accumulator - // — the readout's guarantee is exactly KLL's own local guarantee. + // — the evaluation's guarantee is exactly KLL's own local guarantee. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); let root = realize(&outer).unwrap(); let guarantee = root .guarantee .as_ref() - .expect("a readout carries a guarantee"); + .expect("a evaluation carries a guarantee"); assert_eq!(guarantee.metric, ErrorMetric::Rank); assert_eq!( guarantee.bound.evaluate(), @@ -10894,7 +10924,7 @@ mod tests { ))); // The sketch *state* node carries no guarantee; the exact // accumulator's state is its value and does. - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { panic!() }; assert!(summary_input.guarantee.is_none()); @@ -10905,16 +10935,19 @@ mod tests { } #[test] - fn exact_sum_can_consume_an_approximate_readout() { + fn exact_sum_can_consume_an_approximate_evaluation() { // sum(count_distinct by (job) (m)) is an outer exact summary over - // the inner HLL readout. Both summary levels remain explicit. + // the inner HLL evaluation. Both summary levels remain explicit. let inner = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let outer = agg(vec![], AggIntent::Sum { col: None }, inner); let root = realize(&outer).unwrap(); - let SummaryExpr::SummaryAgg { child, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &root.operator else { panic!("outer exact sum should remain a SummaryAgg") }; - assert!(matches!(child.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); assert!(root.guarantee.is_some()); // count(...) over the same child is exact: a row count does not @@ -10935,12 +10968,12 @@ mod tests { } #[test] - fn equal_split_allocation_supports_nested_summary_readouts() { + fn equal_split_allocation_supports_nested_summary_evaluations() { // A registered rank-additive rule and valid budget split make both // summary levels explicit while preserving the composed guarantee. let inner = agg(vec![2], quantile_eps(0.5, 0.1), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], quantile_eps(0.99, 0.1), inner)); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs( + let outer = agg(vec![], quantile_eps(0.99, 0.1), inner); + let strategy = ASAPStrategies::new_with_planning_inputs( &DefaultCostModel, &RankAdditiveModel, &EqualSplitAllocator, @@ -10949,13 +10982,15 @@ mod tests { assert!(!proposals.candidates.is_empty()); assert!(proposals.candidates.iter().all(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return false; }; - matches!(node.expr, SummaryExpr::SummaryEstimate { .. }) - && node.guarantee.as_ref().is_some_and(|guarantee| { - DefaultAccuracyModel.satisfies(guarantee, &AccuracyTarget::Epsilon(0.1)) - }) + matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ) && node.guarantee.as_ref().is_some_and(|guarantee| { + DefaultAccuracyModel.satisfies(guarantee, &AccuracyTarget::Epsilon(0.1)) + }) })); } @@ -10964,9 +10999,9 @@ mod tests { // The same nested summary remains available through workload search // and global cost ranking. let inner = agg(vec![2], quantile_eps(0.5, 0.1), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], quantile_eps(0.99, 0.1), inner)); + let outer = agg(vec![], quantile_eps(0.99, 0.1), inner); let strategies: Vec> = - vec![Box::new(SketchAlgorithmStrategy::new_with_planning_inputs( + vec![Box::new(ASAPStrategies::new_with_planning_inputs( &DefaultCostModel, &RankAdditiveModel, &EqualSplitAllocator, @@ -10976,10 +11011,14 @@ mod tests { let group = space.candidates_for_target(root).unwrap(); assert!(!group.rejected.is_empty()); assert!(group.candidates.iter().all(|c| match &c.replacement { - Replacement::Summary(node) => node.guarantee.as_ref().is_some_and(|g| { - DefaultAccuracyModel.satisfies(g, &AccuracyTarget::Epsilon(0.1)) - }), - Replacement::Rewrite(_) => false, + // A summary candidate (old `Replacement::Summary`) contains an + // ASAP node; a logical rewrite (old `Replacement::Rewrite`) does not. + Replacement::SubDAG(node) if node.contains_asap() => { + node.guarantee.as_ref().is_some_and(|g| { + DefaultAccuracyModel.satisfies(g, &AccuracyTarget::Epsilon(0.1)) + }) + } + Replacement::SubDAG(_) => false, Replacement::ExactComposition(_) => false, })); let ranked = space.cost_sorted(&DefaultCostModel); @@ -10992,15 +11031,18 @@ mod tests { .unwrap() .chosen .expect("a nested summary candidate wins"); - let Replacement::Summary(node) = &chosen.replacement else { + let Replacement::SubDAG(node) = &chosen.replacement else { panic!() }; - assert!(matches!(node.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); } #[test] fn root_target_check_removes_candidates_before_cost_ranking() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); // A root target tighter than the node's own ε=0.01: every sketch // candidate misses it and is moved to `rejected`; nothing is left // for the cost model to rank. @@ -11014,7 +11056,7 @@ mod tests { assert!(group .candidates .iter() - .all(|c| matches!(c.replacement, Replacement::Rewrite(_)))); + .all(|c| matches!(&c.replacement, Replacement::SubDAG(n) if !n.contains_asap()))); assert!(group.rejected.iter().all(|r| matches!( r.error, AccuracyError::TargetNotSatisfied { target: AccuracyTarget::Epsilon(e), .. } if e == 0.001 @@ -11033,7 +11075,7 @@ mod tests { assert!(group .candidates .iter() - .any(|c| matches!(c.replacement, Replacement::Summary(_)))); + .any(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap()))); // An `Exact` root target admits only exact candidates. let space = search_workload_with_targets( @@ -11043,11 +11085,11 @@ mod tests { ); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); assert!(group.candidates.iter().all(|c| match &c.replacement { - Replacement::Summary(node) => node + Replacement::SubDAG(node) if node.contains_asap() => node .guarantee .as_ref() .is_some_and(ResultGuarantee::is_exact), - Replacement::Rewrite(_) => true, + Replacement::SubDAG(_) => true, Replacement::ExactComposition(_) => false, })); } @@ -11061,14 +11103,14 @@ mod tests { }, metric_scan(&["job"]), ); - let q = Rc::new(agg( + let q = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); + ); let space = search_workload_with_targets( vec![("q", Rc::clone(&q), Some(AccuracyTarget::Epsilon(0.01)))], &default_strategies(), @@ -11078,14 +11120,14 @@ mod tests { assert!(group.candidates.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.as_ref().is_some_and(ResultGuarantee::has_unknown) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(ResultGuarantee::has_unknown) ))); let candidate = group .candidates .iter() .find(|candidate| candidate.has_missing_accuracy_evidence()) .unwrap(); - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { unreachable!() }; let exported = asap_types::dag_export::export_summary(node); @@ -11109,13 +11151,13 @@ mod tests { fn scoped_hll_evidence_sizes_and_certifies_without_a_deployment_model() { use crate::accuracy::EstimatorContract; struct SourceEvidence { - expression: QueryExpr, + expression: OperatorNode, max_distinct: u32, } impl AccuracyEvidenceProvider for SourceEvidence { - fn estimator_contract(&self, expression: &QueryExpr) -> Option { + fn estimator_contract(&self, expression: &OperatorNode) -> Option { (expression == &self.expression).then_some(EstimatorContract::ClassicHll { - max_distinct_per_readout: self.max_distinct, + max_distinct_per_evaluation: self.max_distinct, }) } } @@ -11123,19 +11165,19 @@ mod tests { epsilon: 0.05, delta: 0.01, }; - let root = Rc::new(agg( + let root = agg( vec![], AggIntent::Cardinality { cols: vec![], accuracy: target.clone(), }, metric_scan(&[]), - )); + ); let evidence = SourceEvidence { expression: (*root).clone(), max_distinct: 128, }; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -11145,7 +11187,7 @@ mod tests { let hll = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll => { Some(node) @@ -11155,13 +11197,13 @@ mod tests { .expect("HLL candidate"); assert!(DefaultAccuracyModel .satisfies(hll.guarantee.as_ref().expect("HLL confidence"), &target)); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &hll.expr else { - panic!("readout") + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &hll.operator else { + panic!("evaluation") }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = &summary_input.operator else { panic!("HLL state") }; @@ -11175,9 +11217,8 @@ mod tests { precision: expected } ); - let absent = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); - assert!(!absent.iter().any(|candidate| matches!(&candidate.replacement, Replacement::Summary(node) + let absent = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); + assert!(!absent.iter().any(|candidate| matches!(&candidate.replacement, Replacement::SubDAG(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll && node.guarantee.as_ref().is_some_and(|g| DefaultAccuracyModel.satisfies(g, &target))))); // Invalid contracts, infeasible targets and evidence for another source // must never authorize a confidence-bearing HLL candidate. @@ -11191,30 +11232,30 @@ mod tests { epsilon: 0.05, delta, }; - let query = Rc::new(agg( + let query = agg( vec![], AggIntent::Cardinality { cols: vec![], accuracy: target.clone(), }, metric_scan(&[]), - )); + ); let evidence = SourceEvidence { expression: if wrong_scope { - metric_scan(&["other"]) + (*metric_scan(&["other"])).clone() } else { (*query).clone() }, max_distinct, }; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, ); assert!(!strategy.replacements(&TargetSubDAG::new(&query)).iter().any(|candidate| - matches!(&candidate.replacement, Replacement::Summary(node) + matches!(&candidate.replacement, Replacement::SubDAG(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll && node.guarantee.as_ref().is_some_and(|g| DefaultAccuracyModel.satisfies(g, &target))))); } } @@ -11222,40 +11263,42 @@ mod tests { // A value projection cannot consume an opaque exact accumulator edge. #[test] fn residual_projection_finalizes_selected_exact_state() { - let inner = Rc::new(agg(vec![], AggIntent::Sum { col: None }, metric_scan(&[]))); - let root = Rc::new(QueryExpr::Project { - cols: vec![asap_types::pre_asap::ProjectItem { - expr: QueryExpr::Column(0), - alias: Some("result".into()), - }], - qualifier: None, - child: inner.clone(), - }); + let inner = agg(vec![], AggIntent::Sum { col: None }, metric_scan(&[])); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols: vec![ProjectItem { + expr: ScalarExpr::Column(0), + alias: Some("result".into()), + }], + qualifier: None, + child: inner.clone(), + })) + .unwrap(); let space = search_workload_with_targets( vec![("q", root.clone(), Some(AccuracyTarget::Exact))], &default_strategies(), &DefaultAccuracyModel, ); let selected = space.global_selection(&DefaultCostModel); + // CSE re-interns the workload, so the space's root/child `Rc`s are not + // the fixture's. Assembly only assembles children that are discovered + // targets, so seed the memo under the space's own child pointer. + let root = Rc::clone(&space.roots[0].1); + let Some(NonASAPOp::Project { child: inner, .. }) = root.non_asap() else { + unreachable!() + }; + assert!(space.candidates_for_target(inner).is_some()); selected .assembled_nodes .borrow_mut() - .insert(Rc::as_ptr(&inner), realize(inner.as_ref()).unwrap()); + .insert(Rc::as_ptr(inner), realize(inner.as_ref()).unwrap()); let node = selected.assemble_target(&root).unwrap(); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { .. }, - .. - } = &node.expr - else { + let Operator::NonASAP(NonASAPOp::Project { child, .. }) = &node.operator else { panic!("expected Project"); }; assert!(matches!( - child.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + child.operator, + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) )); assert!(child .schema @@ -11267,18 +11310,16 @@ mod tests { #[test] fn ranking_uses_aggregate_output_position_not_first_numeric_column() { let logical = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["id"])); - let mut values = lift(&logical.output_schema().unwrap()); + let mut values = logical.schema.clone(); values.fields[0].dtype = FieldDataType::Plain(DataType::Int64); assert_eq!(ranking_score_index(&logical, &values).unwrap(), 1); } // A heap's key schema is derived from its encoded item, not all label columns. #[test] - fn heap_readout_preserves_numeric_item_identity() { - let mut raw = metric_scan(&["id", "description"]); - let QueryExpr::Scan { schema, .. } = &mut raw else { - unreachable!() - }; + fn heap_evaluation_preserves_numeric_item_identity() { + let mut schema = metric_scan(&["id", "description"]).schema.clone(); schema.fields[2].dtype = FieldDataType::Plain(DataType::Int64); + let raw = crate::test_support::scan("m", schema); let node = agg( vec![], AggIntent::TopK { @@ -11288,7 +11329,7 @@ mod tests { agg(vec![2], AggIntent::Sum { col: None }, raw.clone()), ); let input = PhysicalSummaryInput { - child: Rc::new(raw), + child: raw, input: SummaryUpdate { item: Some(SummaryInputExpr::Column(ColumnRef::Named("id".into()))), weight: SummaryInputExpr::Constant(1.0), @@ -11297,7 +11338,7 @@ mod tests { }, }, }; - let schema = keyed_heap_readout_schema(&input, &node).unwrap(); + let schema = keyed_heap_evaluation_schema(&input, &node).unwrap(); assert_eq!( schema .fields @@ -11311,4 +11352,44 @@ mod tests { FieldDataType::Plain(DataType::Int64) ); } + + // Every SummaryAgg a strategy proposes derives coverage of its whole + // source: an unrestricted selection over a definition reading that source. + #[test] + fn proposed_summary_states_cover_their_whole_source() { + let root = agg( + vec![], + AggIntent::Quantile { + q: 0.9, + col: None, + accuracy: AccuracyTarget::Epsilon(0.01), + }, + metric_scan(&["job"]), + ); + let source = Source::TimeSeries { metric: "m".into() }; + let proposals = ASAPStrategies::default_cost_model().propose(&TargetSubDAG::new(&root)); + let states: Vec<_> = proposals + .candidates + .iter() + .filter_map(|candidate| match &candidate.replacement { + Replacement::SubDAG(node) => Some(node), + _ => None, + }) + .flat_map(OperatorNode::reachable) + .filter(|node| matches!(node.asap(), Some(ASAPOp::SummaryAgg { .. }))) + .collect(); + assert!(!states.is_empty()); + for state in states { + let coverage = state.coverage().expect("summary state has coverage"); + assert_eq!(coverage.selection, [Default::default()]); + let reads = OperatorNode::reachable(&coverage.definition) + .into_iter() + .filter_map(|node| match node.non_asap() { + Some(NonASAPOp::Scan { source, .. }) => Some(source.clone()), + _ => None, + }) + .collect::>(); + assert_eq!(reads, std::slice::from_ref(&source)); + } + } } diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 94a3d638c..ca752a971 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -35,7 +35,7 @@ //! - **`without(...)` grouping** leaves an `Aggregate`'s own output schema //! *open* (`closed: false`, see `without_output_schema`), while the //! `Project` this strategy always wraps the rewrite in forces -//! `closed: true` (see `QueryExpr::output_schema`'s `Project` arm). Under +//! `closed: true` (see `NonASAPOp::output_schema`'s `Project` arm). Under //! `without(...)` the rewritten form's `closed` flag would silently flip //! relative to the original — exactly the kind of schema drift this //! module exists to avoid. @@ -43,7 +43,7 @@ //! Both are follow-ups (issue #253 itself scopes to "the concrete case in //! Peilin's comment"), not correctness bugs in what ships here — a node //! outside this scope simply doesn't `match`, the same "safe but -//! uninformative" fallback [`SketchAlgorithmStrategy`]/[`SharedSubDAGStrategy`] +//! uninformative" fallback [`ASAPStrategies`]/[`SharedSubDAGStrategy`] //! already use for shapes they don't have an opinion on. //! //! ## Non-goals (mirrors [`replacement`]'s own discipline) @@ -57,14 +57,15 @@ //! the rewritten form is actually worth picking, by letting the original //! and rewritten forms compete on cost — not this strategy. +use asap_types::ir::non_asap::any_measure_filtered; use std::rc::Rc; +use asap_types::ir::operator_properties::{BinaryOpKind, Reduction}; +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ProjectItem, ScalarExpr}; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::expr_ir::ArithmeticOpKind; -use asap_types::pre_asap::query_expr::{ - any_measure_filtered, BinaryOpKind, ProjectItem, QueryExpr, Reduction, -}; use asap_types::pre_asap::schema::{ColumnId, DataType}; + use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; @@ -74,15 +75,35 @@ use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, Ta /// the module docs' "Scope" for why `without(...)`/`PerEntity` are /// excluded). Returns the grouping key count and the summed column so /// [`build_rewrite`] doesn't have to re-match. -fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { - let QueryExpr::Aggregate { +/// `a / b` with PromQL arithmetic semantics and no vector matching. +fn arithmetic( + op: ArithmeticOpKind, + lhs: Rc, + rhs: Rc, +) -> Option> { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Arithmetic(op), + vector_match: None, + }, + return_bool: false, + lhs, + rhs, + })) + .ok() +} + +fn avg_rewrite_target(node: &OperatorNode) -> Option<(usize, Option)> { + let Some(NonASAPOp::Aggregate { reduction, measures, filters, having: None, child, .. - } = node + }) = node.non_asap() else { return None; }; @@ -102,7 +123,7 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { // therefore be decomposed through it only when the averaged input is // provably non-null; otherwise NULL rows would incorrectly contribute to // the denominator. - let input_schema = child.output_schema().ok()?; + let input_schema = &child.schema; let value_col = col .or_else(|| input_schema.column_id("value")) .or_else(|| (0..input_schema.fields.len()).find(|i| !by.contains(i)))?; @@ -129,21 +150,21 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { /// exactly regardless of the summed column's own type (integer division /// would otherwise silently reappear whenever the input column is itself /// integer-typed: `Sum`'s output type tracks its input, `Count`'s is always -/// `Int64`, and `QueryExpr::output_schema`'s own `Arithmetic` type inference +/// `Int64`, and `ScalarExpr::scalar_type`'s own `Arithmetic` type inference /// types a `Div` of two `Int64` operands as `Int64` — the explicit operand /// `Cast` is what keeps both the division and rewritten `avg` column /// `Float64` the way the original always was, not an incidental extra step). // These are conditional physical components, never an unconditional Rewrite. // The caller must attach the finite-division execution guard before admission. -pub(crate) fn temporal_average_components(root: &Rc) -> Option> { - let QueryExpr::Aggregate { +pub(crate) fn temporal_average_components(root: &Rc) -> Option> { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, filters, child, having: None, .. - } = root.as_ref() + }) = root.non_asap() else { return None; }; @@ -153,10 +174,10 @@ pub(crate) fn temporal_average_components(root: &Rc) -> Option) -> Option) -> Option> { +fn build_rewrite(root: &Rc) -> Option> { let (group_count, col) = avg_rewrite_target(root)?; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, output_names, child, .. - } = root.as_ref() + }) = root.non_asap() else { unreachable!("avg_rewrite_target already confirmed an Aggregate shape"); }; // The original `avg` column's own name: `output_names[0]` if the // producing front end overrode it (SQL threading DataFusion's own - // generated name — see `QueryExpr::Aggregate::output_names`'s docs), + // generated name — see `NonASAPOp::Aggregate::output_names`'s docs), // else `AggIntent::Avg`'s synthetic default. Either way this is the // *only* thing about the original output column this rewrite needs to // reproduce — `AggIntent::Avg::output_column`'s `(Float64, nullable: @@ -210,77 +231,78 @@ fn build_rewrite(root: &Rc) -> Option> { .cloned() .unwrap_or_else(|| "avg".to_string()); - let sum_agg = Rc::new(QueryExpr::Aggregate { - reduction: reduction.clone(), - measures: vec![AggIntent::Sum { col }], - output_names: Vec::new(), - filters: vec![], - having: None, - child: Rc::clone(child), - }); - let count_agg = Rc::new(QueryExpr::Aggregate { - reduction: reduction.clone(), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: Vec::new(), - filters: vec![], - having: None, - child: Rc::clone(child), - }); + let sum_agg = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: reduction.clone(), + measures: vec![AggIntent::Sum { col }], + output_names: Vec::new(), + filters: vec![], + having: None, + child: Rc::clone(child), + })) + .ok()?; + let count_agg = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: reduction.clone(), + measures: vec![AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }], + output_names: Vec::new(), + filters: vec![], + having: None, + child: Rc::clone(child), + })) + .ok()?; let sum_idx = group_count; let mut cols: Vec = (0..group_count) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: ScalarExpr::Column(i), }) .collect(); cols.push(ProjectItem { alias: Some(avg_name), - expr: QueryExpr::Cast { - expr: Rc::new(QueryExpr::Column(sum_idx)), + expr: ScalarExpr::Cast { + expr: Box::new(ScalarExpr::Column(sum_idx)), to: DataType::Float64, try_cast: false, }, }); - let float_sum = Rc::new(QueryExpr::Project { - cols, - qualifier: None, - child: sum_agg, - }); - Some(Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: float_sum, - rhs: count_agg, - vector_match: None, - })) + let float_sum = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols, + qualifier: None, + child: sum_agg, + })) + .ok()?; + arithmetic(ArithmeticOpKind::Div, float_sum, count_agg) } /// Compose adjacent per-entity and cross-entity accumulators when their /// algebra, rather than a query-language spelling, proves equivalence. -pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { - let original_schema = root.output_schema().ok()?; - let QueryExpr::Aggregate { +pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { + let original_schema = &root.schema; + let Some(NonASAPOp::Aggregate { reduction: outer_reduction @ Reduction::Reduce(_), measures: outer_measures, output_names, filters: outer_filters, having: None, child, - } = root.as_ref() + }) = root.non_asap() else { return None; }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: inner_measures, filters: inner_filters, having: None, child: inner_child, .. - } = child.as_ref() + }) = child.non_asap() else { return None; }; @@ -297,14 +319,16 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option inner.clone(), _ => return None, }; - let aggregate = Rc::new(QueryExpr::Aggregate { - reduction: outer_reduction.clone(), - measures: vec![composed], - output_names: output_names.clone(), - filters: vec![], - having: None, - child: Rc::clone(inner_child), - }); + let aggregate = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: outer_reduction.clone(), + measures: vec![composed], + output_names: output_names.clone(), + filters: vec![], + having: None, + child: Rc::clone(inner_child), + })) + .ok()?; // The outer Sum sees PromQL's Float64 sample value, whereas the composed // Count accumulator is Int64. Keep the original observable type. @@ -321,7 +345,7 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option = (0..by.keys().len()) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: ScalarExpr::Column(i), }) .collect(); cols.push(ProjectItem { @@ -332,17 +356,18 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option) -> Option) -> Option QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } + use crate::test_support::metric_scan; + use asap_types::ir::TimeRangeKind; + + fn avg_agg( + by: Vec, + col: Option, + child: Rc, + ) -> Rc { + avg_agg_with(by, col, vec![], None, child) } - fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn avg_agg_with( + by: Vec, + col: Option, + output_names: Vec, + having: Option, + child: Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![AggIntent::Avg { col }], - output_names: vec![], + output_names, filters: vec![], - having: None, - child: Rc::new(child), - } + having, + child, + })) + .unwrap() } // Temporal averages expose two single-measure children without closing labels. #[test] fn temporal_average_components_preserves_schema_and_exposes_sum_count() { - let root = Rc::new(lower_promql( - "avg_over_time(a{job=\"api\"}[5m])", - AccuracyTarget::Exact, - )); + let root = lower_promql("avg_over_time(a{job=\"api\"}[5m])", AccuracyTarget::Exact); assert!(SemanticEquivalentRewriteStrategy .replacements(&TargetSubDAG::new(&root)) .is_empty()); let rewritten = temporal_average_components(&root).expect("conditional sum/count components"); - assert_eq!( - root.output_schema().unwrap(), - rewritten.output_schema().unwrap() - ); - assert!(matches!(rewritten.as_ref(), QueryExpr::BinaryOp { .. })); + assert_eq!(root.schema.clone(), rewritten.schema.clone()); + assert!(matches!( + rewritten.non_asap(), + Some(NonASAPOp::BinaryOp { .. }) + )); } // ── matches ────────────────────────────────────────────────────────── #[test] fn matches_a_bare_avg_aggregate() { - let q = Rc::new(avg_agg(vec![], None, metric_scan(&[]))); + let q = avg_agg(vec![], None, metric_scan(&[])); let target = TargetSubDAG::new(&q); assert!(AvgToSumOverCountStrategy.matches(&target)); } #[test] fn matches_a_grouped_avg_aggregate() { - let q = Rc::new(avg_agg(vec![2], None, metric_scan(&["job"]))); + let q = avg_agg(vec![2], None, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); assert!(AvgToSumOverCountStrategy.matches(&target)); } #[test] fn does_not_match_a_multi_measure_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -494,13 +518,15 @@ mod tests { #[test] fn does_not_match_a_having_bearing_avg_aggregate() { - let mut q = avg_agg(vec![2], None, metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { - *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), - ))); - } - let q = Rc::new(q); + let q = avg_agg_with( + vec![2], + None, + vec![], + Some(asap_types::ir::Predicate(ScalarExpr::Literal( + asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true), + ))), + metric_scan(&["job"]), + ); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -515,14 +541,16 @@ mod tests { }, AggIntent::Min { col: None }, ] { - let q = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![intent.clone()], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }); + let q = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![intent.clone()], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!( !AvgToSumOverCountStrategy.matches(&target), @@ -534,16 +562,17 @@ mod tests { #[test] fn does_not_match_a_without_grouped_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(asap_types::pre_asap::query_expr::GroupKeys::without( + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(asap_types::ir::operator_properties::GroupKeys::without( vec![2], )), measures: vec![AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -551,14 +580,15 @@ mod tests { #[test] fn does_not_match_a_per_entity_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&[])), - }); + child: metric_scan(&[]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -566,7 +596,7 @@ mod tests { #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -579,7 +609,7 @@ mod tests { #[test] fn avg_rewrites_and_schema_matches_exactly_when_ungrouped() { let original = avg_agg(vec![], None, metric_scan(&[])); - let original_rc = Rc::new(original.clone()); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); @@ -587,28 +617,26 @@ mod tests { assert!(!replacements[0].rationale.is_empty()); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = rewritten.non_asap() else { panic!("expected sum/count BinaryOp, got {rewritten:?}"); }; - let QueryExpr::Project { child: sum, .. } = lhs.as_ref() else { + let Some(NonASAPOp::Project { child: sum, .. }) = lhs.non_asap() else { panic!("expected cast Project above Sum, got {lhs:?}"); }; - assert!(matches!( - sum.as_ref(), - QueryExpr::Aggregate { measures, .. } + assert!(matches!(sum.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Sum { col: None }]) )); - assert!(matches!( - rhs.as_ref(), - QueryExpr::Aggregate { measures, .. } + assert!(matches!(rhs.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Count { accuracy: AccuracyTarget::Exact }]) )); - let original_schema = original.output_schema().unwrap(); - let rewritten_schema = rewritten.output_schema().unwrap(); + let original_schema = original.schema.clone(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!( original_schema, rewritten_schema, "the rewritten DAG must report exactly the same output schema as the original avg" @@ -620,20 +648,22 @@ mod tests { /// synthetic `"avg"` default. #[test] fn preserves_an_explicit_output_name_override() { - let mut q = avg_agg(vec![], None, metric_scan(&[])); - if let QueryExpr::Aggregate { output_names, .. } = &mut q { - *output_names = vec!["avg_latency".to_string()]; - } - let original_schema = q.output_schema().unwrap(); - let q = Rc::new(q); + let q = avg_agg_with( + vec![], + None, + vec!["avg_latency".to_string()], + None, + metric_scan(&[]), + ); + let original_schema = q.schema.clone(); let target = TargetSubDAG::new(&q); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!(original_schema, rewritten_schema); assert_eq!(rewritten_schema.fields[0].name, "avg_latency"); } @@ -643,23 +673,23 @@ mod tests { #[test] fn grouped_avg_rewrite_preserves_the_whole_schema() { let original = avg_agg(vec![2], None, metric_scan(&["job"])); - let original_schema = original.output_schema().unwrap(); - let original_rc = Rc::new(original); + let original_schema = original.schema.clone(); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!(rewritten_schema, original_schema); } #[test] fn default_search_discovers_bindable_sum_and_count_targets() { - let root = Rc::new(avg_agg(vec![2], None, metric_scan(&["job"]))); + let root = avg_agg(vec![2], None, metric_scan(&["job"])); let space = crate::replacement::search_workload(vec![("avg", Rc::clone(&root))]); let avg_group = space @@ -672,7 +702,7 @@ mod tests { let mut found_sum = false; let mut found_count = false; for group in space.target_subdag_candidates() { - let QueryExpr::Aggregate { measures, .. } = group.target.as_ref() else { + let Some(NonASAPOp::Aggregate { measures, .. }) = group.target.non_asap() else { continue; }; let expected = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) @@ -689,7 +719,8 @@ mod tests { group .candidates .iter() - .any(|candidate| matches!(candidate.replacement, Replacement::Summary(_))), + .any(|candidate| matches!(&candidate.replacement, + Replacement::SubDAG(node) if node.contains_asap())), "rewritten accumulator must be independently bindable: {measures:?}" ); found_sum |= matches!(measures.as_slice(), [AggIntent::Sum { .. }]); @@ -710,25 +741,26 @@ mod tests { Field::plain("job", DataType::Utf8, true), Field::plain("bytes", DataType::Int64, false), ]; - let child = QueryExpr::Scan { + let child = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: { let cols = std::mem::take(&mut schema_cols); Schema::with_time_index(cols, 0, vec![]) }, - }; + })) + .unwrap(); let original = avg_agg(vec![1], Some(2), child); - let original_schema = original.output_schema().unwrap(); - let original_rc = Rc::new(original); + let original_schema = original.schema.clone(); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); // The whole reason for the explicit `Cast` in `build_rewrite`: an // `Int64` input column (`bytes`) makes `Sum`'s own output `Int64` @@ -741,15 +773,15 @@ mod tests { DataType::Float64 ); - let QueryExpr::BinaryOp { lhs, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, .. }) = rewritten.non_asap() else { panic!("expected sum/count BinaryOp"); }; - let QueryExpr::Project { cols, .. } = lhs.as_ref() else { + let Some(NonASAPOp::Project { cols, .. }) = lhs.non_asap() else { panic!("expected cast Project above Sum"); }; assert!(matches!( &cols.last().unwrap().expr, - QueryExpr::Cast { + ScalarExpr::Cast { to: DataType::Float64, .. } @@ -758,7 +790,7 @@ mod tests { #[test] fn does_not_rewrite_avg_of_a_nullable_column_via_count_star() { - let child = QueryExpr::Scan { + let child = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -770,27 +802,34 @@ mod tests { 0, vec![], ), - }; - let q = Rc::new(avg_agg(vec![], Some(2), child)); + })) + .unwrap(); + let q = avg_agg(vec![], Some(2), child); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); } - fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { - let temporal = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![inner], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["service"])), - }), - }; - Rc::new(QueryExpr::Aggregate { + fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { + let temporal = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![inner], + output_names: vec![], + filters: vec![], + having: None, + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + range: Duration::from_secs(300), + child: metric_scan(&["service"]), + }, + )) + .unwrap(), + })) + .unwrap(); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![outer], // Match the PromQL front end: an empty entry selects the intent's @@ -798,8 +837,9 @@ mod tests { output_names: vec![String::new()], filters: vec![], having: None, - child: Rc::new(temporal), - }) + child: temporal, + })) + .unwrap() } #[test] @@ -820,34 +860,30 @@ mod tests { let [candidate] = candidates.as_slice() else { panic!("supported pair should produce exactly one rewrite") }; - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDAG(rewritten) = &candidate.replacement else { panic!("expected a logical rewrite") }; - assert_eq!( - original.output_schema().unwrap(), - rewritten.output_schema().unwrap() - ); - let aggregate = match rewritten.as_ref() { - QueryExpr::Aggregate { .. } => rewritten.as_ref(), - QueryExpr::Project { child, .. } => child.as_ref(), + assert_eq!(original.schema.clone(), rewritten.schema.clone()); + let aggregate = match rewritten.non_asap() { + Some(NonASAPOp::Aggregate { .. }) => rewritten.as_ref(), + Some(NonASAPOp::Project { child, .. }) => child.as_ref(), other => panic!("expected Aggregate or cast Project, got {other:?}"), }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, child, .. - } = aggregate + }) = aggregate.non_asap() else { panic!("expected composed cross-entity aggregate") }; assert_eq!(by.keys(), &[2]); assert_eq!(measures, &[expected]); - assert!(matches!( - child.as_ref(), - QueryExpr::TimeRange { range, child } + assert!(matches!(child.non_asap(), + Some(NonASAPOp::TimeRange { range, child, .. }) if *range == Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Scan { .. }) + && matches!(child.non_asap(), Some(NonASAPOp::Scan { .. })) )); } } @@ -880,9 +916,9 @@ mod tests { .iter() .find(|candidate| candidate.strategy == "SemanticEquivalentRewriteStrategy") .expect("default search should run semantic rewrites"); - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDAG(rewritten) = &candidate.replacement else { panic!("expected logical rewrite") }; - assert_eq!(rewritten.output_schema().unwrap().fields[1].name, "sum"); + assert_eq!(rewritten.schema.clone().fields[1].name, "sum"); } } diff --git a/crates/asap-aware-mapping/src/rollup.rs b/crates/asap-aware-mapping/src/rollup.rs index ab6eff3d9..09dc16fa4 100644 --- a/crates/asap-aware-mapping/src/rollup.rs +++ b/crates/asap-aware-mapping/src/rollup.rs @@ -88,7 +88,7 @@ //! - **No materialized roll-up operator.** Actually building a pre-aggregated //! summary/scan leaf at execution time is separate, larger work outside //! `asap-aware-mapping`'s scope (see issue #254's own "Non-goal" section) -//! — this module only constructs the pre-ASAP [`QueryExpr::Aggregate`] +//! — this module only constructs the pre-ASAP `NonASAPOp::Aggregate` //! rewrite; a `CostModel`/search engine decides whether to prefer it. //! - **No cross-schema reconciliation** (see "`ColumnId` comparability" //! above) and **no `without(...)` grouping support** — `without`'s kept @@ -97,34 +97,37 @@ //! against a superset/subset relationship at all; [`is_legal_rollup_source`] //! declines both directions. +use asap_types::ir::non_asap::any_measure_filtered; use std::collections::HashSet; use std::rc::Rc; +use asap_types::ir::operator_properties::{GroupKeys, Reduction}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, GroupKeys, QueryExpr, Reduction}; use asap_types::pre_asap::schema::{ColumnId, Schema}; + use asap_types::types::AccuracyTarget; use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; /// The `(by, intent, child)` shape this strategy operates on: a single /// measure, no `HAVING` — the same bindable shape -/// [`crate::replacement::SketchAlgorithmStrategy`] requires (see that module's +/// [`crate::replacement::ASAPStrategies`] requires (see that module's /// private `bindable_intent`) — **plus** a genuine [`Reduction::Reduce`] /// grouping to compare (not [`Reduction::PerEntity`], which has no `by` set /// at all). `None` for anything else, including a multi-measure or `HAVING` /// aggregate, a non-`Aggregate` node, or a `PerEntity` reduction. fn bindable_grouped_aggregate( - node: &QueryExpr, -) -> Option<(&GroupKeys, &AggIntent, &Rc)> { - let QueryExpr::Aggregate { + node: &OperatorNode, +) -> Option<(&GroupKeys, &AggIntent, &Rc)> { + let Some(NonASAPOp::Aggregate { reduction, measures, filters, having, child, .. - } = node + }) = node.non_asap() else { return None; }; @@ -263,7 +266,7 @@ fn is_strict_column_superset(finer: &[ColumnId], coarser: &[ColumnId]) -> bool { /// docs' "Non-goals" on why finding the full sibling set across a workload /// is a workload-wide traversal this strategy does not own. pub struct RollupStrategy { - siblings: Vec>, + siblings: Vec>, } impl RollupStrategy { @@ -271,7 +274,7 @@ impl RollupStrategy { /// each as a candidate roll-up source (or target) — typically the full set of `Aggregate` /// nodes a workload-wide discovery pass (issue #252) already found /// sharing at least one child `Rc` with something else. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -280,7 +283,7 @@ impl RollupStrategy { /// Every sibling that is a legal, strictly finer roll-up source for /// `target` — shared between `matches` and `replacements` so the two /// can never disagree about which siblings qualify. - fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { + fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { let Some((coarser_by, coarser_intent, coarser_child)) = bindable_grouped_aggregate(target.root) else { @@ -301,12 +304,9 @@ impl RollupStrategy { if !Rc::ptr_eq(finer_child, coarser_child) && finer_child != coarser_child { return false; } - let Ok(finer_schema) = candidate.output_schema() else { - return false; - }; is_legal_rollup_source( finer_by, - &finer_schema, + &candidate.schema, finer_intent, coarser_by, coarser_intent, @@ -325,7 +325,7 @@ impl ReplacementStrategy for RollupStrategy { let Some((coarser_by, coarser_intent, _)) = bindable_grouped_aggregate(target.root) else { return Vec::new(); }; - let QueryExpr::Aggregate { output_names, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { output_names, .. }) = target.root.non_asap() else { unreachable!("bindable_grouped_aggregate already confirmed Aggregate"); }; self.finer_sources(target) @@ -335,7 +335,7 @@ impl ReplacementStrategy for RollupStrategy { } } -/// Build the coarser replacement: a new `QueryExpr::Aggregate` grouped by +/// Build the coarser replacement: a new `NonASAPOp::Aggregate` grouped by /// `coarser_by`'s columns (repositioned into `finer`'s own output schema — /// see below), computing `rollup_combinator(intent, ..)` over `finer`'s own /// measure column, with `child = finer` instead of the original shared @@ -350,7 +350,7 @@ impl ReplacementStrategy for RollupStrategy { /// position in the shared child to its position in `finer`'s output: the /// index its `ColumnId` occupies within `finer_by`'s own ordered list. fn build_rollup( - finer: &Rc, + finer: &Rc, coarser_by: &GroupKeys, intent: &AggIntent, output_names: &[String], @@ -367,18 +367,20 @@ fn build_rollup( .map(|id| finer_by.keys().iter().position(|f| f == id)) .collect::>>()?; - let rewritten = QueryExpr::Aggregate { - reduction: Reduction::by(remapped_by), - measures: vec![combinator], - output_names: output_names.to_vec(), - filters: vec![], - having: None, - child: Rc::clone(finer), - }; + let rewritten = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(remapped_by), + measures: vec![combinator], + output_names: output_names.to_vec(), + filters: vec![], + having: None, + child: Rc::clone(finer), + })) + .ok()?; Some(ReplacementSubDAG { strategy: "RollupStrategy", - replacement: Replacement::Rewrite(Rc::new(rewritten)), + replacement: Replacement::SubDAG(rewritten), provenance: crate::replacement::ReplacementProvenance::LogicalRewrite, rationale: format!( "rolls up from the finer Aggregate grouped by {:?} (a strict superset of this \ @@ -394,13 +396,13 @@ fn build_rollup( #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::query_expr::Source; + use asap_types::ir::operator_properties::Source; use asap_types::pre_asap::schema::{DataType, Field}; use asap_types::types::AccuracyTarget; /// `[ts(0), value(1), job(2), region(3)]`. - fn metric_scan() -> QueryExpr { - QueryExpr::Scan { + fn metric_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -413,33 +415,36 @@ mod tests { 0, vec![], ), - } + })) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } fn without_agg( excluded: Vec, intent: AggIntent, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![intent], output_names: vec![], filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } // ── is_legal_rollup_source (the standalone predicate) ─────────────── @@ -569,7 +574,7 @@ mod tests { #[test] fn superset_by_over_identical_mergeable_intent_and_shared_child_rolls_up() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); @@ -581,16 +586,16 @@ mod tests { let replacements = strategy.replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDAG(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, child, having, .. - } = rewritten.as_ref() + }) = rewritten.non_asap() else { panic!("expected an Aggregate rewrite, got {rewritten:?}"); }; @@ -620,7 +625,7 @@ mod tests { // Count is not self-combining (see the module docs) — the rewritten // measure must be Sum over the finer Count's own output column, not // Count reapplied. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg( vec![2, 3], AggIntent::Count { @@ -642,10 +647,10 @@ mod tests { let replacements = strategy.replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDAG(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { measures, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::Aggregate { measures, .. }) = rewritten.non_asap() else { panic!("expected an Aggregate rewrite"); }; assert_eq!( @@ -657,7 +662,7 @@ mod tests { #[test] fn approximate_count_does_not_roll_up_via_sum() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let intent = AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), }; @@ -673,8 +678,8 @@ mod tests { #[test] fn default_workload_search_adds_rollup_for_two_query_workload() { - let fine_scan = Rc::new(metric_scan()); - let coarse_scan = Rc::new(metric_scan()); + let fine_scan = metric_scan(); + let coarse_scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &fine_scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &coarse_scan); @@ -682,12 +687,11 @@ mod tests { let coarse_group = space .target_subdag_candidates() .find(|group| { - matches!( - group.target.as_ref(), - QueryExpr::Aggregate { + matches!(group.target.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2] + }) if by.keys() == [2] ) }) .expect("coarser aggregate group"); @@ -696,19 +700,19 @@ mod tests { .candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Rewrite(rewrite) => Some(rewrite), - Replacement::Summary(_) | Replacement::ExactComposition(_) => None, + // Old `Replacement::Rewrite`: a pure pre-ASAP sub-DAG. + Replacement::SubDAG(rewrite) if !rewrite.contains_asap() => Some(rewrite), + Replacement::SubDAG(_) | Replacement::ExactComposition(_) => None, }) .expect("default search must include the roll-up rewrite"); - let QueryExpr::Aggregate { child, .. } = rewrite.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = rewrite.non_asap() else { panic!("expected aggregate rewrite, got {rewrite:?}"); }; - assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { + assert!(matches!(child.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2, 3] + }) if by.keys() == [2, 3] )); } @@ -717,18 +721,17 @@ mod tests { let intent = AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), }; - let fine = agg(vec![2, 3], intent.clone(), &Rc::new(metric_scan())); - let coarse = agg(vec![2], intent, &Rc::new(metric_scan())); + let fine = agg(vec![2, 3], intent.clone(), &metric_scan()); + let coarse = agg(vec![2], intent, &metric_scan()); let space = crate::replacement::search_workload(vec![("fine", fine), ("coarse", coarse)]); let coarse_group = space .target_subdag_candidates() .find(|group| { - matches!( - group.target.as_ref(), - QueryExpr::Aggregate { + matches!(group.target.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2] + }) if by.keys() == [2] ) }) .expect("coarser aggregate group"); @@ -736,32 +739,35 @@ mod tests { assert!(coarse_group .candidates .iter() - .all(|candidate| !matches!(candidate.replacement, Replacement::Rewrite(_)))); + .all(|candidate| !matches!(&candidate.replacement, + Replacement::SubDAG(rewrite) if !rewrite.contains_asap()))); } #[test] fn rollup_preserves_the_coarser_output_name() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let coarse = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![AggIntent::Sum { col: Some(1) }], - output_names: vec!["total_requests".into()], - filters: vec![], - having: None, - child: Rc::clone(&scan), - }); - let original_schema = coarse.output_schema().unwrap(); + let coarse = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![AggIntent::Sum { col: Some(1) }], + output_names: vec!["total_requests".into()], + filters: vec![], + having: None, + child: Rc::clone(&scan), + })) + .unwrap(); + let original_schema = coarse.schema.clone(); let siblings = vec![Rc::clone(&fine), Rc::clone(&coarse)]; let strategy = RollupStrategy::new(&siblings); let replacements = strategy.replacements(&TargetSubDAG::new(&coarse)); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDAG(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Aggregate { output_names, .. } = rewritten.as_ref() else { + assert_eq!(rewritten.schema.clone(), original_schema); + let Some(NonASAPOp::Aggregate { output_names, .. }) = rewritten.non_asap() else { unreachable!(); }; assert_eq!(output_names, &vec!["total_requests".to_string()]); @@ -769,7 +775,7 @@ mod tests { #[test] fn non_mergeable_intent_does_not_roll_up() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Avg { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Avg { col: Some(1) }, &scan); @@ -789,12 +795,12 @@ mod tests { // numerically a superset of the coarser side's *kept* positions — // `is_legal_rollup_source` rejects any `without` grouping outright, // and would reject on the missing unique key regardless. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = without_agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); assert!( - !fine.output_schema().unwrap().has_unique_key(), + !fine.schema.clone().has_unique_key(), "fixture sanity: a without(...) aggregate has no provable unique key" ); @@ -809,7 +815,7 @@ mod tests { #[test] fn unrelated_by_sets_do_not_roll_up() { // Neither `[job]` nor `[region]` is a superset of the other. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let a = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); let b = agg(vec![3], AggIntent::Sum { col: Some(1) }, &scan); @@ -830,7 +836,7 @@ mod tests { // Equal groupings are `SharedSubDAGStrategy`'s CSE-sharing // question (build once and share, or build independently) — a // roll-up requires a *strict* superset, not equality. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let a = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); let b = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); @@ -845,16 +851,8 @@ mod tests { // Scans without unique keys are deliberately not pointer-aliased by // CSE. Structural equality still proves identical schemas and makes // the two aggregates' positional ColumnIds comparable. - let fine = agg( - vec![2, 3], - AggIntent::Sum { col: Some(1) }, - &Rc::new(metric_scan()), - ); - let coarse = agg( - vec![2], - AggIntent::Sum { col: Some(1) }, - &Rc::new(metric_scan()), - ); + let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &metric_scan()); + let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &metric_scan()); let siblings = vec![Rc::clone(&fine), Rc::clone(&coarse)]; let strategy = RollupStrategy::new(&siblings); @@ -865,21 +863,23 @@ mod tests { #[test] fn does_not_match_a_multi_measure_or_having_aggregate() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let multi = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![ - AggIntent::Sum { col: Some(1) }, - AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }, - ], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::clone(&scan), - }); + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![ + AggIntent::Sum { col: Some(1) }, + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }, + ], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::clone(&scan), + })) + .unwrap(); let siblings = vec![Rc::clone(&fine), Rc::clone(&multi)]; let strategy = RollupStrategy::new(&siblings); diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs index eac94a2cb..911d90275 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs @@ -1,7 +1,7 @@ use super::*; pub(super) fn estimate_heterogeneous_summary( - root: &SummaryNode, + root: &OperatorNode, deployments: &[CostedSummaryDeployment<'_>], evidence: &SummaryNodeEvidence, scope: &ComparisonScope, @@ -19,46 +19,6 @@ pub(super) fn estimate_heterogeneous_summary( .map(|(deployment, framework)| (deployment.summary as *const _, framework)) .collect(); validate_summary_edges_and_physical_ids(root, evidence, &frameworks_by_node)?; - fn summary_source_selections( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut Vec, - ) -> Result<(), AnalyticalCostError> { - if !seen.insert(node as *const _) { - return Ok(()); - } - match &node.expr { - SummaryExpr::KeepPreAsap(query) => query_source_selections(query, out)?, - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - summary_source_selections(child, seen, out)? - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - summary_source_selections(child, seen, out)?; - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - summary_source_selections(left, seen, out)?; - summary_source_selections(right, seen, out)?; - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - summary_source_selections(summary_input, seen, out)? - } - } - Ok(()) - } let evaluation_count = scope.validate()?; let by_node: HashMap<_, _> = deployments .iter() @@ -81,7 +41,7 @@ pub(super) fn estimate_heterogeneous_summary( let node_evidence = evidence .aggregation(deployment.summary) .ok_or(AnalyticalCostError::MissingOrStale("summary_agg"))?; - let SummaryExpr::SummaryAgg { child, .. } = &deployment.summary.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &deployment.summary.operator else { return Err(AnalyticalCostError::UnsupportedCandidate); }; let inputs = node_evidence.inputs.validate()?; @@ -95,7 +55,7 @@ pub(super) fn estimate_heterogeneous_summary( .ok_or(AnalyticalCostError::MissingComparisonScope( "summary scan selection", ))?; - if !matches!(&child.expr, SummaryExpr::KeepPreAsap(_)) + if !has_retained_subdag_evidence(child, evidence) || inputs.initial_input_rows != raw.planning_time_input_rows || inputs.initial_input_bytes != raw.planning_time_input_bytes || inputs.initial_source_scan_bytes != raw.planning_time_source_scan_bytes @@ -106,7 +66,7 @@ pub(super) fn estimate_heterogeneous_summary( )); } let mut actual_selections = Vec::new(); - summary_source_selections(child, &mut HashSet::new(), &mut actual_selections)?; + query_source_selections(child, &mut HashSet::new(), &mut actual_selections)?; let actual_selections = deduplicate_source_selections(actual_selections); let expected = ( declared.source.clone(), @@ -120,7 +80,7 @@ pub(super) fn estimate_heterogeneous_summary( } } None => { - if matches!(&child.expr, SummaryExpr::KeepPreAsap(_)) + if has_retained_subdag_evidence(child, evidence) || inputs.initial_source_scan_bytes != 0 || !node_evidence.bootstrap_read_identity.is_empty() { @@ -131,7 +91,7 @@ pub(super) fn estimate_heterogeneous_summary( } } validate_guarantee(deployment.guarantee, scope.data_arrival)?; - let logical_state = format!("{:?}", deployment.summary.expr); + let logical_state = format!("{:?}", deployment.summary.operator); let window_framework = (*frameworks_by_node .get(&(deployment.summary as *const _)) .ok_or(AnalyticalCostError::MissingOrStale( @@ -253,9 +213,9 @@ pub(super) fn estimate_heterogeneous_summary( #[expect(clippy::too_many_arguments, reason = "CPU and I/O traversal state")] fn visit_ops( - node: &SummaryNode, + node: &OperatorNode, seen: &mut HashSet, - by_node: &HashMap<*const SummaryNode, &CostedSummaryDeployment<'_>>, + by_node: &HashMap<*const OperatorNode, &CostedSummaryDeployment<'_>>, evidence: &SummaryNodeEvidence, scope: &ComparisonScope, evaluation_count: u64, @@ -266,13 +226,29 @@ pub(super) fn estimate_heterogeneous_summary( if !seen.insert(physical_id) { return Ok(()); } - match &node.expr { - SummaryExpr::BinaryOp { lhs, rhs, .. } - | SummaryExpr::RelationalJoin { + if has_retained_subdag_evidence(node, evidence) { + let retained = evidence + .retained_queries + .get(&(node as *const _)) + .ok_or(AnalyticalCostError::MissingOrStale("retain_exact"))?; + if !retained.preprocessing_cpu_ops_over_horizon.is_finite() + || retained.preprocessing_cpu_ops_over_horizon < 0.0 + { + return Err(AnalyticalCostError::InvalidOperationCost( + "retain_exact", + retained.preprocessing_cpu_ops_over_horizon, + )); + } + *cpu_ops += retained.preprocessing_cpu_ops_over_horizon; + return Ok(()); + } + match &node.operator { + Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) + | Operator::NonASAP(NonASAPOp::Join { left: lhs, right: rhs, .. - } => { + }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 * validated_operator_executions("exact_binary", operation)? as f64 @@ -300,39 +276,31 @@ pub(super) fn estimate_heterogeneous_summary( )?; } - SummaryExpr::ValueOperation { child, .. } => { + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. }, + ) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 * validated_operator_executions("value_operation", operation)? as f64 * validated_operator_cpu("value_operation", operation.cpu_ops)?; add_operator_io(io_bytes, operation, evaluation_count)?; - visit_ops( - child, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - SummaryExpr::KeepPreAsap(_) => { - let retained = evidence - .retained_queries - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap"))?; - if !retained.preprocessing_cpu_ops_over_horizon.is_finite() - || retained.preprocessing_cpu_ops_over_horizon < 0.0 - { - return Err(AnalyticalCostError::InvalidOperationCost( - "keep_pre_asap", - retained.preprocessing_cpu_ops_over_horizon, - )); + for child in node.children() { + visit_ops( + child, + seen, + by_node, + evidence, + scope, + evaluation_count, + cpu_ops, + io_bytes, + )?; } - *cpu_ops += retained.preprocessing_cpu_ops_over_horizon; } - SummaryExpr::SummaryAgg { child, .. } => { + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => { visit_ops( child, seen, @@ -344,7 +312,7 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryMerge { children, .. } => { + Operator::ASAP(ASAPOp::SummaryMerge { children }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); let merge = validated_operator_cpu("summary_merge", operation.cpu_ops)?; *cpu_ops += evaluation_count as f64 @@ -364,7 +332,7 @@ pub(super) fn estimate_heterogeneous_summary( )?; } } - SummaryExpr::SummarySubtract { left, right } => { + Operator::ASAP(ASAPOp::SummarySubtract { left, right }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 * validated_operator_executions("summary_subtract", operation)? as f64 @@ -391,7 +359,7 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryDelete { summary_input, .. } => { + Operator::ASAP(ASAPOp::SummaryDelete { summary_input, .. }) => { let delete = summary_operation_evidence(node, evidence)?; let SummaryOperatorEvidence::Delete { resource: operation, @@ -406,44 +374,18 @@ pub(super) fn estimate_heterogeneous_summary( .get(&(node as *const _)) .ok_or(AnalyticalCostError::MissingOrStale("summary_delete_owner"))?; fn collect_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut Vec<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + out: &mut Vec<*const OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - out.push(node as *const _); - collect_aggs(child, seen, out); - } - SummaryExpr::ValueOperation { child, .. } => collect_aggs(child, seen, out), - SummaryExpr::SummaryMerge { children, .. } => { - children - .iter() - .for_each(|child| collect_aggs(child, seen, out)); - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - collect_aggs(left, seen, out); - collect_aggs(right, seen, out); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_aggs(summary_input, seen, out) - } - SummaryExpr::KeepPreAsap(_) => {} + if matches!(node.operator, Operator::ASAP(ASAPOp::SummaryAgg { .. })) { + out.push(node as *const _); + } + for child in node.children() { + collect_aggs(child, seen, out); } } let mut reachable = Vec::new(); @@ -491,11 +433,11 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryEstimate { summary_input, .. } => { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { let operation = summary_operation_evidence(node, evidence)?.resource(); *cpu_ops += evaluation_count as f64 - * validated_operator_executions("summary_readout", operation)? as f64 - * validated_operator_cpu("summary_readout", operation.cpu_ops)?; + * validated_operator_executions("summary_evaluation", operation)? as f64 + * validated_operator_cpu("summary_evaluation", operation.cpu_ops)?; add_operator_io(io_bytes, operation, evaluation_count)?; visit_ops( summary_input, @@ -508,7 +450,7 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } - SummaryExpr::SummaryJoin { outer, inner, .. } => { + Operator::ASAP(ASAPOp::SummaryJoin { outer, inner, .. }) => { let join = evidence .joins .get(&(node as *const _)) @@ -555,6 +497,9 @@ pub(super) fn estimate_heterogeneous_summary( io_bytes, )?; } + Operator::ASAP(ASAPOp::Extension { .. }) => { + return Err(AnalyticalCostError::UnsupportedCandidate); + } } Ok(()) } @@ -613,55 +558,37 @@ fn add_operator_io( } fn validate_summary_edges_and_physical_ids( - root: &SummaryNode, + root: &OperatorNode, evidence: &SummaryNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, + frameworks_by_node: &HashMap<*const OperatorNode, &Option>, ) -> Result<(), AnalyticalCostError> { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - vec![child] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|child| child.as_ref()).collect() - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - } + fn children<'a>( + node: &'a OperatorNode, + evidence: &SummaryNodeEvidence, + ) -> Vec<&'a OperatorNode> { + summary_children(node, evidence) } fn metadata( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result<(String, Vec, EdgeStatistics), AnalyticalCostError> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - let retained = evidence - .retained_queries - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap"))?; - Ok((retained.physical_id.clone(), vec![], retained.output)) + if let Some(retained) = evidence.retained_queries.get(&(node as *const _)) { + if node.contains_asap() { + return Err(AnalyticalCostError::InvalidPhysicalDAG( + "retained sub-DAG evidence covers summary operators", + )); } - SummaryExpr::SummaryAgg { .. } => { + return Ok((retained.physical_id.clone(), vec![], retained.output)); + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => { let value = evidence .aggregations .get(&(node as *const _)) .ok_or(AnalyticalCostError::MissingOrStale("summary_agg"))?; Ok((value.physical_id.clone(), vec![value.input], value.output)) } - SummaryExpr::SummaryJoin { .. } => { + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => { let value = evidence .joins .get(&(node as *const _)) @@ -683,16 +610,16 @@ fn validate_summary_edges_and_physical_ids( } } fn visit( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, - seen: &mut HashSet<*const SummaryNode>, + frameworks_by_node: &HashMap<*const OperatorNode, &Option>, + seen: &mut HashSet<*const OperatorNode>, physical: &mut HashMap, EdgeStatistics, String)>, ) -> Result { if !seen.insert(node as *const _) { return metadata(node, evidence).map(|(_, _, output)| output); } - let child_nodes = children(node); + let child_nodes = children(node, evidence); let child_outputs = child_nodes .iter() .map(|child| visit(child, evidence, frameworks_by_node, seen, physical)) @@ -702,17 +629,18 @@ fn validate_summary_edges_and_physical_ids( .map(|child| summary_physical_id(child, evidence)) .collect::, _>>()?; let (id, inputs, output) = metadata(node, evidence)?; - let local_fingerprint = match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - format!("{:?}", evidence.retained_queries.get(&(node as *const _))) - } - SummaryExpr::SummaryAgg { .. } => { - format!("{:?}", evidence.aggregations.get(&(node as *const _))) - } - SummaryExpr::SummaryJoin { .. } => { - format!("{:?}", evidence.joins.get(&(node as *const _))) + let local_fingerprint = if has_retained_subdag_evidence(node, evidence) { + format!("{:?}", evidence.retained_queries.get(&(node as *const _))) + } else { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => { + format!("{:?}", evidence.aggregations.get(&(node as *const _))) + } + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => { + format!("{:?}", evidence.joins.get(&(node as *const _))) + } + _ => format!("{:?}", evidence.operations.get(&(node as *const _))), } - _ => format!("{:?}", evidence.operations.get(&(node as *const _))), }; // A provider identity names the complete physical operator, including // its inputs. Equal local widths/costs do not make operators consuming @@ -720,7 +648,7 @@ fn validate_summary_edges_and_physical_ids( let framework = frameworks_by_node.get(&(node as *const _)); let fingerprint = format!( "logical={:?}|framework={framework:?}|{local_fingerprint}|children={child_physical_ids:?}", - node.expr + node.operator ); if id.is_empty() || inputs != child_outputs @@ -756,20 +684,49 @@ fn validate_summary_edges_and_physical_ids( .map(|_| ()) } +/// Whether `node` is a retained non-ASAP sub-DAG costed as one unit: the +/// provider bound retained-query evidence to it instead of per-operator +/// evidence. Its children are then not visited. No ASAP descendant may be +/// hidden by this boundary; dag validation rejects such evidence. +fn has_retained_subdag_evidence(node: &OperatorNode, evidence: &SummaryNodeEvidence) -> bool { + evidence.retained_queries.contains_key(&(node as *const _)) && !node.contains_asap() +} + +/// The inputs the estimator visits below `node`: none for a retained +/// sub-DAG, every direct input otherwise. +fn summary_children<'a>( + node: &'a OperatorNode, + evidence: &SummaryNodeEvidence, +) -> Vec<&'a OperatorNode> { + if has_retained_subdag_evidence(node, evidence) { + vec![] + } else { + node.children() + .into_iter() + .map(|child| child.as_ref()) + .collect() + } +} + fn summary_physical_id( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => evidence + if has_retained_subdag_evidence(node, evidence) { + return evidence .retained_queries .get(&(node as *const _)) - .map(|value| value.physical_id.clone()), - SummaryExpr::SummaryAgg { .. } => evidence + .map(|value| value.physical_id.clone()) + .ok_or(AnalyticalCostError::MissingOrStale( + "summary physical identity", + )); + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => evidence .aggregations .get(&(node as *const _)) .map(|value| value.physical_id.clone()), - SummaryExpr::SummaryJoin { .. } => evidence + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => evidence .joins .get(&(node as *const _)) .map(|value| value.physical_id.clone()), @@ -786,45 +743,26 @@ fn summary_physical_id( /// child output buffers remain live until their final consumer executes; /// operator workspace and its output buffer coexist during that execution. pub(super) fn estimate_transient_liveness( - root: &SummaryNode, + root: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - vec![child] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|child| child.as_ref()).collect() - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - } + fn children<'a>( + node: &'a OperatorNode, + evidence: &SummaryNodeEvidence, + ) -> Vec<&'a OperatorNode> { + summary_children(node, evidence) } fn visit<'a>( - node: &'a SummaryNode, + node: &'a OperatorNode, evidence: &SummaryNodeEvidence, seen: &mut HashSet, uses: &mut HashMap, - order: &mut Vec<&'a SummaryNode>, + order: &mut Vec<&'a OperatorNode>, ) -> Result<(), AnalyticalCostError> { if !seen.insert(summary_physical_id(node, evidence)?) { return Ok(()); } - for child in children(node) { + for child in children(node, evidence) { *uses .entry(summary_physical_id(child, evidence)?) .or_default() += 1; @@ -834,28 +772,24 @@ pub(super) fn estimate_transient_liveness( Ok(()) } fn memory( - node: &SummaryNode, + node: &OperatorNode, evidence: &SummaryNodeEvidence, ) -> Result<(u64, u64), AnalyticalCostError> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => evidence + if has_retained_subdag_evidence(node, evidence) { + return evidence .retained_queries .get(&(node as *const _)) .map(|value| (value.working_memory_bytes, value.output_buffer_bytes)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap")), - SummaryExpr::SummaryAgg { .. } => Ok((0, 0)), - SummaryExpr::SummaryJoin { .. } => evidence + .ok_or(AnalyticalCostError::MissingOrStale("retain_exact")); + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => Ok((0, 0)), + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => evidence .joins .get(&(node as *const _)) .map(|value| (value.working_memory_bytes, value.output_buffer_bytes)) .ok_or(AnalyticalCostError::MissingOrStale("summary_join")), - SummaryExpr::SummaryMerge { .. } - | SummaryExpr::BinaryOp { .. } - | SummaryExpr::RelationalJoin { .. } - | SummaryExpr::ValueOperation { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryEstimate { .. } => { + _ => { let value = summary_operation_evidence(node, evidence)?.resource(); Ok((value.working_memory_bytes, value.output_buffer_bytes)) } @@ -884,7 +818,7 @@ pub(super) fn estimate_transient_liveness( live = live .checked_add(output) .ok_or(AnalyticalCostError::Overflow)?; - for child in children(node) { + for child in children(node, evidence) { let child_id = summary_physical_id(child, evidence)?; let remaining = uses.get_mut(&child_id) @@ -902,50 +836,23 @@ pub(super) fn estimate_transient_liveness( Ok(peak) } #[cfg(test)] -pub(super) fn evidence_nodes(root: &SummaryNode) -> (Vec<&SummaryNode>, Vec<&SummaryNode>) { +pub(super) fn evidence_nodes(root: &OperatorNode) -> (Vec<&OperatorNode>, Vec<&OperatorNode>) { fn visit<'a>( - node: &'a SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - aggregations: &mut Vec<&'a SummaryNode>, - joins: &mut Vec<&'a SummaryNode>, + node: &'a OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + aggregations: &mut Vec<&'a OperatorNode>, + joins: &mut Vec<&'a OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - aggregations.push(node); - visit(child, seen, aggregations, joins); - } - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, aggregations, joins), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - visit(child, seen, aggregations, joins); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - if matches!(&node.expr, SummaryExpr::SummaryJoin { .. }) { - joins.push(node); - } - visit(left, seen, aggregations, joins); - visit(right, seen, aggregations, joins); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - visit(summary_input, seen, aggregations, joins); - } - SummaryExpr::KeepPreAsap(_) => {} + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => aggregations.push(node), + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => joins.push(node), + _ => {} + } + for child in node.children() { + visit(child, seen, aggregations, joins); } } let mut aggregations = Vec::new(); @@ -961,7 +868,7 @@ struct SummaryOperationCounts { merges_per_read: u64, subtracts_per_read: u64, deletes_per_update: u64, - readouts_per_read: u64, + evaluations_per_read: u64, joins_per_read: u64, } @@ -970,7 +877,7 @@ struct SummaryOperationCounts { /// once; explicit delete frequency comes from deletion evidence. #[cfg(test)] pub(super) fn estimate_incremental_summary_maintenance( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -980,7 +887,7 @@ pub(super) fn estimate_incremental_summary_maintenance( } #[cfg(test)] pub(super) fn estimate_incremental_summary_maintenance_with_join( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -1030,10 +937,10 @@ pub(super) fn estimate_incremental_summary_maintenance_with_join( .checked_mul(fanout) .ok_or(AnalyticalCostError::Overflow)? }; - let readout = required_cpu_when( - counts.readouts_per_read, - "readout_cpu_ops", - cpu.readout_cpu_ops, + let evaluation = required_cpu_when( + counts.evaluations_per_read, + "evaluation_cpu_ops", + cpu.evaluation_cpu_ops, )?; let join_cpu = match (counts.joins_per_read, join.as_ref()) { (0, _) => 0.0, @@ -1071,7 +978,7 @@ pub(super) fn estimate_incremental_summary_maintenance_with_join( + evaluations * counts.merges_per_read as f64 * instances * merge + evaluations * counts.subtracts_per_read as f64 * instances * subtract + delete_events as f64 * counts.deletes_per_update as f64 * delete - + evaluations * counts.readouts_per_read as f64 * instances * readout + + evaluations * counts.evaluations_per_read as f64 * instances * evaluation + evaluations * counts.joins_per_read as f64 * join_cpu; if !cpu_ops.is_finite() { return Err(AnalyticalCostError::Overflow); @@ -1238,25 +1145,23 @@ fn required_cpu_when( } #[cfg(test)] -fn count_operations(root: &SummaryNode) -> Result { +fn count_operations(root: &OperatorNode) -> Result { fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, counts: &mut SummaryOperationCounts, ) -> Result<(), AnalyticalCostError> { - if !seen.insert(node as *const SummaryNode) { + if !seen.insert(node as *const OperatorNode) { return Ok(()); } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => { counts.state_builds = counts .state_builds .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(child, seen, counts)?; } - SummaryExpr::SummaryMerge { children, .. } => { + Operator::ASAP(ASAPOp::SummaryMerge { children }) => { if children.is_empty() { return Err(AnalyticalCostError::InvalidPhysicalDAG( "summary merge has no children", @@ -1266,50 +1171,44 @@ fn count_operations(root: &SummaryNode) -> Result { + Operator::ASAP(ASAPOp::SummarySubtract { .. }) => { counts.subtracts_per_read = counts .subtracts_per_read .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(left, seen, counts)?; - visit(right, seen, counts)?; - } - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - visit(lhs, seen, counts)?; - visit(rhs, seen, counts)?; } - SummaryExpr::RelationalJoin { left, right, .. } => { - visit(left, seen, counts)?; - visit(right, seen, counts)?; - } - - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, counts)?, - SummaryExpr::SummaryDelete { summary_input, .. } => { + Operator::ASAP(ASAPOp::SummaryDelete { .. }) => { counts.deletes_per_update = counts .deletes_per_update .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(summary_input, seen, counts)?; } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - counts.readouts_per_read = counts - .readouts_per_read + Operator::ASAP( + ASAPOp::SummaryEstimate { .. } | ASAPOp::FinalizeExactAccumulator { .. }, + ) => { + counts.evaluations_per_read = counts + .evaluations_per_read .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(summary_input, seen, counts)?; } - SummaryExpr::SummaryJoin { outer, inner, .. } => { + Operator::ASAP(ASAPOp::SummaryJoin { .. }) => { counts.joins_per_read = counts .joins_per_read .checked_add(1) .ok_or(AnalyticalCostError::Overflow)?; - visit(outer, seen, counts)?; - visit(inner, seen, counts)?; } + // Retained relational work, accumulator/population boundaries and + // exact query-time operators add no summary operation. + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. } + | ASAPOp::Extension { .. }, + ) => {} + } + for child in node.children() { + visit(child, seen, counts)?; } Ok(()) } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs index 7c8607685..6ae5999bd 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs @@ -134,7 +134,7 @@ pub struct SummaryOperationCpuEvidence { pub delete_events_per_second: Option, /// Concrete state instances touched by one delete event. pub delete_routing_fanout: Option, - pub readout_cpu_ops: Option, + pub evaluation_cpu_ops: Option, } /// Physical evidence for one `SummaryJoin` implementation. Total work, @@ -185,7 +185,7 @@ pub struct SummaryOperatorResourceEvidence { } /// Evidence is structured by logical summary operation so delete-only facts -/// cannot be attached to merge, subtract, or readout nodes. +/// cannot be attached to merge, subtract, or evaluation nodes. #[derive(Debug, Clone, PartialEq)] pub enum SummaryOperatorEvidence { /// Exact query-time arithmetic over two independently realized operands. @@ -201,7 +201,7 @@ pub enum SummaryOperatorEvidence { events_per_second: f64, routing_fanout: u64, }, - Readout(SummaryOperatorResourceEvidence), + Evaluation(SummaryOperatorResourceEvidence), } impl SummaryOperatorEvidence { @@ -212,7 +212,7 @@ impl SummaryOperatorEvidence { | Self::Merge(resource) | Self::Subtract(resource) | Self::Delete { resource, .. } - | Self::Readout(resource) => resource, + | Self::Evaluation(resource) => resource, } } @@ -224,7 +224,7 @@ impl SummaryOperatorEvidence { | Self::Merge(resource) | Self::Subtract(resource) | Self::Delete { resource, .. } - | Self::Readout(resource) => resource, + | Self::Evaluation(resource) => resource, } } } @@ -247,27 +247,27 @@ pub struct RetainedSubDAGEvidence { /// structurally equal node is not silently treated as the same deployment. #[derive(Debug, Clone, Default)] pub struct SummaryNodeEvidence { - pub(super) aggregations: HashMap<*const SummaryNode, SummaryAggregateEvidence>, - pub(super) joins: HashMap<*const SummaryNode, SummaryJoinEvidence>, - pub(super) operations: HashMap<*const SummaryNode, SummaryOperatorEvidence>, - pub(super) operation_state_owners: HashMap<*const SummaryNode, *const SummaryNode>, - pub(super) retained_queries: HashMap<*const SummaryNode, RetainedSubDAGEvidence>, + pub(super) aggregations: HashMap<*const OperatorNode, SummaryAggregateEvidence>, + pub(super) joins: HashMap<*const OperatorNode, SummaryJoinEvidence>, + pub(super) operations: HashMap<*const OperatorNode, SummaryOperatorEvidence>, + pub(super) operation_state_owners: HashMap<*const OperatorNode, *const OperatorNode>, + pub(super) retained_queries: HashMap<*const OperatorNode, RetainedSubDAGEvidence>, } impl SummaryNodeEvidence { pub fn insert_aggregation( &mut self, - node: &Rc, + node: &Rc, evidence: SummaryAggregateEvidence, ) { self.aggregations.insert(Rc::as_ptr(node), evidence); } - pub fn insert_join(&mut self, node: &Rc, evidence: SummaryJoinEvidence) { + pub fn insert_join(&mut self, node: &Rc, evidence: SummaryJoinEvidence) { self.joins.insert(Rc::as_ptr(node), evidence); } - pub fn insert_operation(&mut self, node: &Rc, evidence: SummaryOperatorEvidence) { + pub fn insert_operation(&mut self, node: &Rc, evidence: SummaryOperatorEvidence) { self.operations.insert(Rc::as_ptr(node), evidence); } @@ -275,8 +275,8 @@ impl SummaryNodeEvidence { /// aggregation deployment whose active interval it follows. pub fn insert_state_operation( &mut self, - node: &Rc, - state: &Rc, + node: &Rc, + state: &Rc, evidence: SummaryOperatorEvidence, ) { self.operations.insert(Rc::as_ptr(node), evidence); @@ -286,52 +286,57 @@ impl SummaryNodeEvidence { pub fn insert_retained_query( &mut self, - node: &Rc, + node: &Rc, evidence: RetainedSubDAGEvidence, ) { self.retained_queries.insert(Rc::as_ptr(node), evidence); } - pub(super) fn aggregation(&self, node: &SummaryNode) -> Option { + pub(super) fn aggregation(&self, node: &OperatorNode) -> Option { self.aggregations.get(&(node as *const _)).cloned() } } pub(super) fn summary_operation_evidence<'a>( - node: &SummaryNode, + node: &OperatorNode, evidence: &'a SummaryNodeEvidence, ) -> Result<&'a SummaryOperatorEvidence, AnalyticalCostError> { let operation = evidence .operations .get(&(node as *const _)) .ok_or(AnalyticalCostError::MissingOrStale("summary operation"))?; - let matches = matches!( - (&node.expr, operation), + // A binary operator or join over two inputs is `Binary` evidence; every + // other non-ASAP operator, and the accumulator/population boundaries, + // is a `ValueOperation`. + let matches = match (&node.operator, operation) { ( - SummaryExpr::BinaryOp { .. }, - SummaryOperatorEvidence::Binary(_) - ) | ( - SummaryExpr::ValueOperation { .. }, - SummaryOperatorEvidence::ValueOperation(_) - ) | ( - SummaryExpr::SummaryMerge { .. }, - SummaryOperatorEvidence::Merge(_) - ) | ( - SummaryExpr::SummarySubtract { .. }, - SummaryOperatorEvidence::Subtract(_) - ) | ( - SummaryExpr::SummaryDelete { .. }, - SummaryOperatorEvidence::Delete { .. } - ) | ( - SummaryExpr::SummaryEstimate { .. }, - SummaryOperatorEvidence::Readout(_) - ) - ); + Operator::NonASAP(NonASAPOp::BinaryOp { .. } | NonASAPOp::Join { .. }), + SummaryOperatorEvidence::Binary(_), + ) => true, + (Operator::NonASAP(NonASAPOp::BinaryOp { .. } | NonASAPOp::Join { .. }), _) => false, + ( + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. }, + ), + SummaryOperatorEvidence::ValueOperation(_), + ) => true, + (Operator::ASAP(ASAPOp::SummaryMerge { .. }), SummaryOperatorEvidence::Merge(_)) + | (Operator::ASAP(ASAPOp::SummarySubtract { .. }), SummaryOperatorEvidence::Subtract(_)) + | (Operator::ASAP(ASAPOp::SummaryDelete { .. }), SummaryOperatorEvidence::Delete { .. }) + | ( + Operator::ASAP(ASAPOp::SummaryEstimate { .. }), + SummaryOperatorEvidence::Evaluation(_), + ) => true, + _ => false, + }; if matches { Ok(operation) } else { Err(AnalyticalCostError::InconsistentOperatorStatistics( - "summary operation evidence kind does not match SummaryExpr", + "summary operation evidence kind does not match the operator", )) } } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs index bc1b7ddaf..ff4c69c1d 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs @@ -1,4 +1,4 @@ -//! Analytical resource cost for at-rest and incrementally maintained summary deployments. +//! Analytical resource cost for at-rest and at-rest and incrementally maintained summary deployments. //! //! The canonical workload and lifecycle types own deployment semantics. This //! module only adds physical evidence absent from those schemas: state size, @@ -7,14 +7,13 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, Predicate}; use asap_types::post_asap::{ BoundExpr, ErrorMetric, ExactKind, FieldDataType, GuaranteeSource, ProbabilityExpr, - ResultGuarantee, SketchAlgorithm, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, -}; -use asap_types::pre_asap::{ - agg_intent::AggIntent, CompareOpKind, InfoMatcher, Predicate, QueryExpr, Source, + ResultGuarantee, SketchAlgorithm, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryWindowFramework, }; +use asap_types::pre_asap::{agg_intent::AggIntent, CompareOpKind, InfoMatcher, Source}; use asap_types::types::AccuracyTarget; use asap_types::workload::{DataArrival, DataWorkload, QueryRecurrence, RepeatedDemand}; use serde::{Deserialize, Serialize}; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs index 7859db9b8..a7f386a8b 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs @@ -7,7 +7,7 @@ pub struct SummaryMaintenanceCostModel { pub node_evidence: SummaryNodeEvidence, pub calibration: ResourceCalibration, pub capabilities: SummaryMaintenanceCapabilities, - target_comparisons: HashMap<*const QueryExpr, SummaryTargetComparison>, + target_comparisons: HashMap<*const OperatorNode, SummaryTargetComparison>, candidate_comparisons: HashMap, physical_plan_alternatives: HashMap>, @@ -15,17 +15,17 @@ pub struct SummaryMaintenanceCostModel { HashMap>, } -type CandidateComparisonKey = (*const QueryExpr, *const SummaryNode); +type CandidateComparisonKey = (*const OperatorNode, *const OperatorNode); #[derive(Debug, Clone)] struct BoundCandidateIdentity { - _target: Rc, - _root: Rc, + _target: Rc, + _root: Rc, } #[derive(Debug, Clone)] struct SummaryTargetComparison { - _target: Rc, + _target: Rc, scope: ComparisonScope, raw: RawInputEvidence, } @@ -59,73 +59,40 @@ fn info_source(selector: &[InfoMatcher]) -> Result }) } +/// Collect the source selections (scan sources with their predicates, and +/// info-metric selectors) of every leaf reachable from `node`, visiting a +/// shared node once. pub(super) fn query_source_selections( - query: &QueryExpr, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, out: &mut Vec, ) -> Result<(), AnalyticalCostError> { - use QueryExpr::*; - match query { - Scan { + if !seen.insert(node as *const _) { + return Ok(()); + } + match &node.operator { + Operator::NonASAP(NonASAPOp::Scan { source, predicates, .. - } => out.push((source.clone(), predicates.clone(), vec![])), - PromqlVectorFromScalar(child) | PromqlScalarFromVector(child) => { - query_source_selections(child, out)? - } - PromqlInfoEnrich { selector, child } => { - query_source_selections(child, out)?; + }) => out.push((source.clone(), predicates.clone(), vec![])), + Operator::NonASAP(NonASAPOp::PromqlInfoEnrich { selector, child }) => { + query_source_selections(child, seen, out)?; out.push((info_source(selector)?, vec![], selector.clone())); } - PromqlRelabel { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | PromqlSeriesSample { child, .. } - | Sort { child, .. } - | Limit { child, .. } => query_source_selections(child, out)?, - Concat { children, .. } => { - for child in children { - query_source_selections(child, out)?; + _ => { + for child in node.children() { + query_source_selections(child, seen, out)?; } } - Join { left, right, .. } | SetOp { left, right, .. } => { - query_source_selections(left, out)?; - query_source_selections(right, out)?; - } - BinaryOp { lhs, rhs, .. } => { - query_source_selections(lhs, out)?; - query_source_selections(rhs, out)?; - } - PromqlScalarBridge(_) - | EvalTimestamp - | CurrentTimestamp - | Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} } Ok(()) } fn validate_query_scope( - target: &QueryExpr, + target: &OperatorNode, scope: &ComparisonScope, ) -> Result<(), AnalyticalCostError> { let mut actual = Vec::new(); - query_source_selections(target, &mut actual)?; + query_source_selections(target, &mut HashSet::new(), &mut actual)?; let actual = deduplicate_source_selections(actual); let mut declared: Vec<_> = scope .sources @@ -428,8 +395,8 @@ impl SummaryMaintenanceCostModel { /// rejected rather than silently replacing the canonical context. pub fn bind_candidate_comparison( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, scope: ComparisonScope, raw: RawInputEvidence, ) -> Result<(), AnalyticalCostError> { @@ -474,8 +441,8 @@ impl SummaryMaintenanceCostModel { /// candidate. Duplicate or empty provider identities are rejected. pub fn bind_physical_plan_alternative( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, alternative: SummaryPhysicalPlanAlternative, ) -> Result<(), AnalyticalCostError> { let key = (Rc::as_ptr(target), Rc::as_ptr(root)); @@ -512,8 +479,8 @@ impl SummaryMaintenanceCostModel { /// evidence keep the implementations distinct during ranking. pub fn bind_window_framework_candidate( &mut self, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, candidate: SummaryWindowFrameworkCandidate, ) -> Result<(), AnalyticalCostError> { let key = (Rc::as_ptr(target), Rc::as_ptr(root)); @@ -565,8 +532,8 @@ impl SummaryMaintenanceCostModel { fn comparison_context( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, horizon: Option, expected_reads: Option, ) -> Option<(CandidateComparisonKey, &SummaryTargetComparison)> { @@ -600,7 +567,7 @@ impl SummaryMaintenanceCostModel { fn complete_cost_with_evidence( &self, - root: &SummaryNode, + root: &OperatorNode, deployments: &[CostedSummaryDeployment<'_>], comparison: &SummaryTargetComparison, evidence: &SummaryNodeEvidence, @@ -619,7 +586,7 @@ impl SummaryMaintenanceCostModel { ) } - fn canonical_inputs(&self, summary: &SummaryNode) -> Option { + fn canonical_inputs(&self, summary: &OperatorNode) -> Option { let evidence = self.node_evidence.aggregation(summary)?; evidence.inputs.validate().ok()?; Some(evidence) @@ -631,7 +598,7 @@ impl SummaryMaintenanceCostModel { fn lifecycle_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, horizon: Option, ) -> Option { let evidence = self.canonical_inputs(summary)?; @@ -657,8 +624,8 @@ impl SummaryMaintenanceCostModel { Some(SummaryMaintenanceLifecycleCostInputs { build_cost: Some(build), maintenance_cost_per_update: Some(maintenance), - // Readout is a separate physical operator in the complete DAG. - // A state-only candidate therefore does not fabricate readout + // Evaluation is a separate physical operator in the complete DAG. + // A state-only candidate therefore does not fabricate evaluation // evidence merely to keep a lifecycle alternative selectable. summary_read_cost: Some(Cost::ZERO), retention_cost_rate: Some(CostRate(retention_total.0 / horizon_seconds)), @@ -681,8 +648,7 @@ impl CostModel for SummaryMaintenanceCostModel { // Lifecycle selection supplies a complete override. If it cannot, // the candidate remains unavailable rather than receiving this // trait's structural fallback. - Replacement::Summary(_) => None, - Replacement::Rewrite(_) => None, + Replacement::SubDAG(_) => None, } } @@ -700,14 +666,14 @@ impl CostModel for SummaryMaintenanceCostModel { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs::default() } fn summary_maintenance_lifecycle_cost_inputs_for_horizon( &self, - summary: &SummaryNode, + summary: &OperatorNode, horizon: Option, ) -> SummaryMaintenanceLifecycleCostInputs { self.lifecycle_inputs(summary, horizon).unwrap_or_default() @@ -715,15 +681,15 @@ impl CostModel for SummaryMaintenanceCostModel { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { self.capabilities } fn complete_summary_candidate_cost( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -742,8 +708,8 @@ impl CostModel for SummaryMaintenanceCostModel { fn complete_summary_candidate_estimate( &self, - root: &SummaryNode, - target: Option<&QueryExpr>, + root: &OperatorNode, + target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], horizon: Option, expected_reads: Option, @@ -847,14 +813,14 @@ impl CostModel for SummaryMaintenanceCostModel { true } - fn raw_query_recompute_cost(&self, target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, target: &OperatorNode) -> Option { let _ = target; None } fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &OperatorNode, expected_reads: f64, ) -> Option { let target_ptr = target as *const _; @@ -880,13 +846,14 @@ use super::estimator::*; mod tests { use std::rc::Rc; + use asap_types::ir::{ASAPOp, BinaryOperator, NonASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ EvaluationSchedule, ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, - OutputRepresentation, Schema, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, + OutputRepresentation, Schema, SummaryMaintenanceLifecycle, + SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummaryUpdate, }; use asap_types::pre_asap::{ - agg_intent::AggIntent, ColumnRef, DataType, QueryExpr, Reduction, Source, + agg_intent::AggIntent, ArithmeticOpKind, BinaryOpKind, DataType, Reduction, Source, }; use asap_types::workload::{ DataWorkload, Evidence, EvidenceSource, Predictability, Query, QueryLanguage, @@ -903,7 +870,7 @@ mod tests { }; fn estimate_test( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -912,7 +879,7 @@ mod tests { } fn estimate_join_test( - root: &SummaryNode, + root: &OperatorNode, guarantee: &SummaryMaintenanceLifecycleGuarantee, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, @@ -1200,7 +1167,7 @@ mod tests { inputs, SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -1253,7 +1220,7 @@ mod tests { inputs, SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -1442,7 +1409,10 @@ mod tests { let mut model = streaming_model(); for group in space.target_subdag_candidates() { for candidate in &group.candidates { - if let Replacement::Summary(root) = &candidate.replacement { + if let Replacement::SubDAG(root) = &candidate.replacement { + if !root.contains_asap() { + continue; + } bind_aggregations( &mut model, &group.target, @@ -1713,10 +1683,13 @@ mod tests { op: CompareOpKind::Eq, value: "api".into(), }]; - let info_target = QueryExpr::PromqlInfoEnrich { - selector: selector.clone(), - child: target, - }; + let info_target = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlInfoEnrich { + selector: selector.clone(), + child: target, + }, + )) + .unwrap(); let mut info_scope = streaming_scope(); info_scope .sources @@ -1739,8 +1712,9 @@ mod tests { } #[test] - fn delete_owner_must_be_the_unique_state_reachable_from_its_input() { - let workload = streaming_workload(); + fn summary_delete_dag_fails_closed_before_costing() { + // SummaryDelete is reserved: planning rejects the DAG even with full + // delete evidence, instead of costing (or owner-checking) the delete. let target = streaming_sum_query(); let root = summary_with_operations(false, false, true); let mut cpu = streaming_cpu(); @@ -1750,34 +1724,32 @@ mod tests { let mut model = streaming_model(); model.capabilities.delete = true; bind_aggregations(&mut model, &target, &root, streaming_inputs(), cpu); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let delete_ptr = Rc::as_ptr(summary_input); - let unrelated = summary_with_operations(false, false, false); - let unrelated_agg = evidence_nodes(&unrelated).0[0] as *const _; - model - .node_evidence - .operation_state_owners - .insert(delete_ptr, unrelated_agg); - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.summary_total_cost, None); + // Under a evaluation, the reserved delete surfaces as an illegal child. + assert!(matches!( + streaming_planning_error(Rc::clone(&root), &model), + asap_types::post_asap::ExecutionDataStateError::IllegalChildDataState { + edge: "FinalizeExactAccumulator.child", + child: asap_types::post_asap::ExecutionDataState::QUERY_ROWS, + } + )); + // As the root, it is reported as the unimplemented operator itself. + assert!(matches!( + streaming_planning_error(evaluation_state(&root), &model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryDelete" + } + )); } #[test] fn summary_edge_and_io_evidence_fail_closed() { + // Over two independent summaries combined by a BinaryOp: a parent + // input edge that disagrees with its child's output, or missing I/O + // evidence on the root, leaves the whole-DAG cost unset. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -1786,19 +1758,15 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let join = evidence_nodes(&root).1[0]; - model.node_evidence.joins.insert( - join as *const _, - SummaryJoinEvidence { - physical_id: "join-edge".into(), - inputs: vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + model.node_evidence.insert_operation( + &root, + SummaryOperatorEvidence::Binary(test_resource( + "binary-edge", + vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], + 1.0, + 1, + 0, + )), ); let bad_edge = plan_summary_maintenance_lifecycles( Rc::clone(&root), @@ -1811,20 +1779,34 @@ mod tests { .unwrap(); assert_eq!(bad_edge.summary_total_cost, None); - model + let root_evidence = model .node_evidence - .joins - .get_mut(&(join as *const _)) + .operations + .get_mut(&Rc::as_ptr(&root)) .unwrap() - .inputs = vec![test_edge(), test_edge()]; + .resource_mut(); + root_evidence.inputs = vec![test_edge(), test_edge()]; + root_evidence.io_bytes_per_execution = None; + let missing_io = plan_summary_maintenance_lifecycles( + Rc::clone(&root), + WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), + 0, + Some(Horizon(5.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &model, + ) + .unwrap(); + assert_eq!(missing_io.summary_total_cost, None); + + // Control: the same evidence with I/O restored is costable. model .node_evidence .operations .get_mut(&Rc::as_ptr(&root)) .unwrap() .resource_mut() - .io_bytes_per_execution = None; - let missing_io = plan_summary_maintenance_lifecycles( + .io_bytes_per_execution = Some(0); + let complete = plan_summary_maintenance_lifecycles( root, WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), 0, @@ -1833,14 +1815,17 @@ mod tests { &model, ) .unwrap(); - assert_eq!(missing_io.summary_total_cost, None); + assert!(complete.summary_total_cost.is_some()); } #[test] fn summary_edges_io_and_physical_identity_fail_closed() { + // Evidence bound to a structurally equal clone of the BinaryOp does not + // count for the real node; a bad input edge or missing I/O on the real + // node still fails closed. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -1849,34 +1834,27 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let (_, joins) = evidence_nodes(&root); - model.node_evidence.insert_join( - &Rc::new(joins[0].clone()), - SummaryJoinEvidence { - physical_id: "unused".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + model.node_evidence.insert_operation( + &Rc::new((*root).clone()), + SummaryOperatorEvidence::Binary(test_resource( + "unused", + vec![test_edge(), test_edge()], + 1.0, + 1, + 0, + )), ); - // Bind the actual join, then make one parent input disagree with its - // child's output. - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "join-edge".into(), - inputs: vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + // Bind the actual BinaryOp, then make one parent input disagree with + // its child's output. + model.node_evidence.insert_operation( + &root, + SummaryOperatorEvidence::Binary(test_resource( + "binary-edge", + vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], + 1.0, + 1, + 0, + )), ); let bad_edge = plan_summary_maintenance_lifecycles( Rc::clone(&root), @@ -1889,19 +1867,14 @@ mod tests { .unwrap(); assert_eq!(bad_edge.summary_total_cost, None); - model - .node_evidence - .joins - .get_mut(&(joins[0] as *const _)) - .unwrap() - .inputs = vec![test_edge(), test_edge()]; - model + let root_evidence = model .node_evidence .operations .get_mut(&Rc::as_ptr(&root)) .unwrap() - .resource_mut() - .io_bytes_per_execution = None; + .resource_mut(); + root_evidence.inputs = vec![test_edge(), test_edge()]; + root_evidence.io_bytes_per_execution = None; let missing_io = plan_summary_maintenance_lifecycles( root, WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), @@ -1955,9 +1928,11 @@ mod tests { #[test] fn conflicting_evidence_cannot_alias_one_provider_physical_identity() { + // Two independent summary states (combined by a BinaryOp) that claim + // one physical id but carry different evidence leave the cost unset. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -1967,6 +1942,7 @@ mod tests { streaming_cpu(), ); let aggregations = evidence_nodes(&root).0; + assert_eq!(aggregations.len(), 2); let first = aggregations[0] as *const _; let second = aggregations[1] as *const _; model @@ -1992,8 +1968,9 @@ mod tests { } #[test] - fn lifecycle_plan_does_not_fall_back_to_partial_agg_cost_for_a_join_root() { - let workload = streaming_workload(); + fn summary_join_root_fails_closed_even_with_join_evidence() { + // SummaryJoin is reserved: planning rejects the DAG whether or not + // join evidence is bound, so no partial or join cost is produced. let root = summary_join(); let target = streaming_sum_query(); let mut model = streaming_model(); @@ -2004,21 +1981,15 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.deployments.len(), 2); - assert_eq!(plan.summary_total_cost, None); + assert!(matches!( + streaming_planning_error(Rc::clone(&root), &model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryJoin" + } + )); - let mut costed = model; let join_node = evidence_nodes(&root).1[0]; - costed.node_evidence.joins.insert( + model.node_evidence.joins.insert( join_node as *const _, SummaryJoinEvidence { physical_id: "costed-join".into(), @@ -2031,28 +2002,28 @@ mod tests { io_bytes_per_execution: Some(0), }, ); - let costed_plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &costed, - ) - .unwrap(); - assert!(costed_plan.summary_total_cost.is_some()); + assert!(matches!( + streaming_planning_error(root, &model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryJoin" + } + )); } #[test] fn whole_dag_cost_requires_and_uses_each_rc_bound_state_evidence() { + // Over two independent summaries combined by a BinaryOp: the cost needs + // evidence for each Rc-bound state, charges peak transient memory, and + // de-duplicates bootstrap scans only on a shared provider read id. let workload = streaming_workload(); - let root = summary_join(); + let root = add_independent_summary_results(); + let (left, right) = binary_operands(&root); let target = streaming_sum_query(); - let (aggregations, joins) = evidence_nodes(&root); + let aggregations = [evaluation_state(&left), evaluation_state(&right)]; let mut model = streaming_model(); bind_comparison(&mut model, &target, &root); - model.node_evidence.aggregations.insert( - aggregations[0] as *const _, + model.node_evidence.insert_aggregation( + &aggregations[0], SummaryAggregateEvidence { physical_id: "left-state".into(), input: test_edge(), @@ -2078,8 +2049,8 @@ mod tests { second_inputs.state_bytes_per_summary = 250; let mut second_cpu = streaming_cpu(); second_cpu.insert_cpu_ops = Some(5.0); - model.node_evidence.aggregations.insert( - aggregations[1] as *const _, + model.node_evidence.insert_aggregation( + &aggregations[1], SummaryAggregateEvidence { physical_id: "right-state".into(), input: test_edge(), @@ -2090,31 +2061,37 @@ mod tests { insert_cpu_ops: second_cpu.insert_cpu_ops.unwrap(), }, ); - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "join".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 6.0, - working_memory_bytes: 64, - output_buffer_bytes: 64, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, + // The left evaluation plays the old join's role: a 64-byte workspace and a + // 64-byte output that stays live until the root BinaryOp consumes it. + model.node_evidence.insert_operation( + &left, + SummaryOperatorEvidence::ValueOperation(test_resource( + "left-evaluation", + vec![test_edge()], + 6.0, + 64, + 64, + )), ); - model.node_evidence.operations.insert( - Rc::as_ptr(&root), - SummaryOperatorEvidence::Readout(SummaryOperatorResourceEvidence { - physical_id: "root-readout".into(), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops: 3.0, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }), + model.node_evidence.insert_operation( + &right, + SummaryOperatorEvidence::ValueOperation(test_resource( + "right-evaluation", + vec![test_edge()], + 3.0, + 0, + 0, + )), + ); + model.node_evidence.insert_operation( + &root, + SummaryOperatorEvidence::Binary(test_resource( + "root-binary", + vec![test_edge(), test_edge()], + 3.0, + 0, + 0, + )), ); let complete = plan_summary_maintenance_lifecycles( Rc::clone(&root), @@ -2143,8 +2120,9 @@ mod tests { &model, ) .unwrap(); - // The join's 64-byte output remains live while the readout's workspace - // is active. The join's execution workspace is released first. + // The left evaluation's 64-byte output remains live while the binary's + // workspace is active (64 + 128 = 192); the evaluation's own workspace is + // released first, so the old peak was 64 + 64 = 128. assert_eq!( larger_workspace.summary_total_cost.unwrap().0 - complete.summary_total_cost.unwrap().0, 64.0 @@ -2184,7 +2162,10 @@ mod tests { }); let target = streaming_sum_query(); let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &root.operator + else { unreachable!(); }; let windowed_summary = Rc::clone(summary_input); @@ -2326,7 +2307,10 @@ mod tests { fn window_framework_candidates_require_unique_nonempty_planner_primitives() { let target = streaming_sum_query(); let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &root.operator + else { unreachable!(); }; let windowed_summary = Rc::clone(summary_input); @@ -2365,9 +2349,12 @@ mod tests { #[test] fn one_physical_identity_cannot_alias_different_window_frameworks() { + // Two independent summaries (combined by a BinaryOp) sharing one + // physical state id but assigned different window frameworks leave + // the cost unset. let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_join(); + let root = add_independent_summary_results(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -2376,44 +2363,22 @@ mod tests { streaming_inputs(), streaming_cpu(), ); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let SummaryExpr::SummaryJoin { outer, inner, .. } = &summary_input.expr else { - unreachable!(); - }; - let aggregation_nodes = [Rc::clone(outer), Rc::clone(inner)]; - let (aggregations, joins) = evidence_nodes(&root); - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "joined-readout".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 8, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); + let (left, right) = binary_operands(&root); + let aggregation_nodes = [evaluation_state(&left), evaluation_state(&right)]; let mut shared_aggregation = - model.node_evidence.aggregations[&(aggregations[0] as *const _)].clone(); + model.node_evidence.aggregations[&Rc::as_ptr(&aggregation_nodes[0])].clone(); shared_aggregation.physical_id = "shared-window-state".into(); - model - .node_evidence - .aggregations - .insert(aggregations[0] as *const _, shared_aggregation.clone()); - model - .node_evidence - .aggregations - .insert(aggregations[1] as *const _, shared_aggregation); + for aggregate in &aggregation_nodes { + model + .node_evidence + .insert_aggregation(aggregate, shared_aggregation.clone()); + } let retained_children: Vec<_> = aggregation_nodes .iter() - .map(|aggregate| match &aggregate.expr { - SummaryExpr::SummaryAgg { child, .. } => Rc::clone(child), + .map(|aggregate| match &aggregate.operator { + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => Rc::clone(child), _ => unreachable!(), }) .collect(); @@ -2428,7 +2393,7 @@ mod tests { } let candidate = SummaryWindowFrameworkCandidate { - physical_plan_id: "mixed-framework-join".into(), + physical_plan_id: "mixed-framework-binary".into(), assignments: vec![ SummaryWindowFrameworkAssignment { summary: Rc::clone(&aggregation_nodes[0]), @@ -2544,7 +2509,10 @@ mod tests { asap_types::workload::AccuracyRequirement::Explicit(AccuracyTarget::Epsilon(1.0)); let target = streaming_sum_query(); let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &root.operator + else { unreachable!(); }; let mut model = streaming_model(); @@ -2583,6 +2551,42 @@ mod tests { assert_eq!(plan.summary_total_cost, None); } + /// Bulk relational evidence cannot hide summary operators below an ordinary root. + #[test] + fn retained_subdag_evidence_cannot_hide_summary_work() { + let workload = streaming_workload(); + let target = streaming_sum_query(); + let root = add_shared_summary_result(); + let mut model = streaming_model(); + bind_aggregations( + &mut model, + &target, + &root, + streaming_inputs(), + streaming_cpu(), + ); + model.node_evidence.insert_retained_query( + &root, + RetainedSubDAGEvidence { + physical_id: "false-retained-root".into(), + output: test_edge(), + preprocessing_cpu_ops_over_horizon: 0.0, + working_memory_bytes: 0, + output_buffer_bytes: 0, + }, + ); + let plan = plan_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), + 0, + Some(Horizon(5.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + &model, + ) + .unwrap(); + assert_eq!(plan.summary_total_cost, None); + } + #[test] fn whole_dag_fails_closed_for_missing_retained_work_or_false_source_lineage() { let workload = streaming_workload(); @@ -2631,23 +2635,22 @@ mod tests { } #[test] - fn aggregate_recurses_into_child_operations_and_state_only_needs_no_readout() { + fn state_only_needs_no_evaluation_and_summary_merge_child_fails_closed() { + // A state-only root is costable without evaluation evidence; a + // SummaryAgg over a reserved SummaryMerge is rejected at planning. let workload = streaming_workload(); let target = streaming_sum_query(); let estimated = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &estimated.expr else { - unreachable!(); - }; - let state_only = Rc::clone(summary_input); - let mut no_readout_cpu = streaming_cpu(); - no_readout_cpu.readout_cpu_ops = None; + let state_only = evaluation_state(&estimated); + let mut no_evaluation_cpu = streaming_cpu(); + no_evaluation_cpu.evaluation_cpu_ops = None; let mut state_model = streaming_model(); bind_aggregations( &mut state_model, &target, &state_only, streaming_inputs(), - no_readout_cpu, + no_evaluation_cpu, ); let state_plan = plan_summary_maintenance_lifecycles( state_only, @@ -2660,27 +2663,24 @@ mod tests { .unwrap(); assert!(state_plan.summary_total_cost.is_some()); - let child_readout = summary_with_operations(true, false, false); - let SummaryExpr::SummaryEstimate { - summary_input: child, - .. - } = &child_readout.expr - else { - unreachable!(); - }; - let child = Rc::clone(child); - let nested = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Wildcard), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::PerSubpopulationInstance, - filter: None, - }, - schema: estimated.schema.clone(), - guarantee: None, - }); + let nested = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: evaluation_state(&summary_with_operations(true, false, false)), + family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), + input: SummaryUpdate { + item: None, + weight: asap_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }, + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::PerSubpopulationInstance, + filter: None, + }), + count_state_schema(), + ) + .with_guarantee(None), + ); let mut nested_cpu = streaming_cpu(); nested_cpu.merge_cpu_ops = Some(1.0); let mut nested_model = streaming_model(); @@ -2691,23 +2691,12 @@ mod tests { streaming_inputs(), nested_cpu, ); - nested_model - .node_evidence - .operations - .retain(|_, operation| { - operation.resource().cpu_ops != 1.0 - || operation.resource().working_memory_bytes == 0 - }); - let nested_plan = plan_summary_maintenance_lifecycles( - nested, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &nested_model, - ) - .unwrap(); - assert_eq!(nested_plan.summary_total_cost, None); + assert!(matches!( + streaming_planning_error(nested, &nested_model), + asap_types::post_asap::ExecutionDataStateError::UnimplementedOperator { + operator: "SummaryMerge" + } + )); } #[test] @@ -2744,7 +2733,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(3.0), + evaluation_cpu_ops: Some(3.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -2778,7 +2767,7 @@ mod tests { delete_cpu_ops: Some(5.0), delete_events_per_second: Some(4.0), delete_routing_fanout: Some(2), - readout_cpu_ops: Some(7.0), + evaluation_cpu_ops: Some(7.0), }, ) .unwrap(); @@ -2808,7 +2797,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ), @@ -2835,7 +2824,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ), @@ -2866,7 +2855,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ), @@ -2901,7 +2890,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -2937,7 +2926,7 @@ mod tests { }, SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }, ) @@ -2981,7 +2970,7 @@ mod tests { }; let cpu = SummaryOperationCpuEvidence { insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), + evaluation_cpu_ops: Some(1.0), ..SummaryOperationCpuEvidence::default() }; assert_eq!( @@ -3009,157 +2998,253 @@ mod tests { assert_eq!(estimate.peak_memory_bytes(), 64); // 4 persistent states + join memory. } - fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { + fn count_state_schema() -> Schema { + Schema::lifted( + vec![Field::new( + "count", + FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), + false, + )], + None, + ) + } + + fn count_evaluation_schema() -> Schema { + Schema::lifted(vec![Field::plain("count", DataType::Int64, false)], None) + } + + /// The retained relational input of every test summary: a bare scan of + /// the `metrics` series. + fn metrics_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::TimeSeries { + metric: "metrics".into(), + }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ], + 0, + vec![], + ), + })) + .unwrap() + } + + fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { let state_type = FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count); - let schema = Schema::lifted(vec![Field::new("count", state_type.clone(), false)], None); - let leaf = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - })), - schema: schema.clone(), - guarantee: None, - }); - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: leaf, + let schema = count_state_schema(); + let child = metrics_scan(); + let agg = OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child, family: state_type, - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Wildcard), + input: SummaryUpdate { + item: None, + weight: asap_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }, reduction: Reduction::by(vec![]), grouping: GroupingStrategy::PerSubpopulationInstance, filter: None, - }, - schema: schema.clone(), - guarantee: None, - }); + }), + schema.clone(), + ) + .with_guarantee(None); + let agg = std::rc::Rc::new(agg); let mut root = Rc::clone(&agg); if merge { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&agg), Rc::clone(&agg)], - }, - schema: schema.clone(), - guarantee: None, - }); + root = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryMerge { + children: vec![Rc::clone(&agg), Rc::clone(&agg)], + }), + schema.clone(), + ) + .with_guarantee(None), + ); } if subtract { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummarySubtract { - left: Rc::clone(&root), - right: Rc::clone(&agg), - }, - schema: schema.clone(), - guarantee: None, - }); + root = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummarySubtract { + left: Rc::clone(&root), + right: Rc::clone(&agg), + }), + schema.clone(), + ) + .with_guarantee(None), + ); } if delete { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryDelete { - summary_input: root, - key: ColumnRef::Wildcard, - }, - schema: schema.clone(), - guarantee: None, - }); + root = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryDelete { + summary_input: root, + key: 0, + }), + schema.clone(), + ) + .with_guarantee(None), + ); } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: root, - query: asap_types::post_asap::SketchStatistic::PointCount { - key: ColumnRef::Wildcard, - value: None, - }, - }, - schema, - guarantee: Some(ResultGuarantee::exact("exact count readout")), - }) + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: root }), + count_evaluation_schema(), + ) + .with_guarantee(Some(ResultGuarantee::exact("exact count evaluation"))), + ) } - fn summary_join() -> Rc { + fn summary_join() -> Rc { let left = summary_with_operations(false, false, false); let right = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { - summary_input: left, - .. - } = &left.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: left }) = &left.operator else { unreachable!() }; - let SummaryExpr::SummaryEstimate { - summary_input: right, - .. - } = &right.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: right }) = &right.operator else { unreachable!() }; - let schema = left.schema.clone(); - let join = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryJoin { - outer: Rc::clone(left), - inner: Rc::clone(right), - key: ColumnRef::Wildcard, - family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), - }, - schema: schema.clone(), - guarantee: None, - }); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: join, - query: asap_types::post_asap::SketchStatistic::PointCount { - key: ColumnRef::Wildcard, - value: None, - }, - }, - schema, - guarantee: None, - }) + let join = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryJoin { + outer: Rc::clone(left), + inner: Rc::clone(right), + key: 0, + family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), + }), + left.schema.clone(), + ) + .with_guarantee(None), + ); + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: join }), + count_evaluation_schema(), + ) + .with_guarantee(None), + ) } - fn summary_binary() -> Rc { + fn add_shared_summary_result() -> Rc { let operand = summary_with_operations(false, false, false); - Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: asap_types::post_asap::ExecutionTiming::QueryTime, + Rc::new( + OperatorNode::new(Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + }, + return_bool: false, lhs: Rc::clone(&operand), rhs: operand, - operator: asap_types::post_asap::BinaryOperator { + })) + .unwrap() + .with_guarantee(Some(ResultGuarantee::exact("test binary"))), + ) + } + + /// Two independent summary states, each read out, combined by an ordinary + /// `BinaryOp`: the non-reserved replacement for a `SummaryJoin` fixture. + #[test] + fn independent_summary_results_form_a_valid_dag() { + add_independent_summary_results() + .validate_structure() + .unwrap(); + } + + fn add_independent_summary_results() -> Rc { + Rc::new( + OperatorNode::new(Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { checked_relative_division: false, checked_finite_division: false, - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), vector_match: None, }, - }, - schema: Schema::lifted( - vec![Field::new( - "value", - FieldDataType::Plain(DataType::Float64), - false, - )], - None, - ), - guarantee: Some(ResultGuarantee::exact("test binary")), - }) + return_bool: false, + lhs: summary_with_operations(false, false, false), + rhs: summary_with_operations(false, false, false), + })) + .unwrap() + .with_guarantee(Some(ResultGuarantee::exact("test binary"))), + ) + } + + /// The `(lhs, rhs)` evaluations of [`add_independent_summary_results`]. + fn binary_operands(root: &OperatorNode) -> (Rc, Rc) { + let Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) = &root.operator else { + unreachable!(); + }; + (Rc::clone(lhs), Rc::clone(rhs)) + } + + /// The `SummaryAgg` under one `SummaryEstimate` evaluation. + fn evaluation_state(evaluation: &OperatorNode) -> Rc { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: summary_input, + }) = &evaluation.operator + else { + unreachable!(); + }; + Rc::clone(summary_input) + } + + /// The DAG-validation error that streaming lifecycle planning of `root` + /// fails closed with. + fn streaming_planning_error( + root: Rc, + model: &SummaryMaintenanceCostModel, + ) -> asap_types::post_asap::ExecutionDataStateError { + let workload = streaming_workload(); + match plan_summary_maintenance_lifecycles( + root, + WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), + 0, + Some(Horizon(5.0)), + SummaryMaintenanceLifecycleCapabilities::ALL, + model, + ) { + Err( + crate::summary_maintenance_lifecycle::SummaryMaintenanceLifecyclePlanError::InvalidPostAsapDAG( + error, + ), + ) => error, + Err(other) => panic!("unexpected planning error: {other}"), + Ok(_) => panic!("planning must fail closed"), + } + } + + fn test_resource( + physical_id: &str, + inputs: Vec, + cpu_ops: f64, + working_memory_bytes: u64, + output_buffer_bytes: u64, + ) -> SummaryOperatorResourceEvidence { + SummaryOperatorResourceEvidence { + physical_id: physical_id.into(), + inputs, + output: test_edge(), + cpu_ops, + working_memory_bytes, + output_buffer_bytes, + executions_per_evaluation: 1, + io_bytes_per_execution: Some(0), + } } #[test] fn exact_binary_is_costable_with_explicit_physical_evidence() { let workload = streaming_workload(); let target = streaming_sum_query(); - let root = summary_binary(); + let root = add_shared_summary_result(); let mut model = streaming_model(); bind_aggregations( &mut model, @@ -3186,29 +3271,16 @@ mod tests { assert!(plan.summary_total_cost.is_some()); } - fn streaming_sum_query() -> Rc { - let scan = Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - }); - Rc::new(QueryExpr::Aggregate { + fn streaming_sum_query() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], filters: vec![], having: None, - child: scan, - }) + child: metrics_scan(), + })) + .unwrap() } fn streaming_workload() -> QueryWorkload { @@ -3303,61 +3375,43 @@ mod tests { } } + /// A retained relational sub-DAG: a non-ASAP node with no summary below + /// it, costed as one unit through retained-query evidence. + fn is_retained(node: &OperatorNode) -> bool { + !node.contains_asap() + } + fn bind_comparison( model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, ) { model .bind_candidate_comparison(target, root, streaming_scope(), streaming_raw()) .unwrap(); fn retained( model: &mut SummaryMaintenanceCostModel, - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, + node: &Rc, + seen: &mut HashSet<*const OperatorNode>, ) { if !seen.insert(Rc::as_ptr(node)) { return; } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - model.node_evidence.insert_retained_query( - node, - RetainedSubDAGEvidence { - physical_id: format!("retained-{node:p}"), - output: test_edge(), - preprocessing_cpu_ops_over_horizon: 1.0, - working_memory_bytes: 8, - output_buffer_bytes: 0, - }, - ); - } - SummaryExpr::SummaryAgg { child, .. } - | SummaryExpr::ValueOperation { child, .. } => retained(model, child, seen), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - retained(model, child, seen); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - retained(model, left, seen); - retained(model, right, seen); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - retained(model, summary_input, seen) - } + if is_retained(node) { + model.node_evidence.insert_retained_query( + node, + RetainedSubDAGEvidence { + physical_id: format!("retained-{node:p}"), + output: test_edge(), + preprocessing_cpu_ops_over_horizon: 1.0, + working_memory_bytes: 8, + output_buffer_bytes: 0, + }, + ); + return; + } + for child in node.children() { + retained(model, child, seen); } } retained(model, root, &mut HashSet::new()); @@ -3384,24 +3438,23 @@ mod tests { fn streaming_cpu() -> SummaryOperationCpuEvidence { SummaryOperationCpuEvidence { insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(3.0), + evaluation_cpu_ops: Some(3.0), ..SummaryOperationCpuEvidence::default() } } fn bind_aggregations( model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, + target: &Rc, + root: &Rc, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, ) { bind_comparison(model, target, root); for node in evidence_nodes(root).0 { let source_root = matches!( - &node.expr, - SummaryExpr::SummaryAgg { child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)) + &node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) if is_retained(child) ); let mut node_inputs = inputs; if !source_root { @@ -3426,147 +3479,131 @@ mod tests { }, ); } + fn resource( + physical_id: String, + inputs: Vec, + cpu_ops: f64, + working_memory_bytes: u64, + ) -> SummaryOperatorResourceEvidence { + SummaryOperatorResourceEvidence { + physical_id, + inputs, + output: test_edge(), + cpu_ops, + working_memory_bytes, + output_buffer_bytes: 0, + executions_per_evaluation: 1, + io_bytes_per_execution: Some(0), + } + } fn bind_ops( model: &mut SummaryMaintenanceCostModel, - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, inputs: SummaryMaintenanceInputs, cpu: SummaryOperationCpuEvidence, ) { if !seen.insert(node as *const _) { return; } - let operation = match &node.expr { - SummaryExpr::BinaryOp { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Binary(SummaryOperatorResourceEvidence { - physical_id: format!("binary-{node:p}"), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) - }), - SummaryExpr::ValueOperation { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::ValueOperation(SummaryOperatorResourceEvidence { - physical_id: format!("value-operation-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), + if is_retained(node) { + return; + } + let operation = match &node.operator { + Operator::NonASAP(NonASAPOp::BinaryOp { .. }) => { + cpu.evaluation_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Binary(resource( + format!("binary-{node:p}"), + vec![test_edge(), test_edge()], + cpu_ops, + 0, + )) }) - }), - SummaryExpr::SummaryMerge { .. } => cpu.merge_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Merge(SummaryOperatorResourceEvidence { - physical_id: format!("merge-{node:p}"), - inputs: match &node.expr { - SummaryExpr::SummaryMerge { children, .. } => { - vec![test_edge(); children.len()] - } - _ => unreachable!(), - }, - output: test_edge(), + } + Operator::NonASAP(NonASAPOp::Join { .. }) => None, + Operator::NonASAP(_) + | Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::MaintainPopulation { .. } + | ASAPOp::EvaluatePopulation { .. }, + ) => cpu.evaluation_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::ValueOperation(resource( + format!("value-operation-{node:p}"), + vec![test_edge()], cpu_ops, - working_memory_bytes: inputs.state_bytes_per_summary, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) + 0, + )) }), - SummaryExpr::SummarySubtract { .. } => cpu.subtract_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Subtract(SummaryOperatorResourceEvidence { - physical_id: format!("subtract-{node:p}"), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: inputs.state_bytes_per_summary, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), + Operator::ASAP(ASAPOp::SummaryMerge { children }) => { + cpu.merge_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Merge(resource( + format!("merge-{node:p}"), + vec![test_edge(); children.len()], + cpu_ops, + inputs.state_bytes_per_summary, + )) }) - }), - SummaryExpr::SummaryDelete { .. } => cpu.delete_cpu_ops.and_then(|cpu_ops| { - Some(SummaryOperatorEvidence::Delete { - resource: SummaryOperatorResourceEvidence { - physical_id: format!("delete-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), + } + Operator::ASAP(ASAPOp::SummarySubtract { .. }) => { + cpu.subtract_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Subtract(resource( + format!("subtract-{node:p}"), + vec![test_edge(), test_edge()], cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - events_per_second: cpu.delete_events_per_second?, - routing_fanout: cpu.delete_routing_fanout?, + inputs.state_bytes_per_summary, + )) }) - }), - SummaryExpr::SummaryEstimate { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Readout(SummaryOperatorResourceEvidence { - physical_id: format!("readout-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), + } + Operator::ASAP(ASAPOp::SummaryDelete { .. }) => { + cpu.delete_cpu_ops.and_then(|cpu_ops| { + Some(SummaryOperatorEvidence::Delete { + resource: resource( + format!("delete-{node:p}"), + vec![test_edge()], + cpu_ops, + 0, + ), + events_per_second: cpu.delete_events_per_second?, + routing_fanout: cpu.delete_routing_fanout?, + }) }) - }), - _ => None, + } + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) => { + cpu.evaluation_cpu_ops.map(|cpu_ops| { + SummaryOperatorEvidence::Evaluation(resource( + format!("evaluation-{node:p}"), + vec![test_edge()], + cpu_ops, + 0, + )) + }) + } + Operator::ASAP( + ASAPOp::SummaryAgg { .. } + | ASAPOp::SummaryJoin { .. } + | ASAPOp::Extension { .. }, + ) => None, }; if let Some(operation) = operation { model .node_evidence .operations .insert(node as *const _, operation); - if let SummaryExpr::SummaryDelete { summary_input, .. } = &node.expr { + if let Operator::ASAP(ASAPOp::SummaryDelete { summary_input, .. }) = &node.operator + { fn owning_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - owners: &mut Vec<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + owners: &mut Vec<*const OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - owners.push(node as *const _); - owning_aggs(child, seen, owners); - } - SummaryExpr::ValueOperation { child, .. } => { - owning_aggs(child, seen, owners) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - owning_aggs(child, seen, owners); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - owning_aggs(left, seen, owners); - owning_aggs(right, seen, owners); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - owning_aggs(summary_input, seen, owners); - } - SummaryExpr::KeepPreAsap(_) => {} + if matches!(node.operator, Operator::ASAP(ASAPOp::SummaryAgg { .. })) { + owners.push(node as *const _); + } + for child in node.children() { + owning_aggs(child, seen, owners); } } let mut owners = Vec::new(); @@ -3581,36 +3618,8 @@ mod tests { } } } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } - | SummaryExpr::ValueOperation { child, .. } => { - bind_ops(model, child, seen, inputs, cpu) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - bind_ops(model, child, seen, inputs, cpu); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - bind_ops(model, left, seen, inputs, cpu); - bind_ops(model, right, seen, inputs, cpu); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - bind_ops(model, summary_input, seen, inputs, cpu) - } - SummaryExpr::KeepPreAsap(_) => {} + for child in node.children() { + bind_ops(model, child, seen, inputs, cpu); } } bind_ops(model, root, &mut HashSet::new(), inputs, cpu); diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs index c3c12c20b..fb80d5ee0 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs @@ -3,7 +3,7 @@ use super::*; /// One per-state window choice within a complete Planner candidate. #[derive(Debug, Clone)] pub struct SummaryWindowFrameworkAssignment { - pub summary: Rc, + pub summary: Rc, /// `None` explicitly means that this state is not window-organized. pub framework: Option, } @@ -29,46 +29,20 @@ pub struct SummaryWindowFrameworkCandidate { pub node_evidence: SummaryNodeEvidence, } -pub(super) fn summary_aggregation_identities(root: &SummaryNode) -> HashSet<*const SummaryNode> { +pub(super) fn summary_aggregation_identities(root: &OperatorNode) -> HashSet<*const OperatorNode> { fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut HashSet<*const SummaryNode>, + node: &OperatorNode, + seen: &mut HashSet<*const OperatorNode>, + out: &mut HashSet<*const OperatorNode>, ) { if !seen.insert(node as *const _) { return; } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => { - out.insert(node as *const _); - visit(child, seen, out); - } - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, out), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - visit(child, seen, out); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - visit(left, seen, out); - visit(right, seen, out); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - visit(summary_input, seen, out); - } + if matches!(node.operator, Operator::ASAP(ASAPOp::SummaryAgg { .. })) { + out.insert(node as *const _); + } + for child in node.children() { + visit(child, seen, out); } } @@ -146,11 +120,11 @@ impl SummaryWindowAccuracyEvidence { eh_summaries.len() == 1 && eh_summaries.iter().all(|assignment| { matches!( - &assignment.summary.expr, - SummaryExpr::SummaryAgg { + &assignment.summary.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } if kind.algorithm() == &SketchAlgorithm::Kll + }) if kind.algorithm() == &SketchAlgorithm::Kll ) }) } @@ -160,14 +134,14 @@ impl SummaryWindowAccuracyEvidence { eh_summaries.len() == 1 && eh_summaries.iter().all(|assignment| { matches!( - &assignment.summary.expr, - SummaryExpr::SummaryAgg { + &assignment.summary.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate( ExactKind::Count | ExactKind::Sum, _ ), .. - } + }) ) }) } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs index 8e63a4ce1..a3a7bb8f8 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs @@ -8,12 +8,14 @@ use std::collections::HashMap; use std::rc::Rc; +use asap_types::ir::OperatorNode; use serde::Serialize; use asap_types::dag_export::{self, SummaryDAG}; +use asap_types::ir::physical_export::PhysicalASAPNodeId; use asap_types::post_asap::{ - PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, + ResultGuarantee, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, + SummaryWindowFramework, }; use crate::summary_maintenance_lifecycle::{ @@ -42,7 +44,7 @@ pub struct SummaryMaintenanceDAGExport { #[derive(Debug, Clone, Serialize)] pub struct SummaryMaintenanceDeploymentExport { - pub post_asap_node_id: PostAsapNodeId, + pub post_asap_node_id: PhysicalASAPNodeId, #[serde(skip_serializing_if = "Option::is_none")] pub selected_window_framework: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -93,13 +95,7 @@ pub fn export_summary_maintenance_plan( .zip(&deployments) .map(|(deployment, export)| (Rc::as_ptr(&deployment.summary), export)) .collect(); - let mut next_node_id = 0; - annotate_lifecycle_deployments( - &plan.root, - &mut dag, - &deployment_by_summary, - &mut next_node_id, - ); + annotate_lifecycle_deployments(&mut dag, &deployment_by_summary); SummaryMaintenanceDAGExport { dag, @@ -116,47 +112,21 @@ pub fn export_summary_maintenance_plan( } } -/// Walk in the same post-order as `dag_export::export_summary` and attach a -/// deployment directly to every flattened occurrence of its state node. -/// This makes the decision visible to DAG consumers without asking them to -/// reconstruct pointer identity from DAG position. +/// Attach a deployment directly to the exported node of its `SummaryAgg`, +/// matched by the `Rc` identity every exported node carries. This makes the +/// decision visible to dag consumers without asking them to reconstruct +/// pointer identity from dag position. fn annotate_lifecycle_deployments( - node: &SummaryNode, dag: &mut SummaryDAG, - deployments: &HashMap<*const SummaryNode, &SummaryMaintenanceDeploymentExport>, - next_node_id: &mut usize, + deployments: &HashMap<*const OperatorNode, &SummaryMaintenanceDeploymentExport>, ) { - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { - for child in summary_children(&node.expr) { - annotate_lifecycle_deployments(child, dag, deployments, next_node_id); + for dag_node in &mut dag.nodes { + let Some(source) = &dag_node.source_node else { + continue; + }; + if let Some(deployment) = deployments.get(&Rc::as_ptr(source)) { + dag_node.detail["summary_maintenance"] = + serde_json::to_value(deployment).expect("lifecycle export is serializable"); } } - let dag_node = &mut dag.nodes[*next_node_id]; - if let Some(deployment) = deployments.get(&(node as *const SummaryNode)) { - dag_node.detail["summary_maintenance"] = - serde_json::to_value(deployment).expect("lifecycle export is serializable"); - } - *next_node_id += 1; -} - -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { - match expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], - SummaryExpr::SummaryAgg { child, .. } => vec![child], - SummaryExpr::ValueOperation { child, .. } => vec![child], - SummaryExpr::SummaryJoin { outer, inner, .. } - | SummaryExpr::RelationalJoin { - left: outer, - right: inner, - .. - } - | SummaryExpr::SummarySubtract { - left: outer, - right: inner, - } => vec![outer, inner], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryMerge { children, .. } => children.iter().collect(), - } } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index cf41efeff..c92e03094 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -17,17 +17,21 @@ //! incrementally. Unknown evidence stays unknown and therefore cannot make a //! long-lived alternative win. +use asap_types::ir::cse::share_common_sub_dags; use std::collections::{HashMap, HashSet}; use std::rc::Rc; +use asap_types::ir::physical_export::{ + compile_physical_asap_dag_with_node_ids, compile_physical_asap_workload_with_node_ids, + PhysicalASAPDAG, PhysicalASAPDAGValidationError, PhysicalASAPNodeId, +}; +use asap_types::ir::timing::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ - compile_post_asap_dag_with_node_ids, share_common_summary_sub_dags, EvaluationSchedule, - ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDAG, - PostAsapDAGValidationError, PostAsapNodeId, ResultGuarantee, SummaryExpr, - SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, - SummaryNode, SummaryWindowFramework, ValueOperation, + EvaluationSchedule, ExecutionDataStateError, ExecutionTiming, OutputRepresentation, + ResultGuarantee, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, + SummaryMaintenanceMode, SummaryWindowFramework, }; -use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DataWorkload, Predictability, QueryRecurrence, QueryWorkload, RepeatedDemand, @@ -44,7 +48,7 @@ use crate::recurrence::{ }; use crate::replacement::{ CandidateCostOverrides, CandidateLogicalASAPDAGs, GlobalSelection, RealizationError, - Replacement, + Replacement, ReplacementProvenance, }; /// Summary-maintenance lifecycle shapes supported by the target runtime. @@ -167,11 +171,9 @@ pub struct SummaryMaintenanceDeployment { /// Identity of this summary in the exported post-ASAP semantic DAG. /// It is scoped to one plan version and is not a summary definition or /// summary instance identity. - pub post_asap_node_id: PostAsapNodeId, - /// The unique materialized `SummaryAgg`, or maintained population - /// (`MaintainPopulation`) not consumed by a `SummaryAgg`, represented by - /// this deployment. Cost-model lifecycle hooks receive this node. - pub summary: Rc, + pub post_asap_node_id: PhysicalASAPNodeId, + /// The unique materialized `SummaryAgg` represented by this deployment. + pub summary: Rc, /// Lifecycle, evaluation, and representation commitment selected for this /// state, or `None` when no alternative is selectable. pub summary_maintenance_lifecycle_guarantee: Option, @@ -187,10 +189,9 @@ pub struct SummaryMaintenanceDeployment { #[derive(Debug, Clone)] pub struct SummaryMaintenanceLifecyclePlan { /// Root of the materialized post-ASAP DAG being deployed. - pub root: Rc, - /// One entry per unique reachable `SummaryAgg`, then per unique - /// maintained population outside any `SummaryAgg`'s inputs; shared `Rc` - /// nodes appear only once. + pub root: Rc, + /// One entry per unique reachable `SummaryAgg`; shared `Rc` nodes appear + /// only once. pub deployments: Vec, /// Caller-supplied optimization horizon used to turn rates into total /// costs. `None` keeps horizon-dependent alternatives unselectable. @@ -223,14 +224,14 @@ pub enum SummaryMaintenanceTimingError { #[error(transparent)] InvalidPostAsapDAG(#[from] ExecutionDataStateError), #[error("summary {0:?} has no selected lifecycle")] - UnselectedLifecycle(PostAsapNodeId), + UnselectedLifecycle(PhysicalASAPNodeId), /// A maintained population outside any `SummaryAgg`'s inputs has no /// deployment, so its timing would be guessed. Enumeration always emits /// one; this arises only for a plan whose root or deployments were edited. #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] - UnplannedMaintainedState(PostAsapNodeId), + UnplannedMaintainedState(PhysicalASAPNodeId), #[error(transparent)] - InvalidPhases(#[from] PostAsapDAGValidationError), + InvalidPhases(#[from] PhysicalASAPDAGValidationError), } impl SummaryMaintenanceLifecyclePlan { @@ -239,65 +240,86 @@ impl SummaryMaintenanceLifecyclePlan { /// /// A retained (non-`Ephemeral`) state outlives one query, so it and every /// input it consumes run at ingestion time. Every other node runs at query - /// time: readouts and consumers of retained state, and each `Ephemeral` + /// time: evaluations and consumers of retained state, and each `Ephemeral` /// state not consumed by retained state together with its inputs, whose /// raw data the deployment must supply as a query source. This applies to /// maintained populations as to `SummaryAgg` states; a population feeding /// a `SummaryAgg` is one of its inputs. Timings already on the root are /// ignored. - pub fn execution_timed_dag(&self) -> Result { - let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; - let dag = compiled.dag; - for population in &standalone_populations(&self.root) { - let id = compiled - .node_ids - .node_id(population) - .expect("collected population belongs to the compiled DAG"); - if !self - .deployments - .iter() - .any(|deployment| deployment.post_asap_node_id == id) - { + pub fn execution_timed_dag(&self) -> Result { + execution_timed_workload_dag(&[self]) + } +} + +/// One physical ASAP DAG for a workload: a root per plan, in order, with +/// sub-DAGs shared between plans exported once. Timing follows the selected +/// lifecycles of every plan's deployments, as in +/// [`SummaryMaintenanceLifecyclePlan::execution_timed_dag`]. +pub fn execution_timed_workload_dag( + plans: &[&SummaryMaintenanceLifecyclePlan], +) -> Result { + // One memo, so a node shared by several roots is timed and exported once. + let mut memo = TimingMemo::new(); + let assignment = LifecycleAssignment::default_maintained(); + let timed = plans + .iter() + .map(|plan| apply_lifecycle_timings(&plan.root, &assignment, &mut memo)) + .collect::, _>>()?; + let compiled = compile_physical_asap_workload_with_node_ids(&timed)?; + let id_of = |node: &Rc| { + compiled + .node_ids + .node_id(memo.timed(node).expect("plan node was timed")) + .expect("timed plan node belongs to the compiled DAG") + }; + let deployments: Vec<_> = plans + .iter() + .flat_map(|plan| &plan.deployments) + .map(|deployment| (id_of(&deployment.summary), deployment)) + .collect(); + for plan in plans { + for population in &standalone_populations(&plan.root) { + let id = id_of(population); + if !deployments.iter().any(|(deployed, _)| *deployed == id) { return Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)); } } - let mut pending = Vec::new(); - for deployment in &self.deployments { - let guarantee = deployment - .summary_maintenance_lifecycle_guarantee - .as_ref() - .ok_or(SummaryMaintenanceTimingError::UnselectedLifecycle( - deployment.post_asap_node_id, - ))?; - if guarantee.summary_maintenance_lifecycle != SummaryMaintenanceLifecycle::Ephemeral { - pending.push(deployment.post_asap_node_id); - } - } - let mut ingestion = HashSet::new(); - while let Some(id) = pending.pop() { - if ingestion.insert(id) { - pending.extend( - dag.edges - .iter() - .filter(|edge| edge.consumer == id) - .map(|edge| edge.producer), - ); - } + } + let dag = compiled.dag; + let mut pending = Vec::new(); + for (id, deployment) in &deployments { + let guarantee = deployment + .summary_maintenance_lifecycle_guarantee + .as_ref() + .ok_or(SummaryMaintenanceTimingError::UnselectedLifecycle(*id))?; + if guarantee.summary_maintenance_lifecycle != SummaryMaintenanceLifecycle::Ephemeral { + pending.push(*id); + } + } + let mut ingestion = HashSet::new(); + while let Some(id) = pending.pop() { + if ingestion.insert(id) { + pending.extend( + dag.edges + .iter() + .filter(|edge| edge.consumer == id) + .map(|edge| edge.producer), + ); } - let phases = dag - .nodes - .iter() - .map(|node| { - let timing = if ingestion.contains(&node.id) { - ExecutionTiming::IngestionTime - } else { - ExecutionTiming::QueryTime - }; - (node.id, timing) - }) - .collect(); - Ok(dag.with_execution_phases(&phases)?) } + let phases = dag + .nodes + .iter() + .map(|node| { + let timing = if ingestion.contains(&node.id) { + ExecutionTiming::IngestionTime + } else { + ExecutionTiming::QueryTime + }; + (node.id, timing) + }) + .collect(); + Ok(dag.with_execution_phases(&phases)?) } /// Explicit association between a materialized target and the normalized @@ -390,23 +412,23 @@ pub struct SummaryMaintenanceLifecycleCandidates<'a> { arrival: DataArrival, required_accuracy: Vec, cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a OperatorNode>, } /// Why an explicit per-state lifecycle choice cannot be bound. #[derive(Debug, thiserror::Error, PartialEq)] pub enum SummaryMaintenanceLifecycleChoiceError { #[error("summary {0:?} is not a deployment of this root")] - UnknownSummary(PostAsapNodeId), + UnknownSummary(PhysicalASAPNodeId), #[error("summary {0:?} is chosen more than once")] - DuplicateChoice(PostAsapNodeId), + DuplicateChoice(PhysicalASAPNodeId), #[error("summary {0:?} has no chosen lifecycle")] - MissingChoice(PostAsapNodeId), + MissingChoice(PhysicalASAPNodeId), #[error("chosen lifecycle is not an enumerated alternative of summary {0:?}")] - NotAnAlternative(PostAsapNodeId), + NotAnAlternative(PhysicalASAPNodeId), #[error("chosen lifecycle of summary {post_asap_node_id:?} is rejected: {rejection:?}")] Rejected { - post_asap_node_id: PostAsapNodeId, + post_asap_node_id: PhysicalASAPNodeId, rejection: Option, }, #[error("summary states on one maintenance path have different evaluation schedules")] @@ -480,7 +502,7 @@ impl SummaryMaintenanceLifecycleCandidates<'_> { /// cost are the model's and unknown cost is never replaced by zero. pub fn select( mut self, - choices: &[(PostAsapNodeId, SummaryMaintenanceLifecycle)], + choices: &[(PhysicalASAPNodeId, SummaryMaintenanceLifecycle)], ) -> Result { use SummaryMaintenanceLifecycleChoiceError as E; let deployments = &self.plan.deployments; @@ -580,7 +602,7 @@ struct SummaryMaintenanceWorkloadFacts { /// unique summary state, and select the cheapest legal alternative whose cost /// is fully known. pub fn plan_summary_maintenance_lifecycles( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -603,7 +625,7 @@ pub fn plan_summary_maintenance_lifecycles( /// alternatives itself binds its choice with /// [`SummaryMaintenanceLifecycleCandidates::select`]. pub fn enumerate_summary_maintenance_lifecycles<'a>( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -627,14 +649,14 @@ pub fn enumerate_summary_maintenance_lifecycles<'a>( /// DAG path multiplicity has been propagated by `CandidateLogicalASAPDAGs`. #[expect(clippy::too_many_arguments, reason = "internal bound planning context")] fn enumerate_with_profile<'a>( - root: Rc, + root: Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &'a dyn CostModel, profile: Option, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a OperatorNode>, ) -> Result, SummaryMaintenanceLifecyclePlanError> { demand.workload.validate()?; if let Some(data) = demand.data_workload { @@ -673,11 +695,33 @@ fn enumerate_with_profile<'a>( StateKind::SummaryAgg, ); summaries.extend(standalone_populations(&root)); - let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; + let mut timing_memo = TimingMemo::new(); + let timed_root = apply_lifecycle_timings( + &root, + &LifecycleAssignment::default_maintained(), + &mut timing_memo, + )?; + let node_ids = compile_physical_asap_dag_with_node_ids(&timed_root)?.node_ids; let components = summary_state_components(&summaries); let deployments: Vec = summaries .into_iter() .map(|summary| { + let capabilities = if OperatorNode::reachable(&summary).iter().any(|node| { + matches!( + node.non_asap(), + Some(asap_types::ir::NonASAPOp::BinaryOp { .. }) + ) && asap_types::ir::timing::validate_default(node, ExecutionTiming::IngestionTime) + .is_err() + }) { + SummaryMaintenanceLifecycleCapabilities { + supports_ephemeral: capabilities.supports_ephemeral, + supports_prepared: false, + supports_shared: false, + supports_continuously_maintained: false, + } + } else { + capabilities + }; let alternatives = alternatives_for( &facts, horizon, @@ -686,8 +730,9 @@ fn enumerate_with_profile<'a>( cost_model.summary_maintenance_lifecycle_cost_inputs_for_horizon(&summary, horizon), ); SummaryMaintenanceDeployment { - post_asap_node_id: node_ids - .node_id(&summary) + post_asap_node_id: timing_memo + .timed(&summary) + .and_then(|timed| node_ids.node_id(timed)) .expect("collected summary belongs to the compiled DAG"), summary, summary_maintenance_lifecycle_guarantee: None, @@ -696,7 +741,7 @@ fn enumerate_with_profile<'a>( } }) .collect(); - let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); + let selected_raw_recompute = !root.contains_asap(); Ok(SummaryMaintenanceLifecycleCandidates { plan: SummaryMaintenanceLifecyclePlan { root, @@ -761,9 +806,14 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( continue; }; for candidate in &group.candidates { - let Replacement::Summary(summary) = &candidate.replacement else { + // Only summary realizations carry a maintenance lifecycle; a + // logical rewrite or CSE share/recompute candidate does not. + let Replacement::SubDAG(summary) = &candidate.replacement else { continue; }; + if candidate.provenance != ReplacementProvenance::SummaryRealization { + continue; + } costs.finalize_target(&group.target); let plan = enumerate_with_profile( Rc::clone(summary), @@ -800,14 +850,14 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( // Intern every member once; members whose outermost state (the // `SummaryAgg` every other state of the candidate feeds) interns to the // same node share it. Classes are kept in first-member order. - let interned = share_common_summary_sub_dags( + let interned = share_common_sub_dags( members .iter() .enumerate() .map(|(index, (_, _, summary))| (index, Rc::clone(summary))) .collect(), ); - let mut classes: Vec<(Rc, Vec)> = Vec::new(); + let mut classes: Vec<(Rc, Vec)> = Vec::new(); for (index, root) in interned { let states = summary_states(&root); let Some(state) = states @@ -826,7 +876,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( } let mut shared = Vec::new(); for (state, class) in classes { - let mut targets: Vec<&Rc> = Vec::new(); + let mut targets: Vec<&Rc> = Vec::new(); for &index in &class { let target = &members[index].0.target; if !targets.iter().any(|t| Rc::ptr_eq(t, target)) { @@ -902,7 +952,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( /// `demand`, or `None` when no lifecycle alternative is selectable for it. /// No comparison target is supplied: the state serves several queries. pub(crate) fn shared_state_cost( - state: &Rc, + state: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -924,7 +974,7 @@ pub(crate) fn shared_state_cost( } /// Every unique `SummaryAgg` reachable from `root`. -pub(crate) fn summary_states(root: &Rc) -> Vec> { +pub(crate) fn summary_states(root: &Rc) -> Vec> { let mut states = Vec::new(); collect_states( root, @@ -939,7 +989,7 @@ pub(crate) fn summary_states(root: &Rc) -> Vec> { /// summary maintenance decisions. This does not create or maintain runtime state. pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( selection: &GlobalSelection<'_>, - target: &Rc, + target: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -966,8 +1016,8 @@ pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( /// [`assemble_selected_dag_with_summary_maintenance_lifecycles`], for a root /// the caller already assembled (and possibly interned across queries). pub(crate) fn plan_assembled_dag( - root: Rc, - target: &Rc, + root: Rc, + target: &Rc, demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, @@ -994,7 +1044,7 @@ pub(crate) fn plan_assembled_dag( .is_none_or(|summary| raw.0 <= summary.0) }) { - plan.root = crate::replacement::keep_pre_asap(target)?; + plan.root = crate::replacement::retain_exact(target)?; plan.deployments.clear(); plan.selected_raw_recompute = true; plan.selected_window_implementation_id = None; @@ -1445,66 +1495,35 @@ enum StateKind { /// Collect every unique node of `kind` reachable from `node`. fn collect_states( - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, - output: &mut Vec>, + node: &Rc, + seen: &mut HashSet<*const OperatorNode>, + output: &mut Vec>, kind: StateKind, ) { if !seen.insert(Rc::as_ptr(node)) { return; } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - if kind == StateKind::SummaryAgg { - output.push(Rc::clone(node)); - } - collect_states(child, seen, output, kind); - } - SummaryExpr::ValueOperation { - child, operation, .. - } => { - if kind == StateKind::Population - && matches!(operation, ValueOperation::MaintainPopulation { .. }) - { - output.push(Rc::clone(node)); - } - collect_states(child, seen, output, kind) - } - SummaryExpr::SummaryJoin { outer, inner, .. } - | SummaryExpr::RelationalJoin { - left: outer, - right: inner, - .. - } - | SummaryExpr::BinaryOp { - lhs: outer, - rhs: inner, - .. - } - | SummaryExpr::SummarySubtract { - left: outer, - right: inner, - } => { - collect_states(outer, seen, output, kind); - collect_states(inner, seen, output, kind); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_states(summary_input, seen, output, kind) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - collect_states(child, seen, output, kind); - } - } - SummaryExpr::KeepPreAsap(_) => {} + if matches!( + (&node.operator, kind), + ( + Operator::ASAP(ASAPOp::SummaryAgg { .. }), + StateKind::SummaryAgg + ) | ( + Operator::ASAP(ASAPOp::MaintainPopulation { .. }), + StateKind::Population + ) + ) { + output.push(Rc::clone(node)); + } + for child in node.children() { + collect_states(child, seen, output, kind); } } /// Maintained populations that are not an input of any `SummaryAgg`. A /// population feeding summary state is on that state's maintenance path, so -/// that state's lifecycle times it, even when a readout also reads it directly. -fn standalone_populations(root: &Rc) -> Vec> { +/// that state's lifecycle times it, even when a evaluation also reads it directly. +fn standalone_populations(root: &Rc) -> Vec> { let mut summaries = Vec::new(); collect_states( root, @@ -1552,7 +1571,7 @@ pub(crate) fn evaluation_schedule( /// Summary states composed on one maintenance path must be produced on the /// same schedule. Return a component id for each collected state. -fn summary_state_components(summaries: &[Rc]) -> Vec { +fn summary_state_components(summaries: &[Rc]) -> Vec { let indices: HashMap<_, _> = summaries .iter() .enumerate() @@ -1568,16 +1587,18 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { } for (parent_index, summary) in summaries.iter().enumerate() { - let SummaryExpr::SummaryAgg { child, .. } = &summary.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary.operator else { continue; }; if !matches!( - child.expr, - SummaryExpr::SummaryAgg { .. } - | SummaryExpr::SummaryJoin { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryMerge { .. } + child.operator, + Operator::ASAP( + ASAPOp::SummaryAgg { .. } + | ASAPOp::SummaryJoin { .. } + | ASAPOp::SummarySubtract { .. } + | ASAPOp::SummaryDelete { .. } + | ASAPOp::SummaryMerge { .. } + ) ) { continue; } @@ -1603,10 +1624,10 @@ fn summary_state_components(summaries: &[Rc]) -> Vec { /// Inputs shared by every complete lifecycle-combination evaluation of one /// root, whether Planner searches combinations or a caller supplies one. struct CompleteCostContext<'a> { - root: &'a SummaryNode, + root: &'a OperatorNode, components: &'a [usize], cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, + comparison_target: Option<&'a OperatorNode>, horizon: Option, expected_reads: Option, required_accuracy: &'a [AccuracyTarget], @@ -1695,12 +1716,12 @@ fn apply_selection( #[expect(clippy::too_many_arguments, reason = "complete combination context")] fn select_complete_lifecycle_combination( - root: &SummaryNode, + root: &OperatorNode, deployments: &mut [SummaryMaintenanceDeployment], components: &[usize], arrival: DataArrival, cost_model: &dyn CostModel, - comparison_target: Option<&QueryExpr>, + comparison_target: Option<&OperatorNode>, horizon: Option, expected_reads: Option, required_accuracy: &[AccuracyTarget], @@ -1836,12 +1857,16 @@ mod tests { } } use super::*; + use asap_types::ir::physical_export::PhysicalASAPOperatorPayload; + use asap_types::ir::{BinaryOperator, NonASAPOp}; use asap_types::post_asap::{ - ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, PostAsapOperatorPayload, - ResultGuarantee, Schema, SketchAlgorithm, + ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, ResultGuarantee, Schema, + SketchAlgorithm, }; use asap_types::pre_asap::AggIntent; - use asap_types::pre_asap::{ColumnRef, DataType, QueryExpr, Reduction, Source}; + use asap_types::pre_asap::{ + ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, Reduction, Source, + }; use asap_types::types::AccuracyTarget; use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, EvidenceSource, Predictability, Query, @@ -1861,7 +1886,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -1874,7 +1899,7 @@ mod tests { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -1897,19 +1922,19 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) } fn summary_maintenance_capabilities( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { UnitCosts.summary_maintenance_capabilities(summary) } - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &OperatorNode) -> Option { Some(Cost(1.0)) } } @@ -1927,14 +1952,14 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) } fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -1949,7 +1974,7 @@ mod tests { impl CostModel for SummaryMaintenancePrefersDdSketch { fn raw_query_recompute_total_cost( &self, - _target: &QueryExpr, + _target: &OperatorNode, _expected_reads: f64, ) -> Option { Some(Cost(1_000.0)) @@ -1967,7 +1992,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { let build = match sketch_algorithm(summary) { Some(SketchAlgorithm::Kll) => 100.0, @@ -1999,8 +2024,8 @@ mod tests { fn complete_summary_candidate_cost( &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, + _root: &OperatorNode, + _target: Option<&OperatorNode>, deployments: &[CostedSummaryDeployment<'_>], _horizon: Option, _expected_reads: Option, @@ -2032,12 +2057,13 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { + // A leaf summary is one built directly over kept pre-ASAP rows + // (its child is not an ASAP node); a nested one reads state. let is_leaf = matches!( - summary.expr, - SummaryExpr::SummaryAgg { ref child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)) + &summary.operator, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) if !child.is_asap() ); SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(if is_leaf { 1.0 } else { 100.0 })), @@ -2050,7 +2076,7 @@ mod tests { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -2060,23 +2086,27 @@ mod tests { } } - fn sketch_algorithm(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_algorithm(summary_input), - SummaryExpr::SummaryAgg { - family: FieldDataType::Sketch(kind, _), - .. - } => Some(kind.algorithm().clone()), - _ => None, + /// The sketch algorithm of the first `SummaryAgg` reachable from `node` + /// (through a evaluation or any relational operator kept above it). + fn sketch_algorithm(node: &OperatorNode) -> Option { + if let Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + }) = &node.operator + { + return Some(kind.algorithm().clone()); } + node.children() + .into_iter() + .find_map(|child| sketch_algorithm(child)) } - fn query_root() -> Rc { + fn query_root() -> Rc { query_root_for("m") } - fn query_root_for(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { + fn query_root_for(metric: &str) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, @@ -2089,22 +2119,24 @@ mod tests { 0, vec![], ), - }) + })) + .unwrap() } - fn sum_query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn sum_query() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Sum { col: None }], output_names: vec![], filters: vec![], having: None, child: query_root(), - }) + })) + .unwrap() } - fn quantile_query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn quantile_query() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Quantile { col: None, @@ -2115,49 +2147,56 @@ mod tests { filters: vec![], having: None, child: query_root(), - }) + })) + .unwrap() } - fn summary() -> Rc { - let child = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(query_root()), - schema: Schema::lifted(vec![], None), - guarantee: Some(ResultGuarantee::exact("raw")), - }); + /// An exact sum accumulator over the kept pre-ASAP scan. + fn summary() -> Rc { + let child = Rc::new( + query_root() + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("raw"))), + ); let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( - "value".into(), - )), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: Some(ResultGuarantee::exact("sum")), - }) + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child, + family: family.clone(), + input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( + "value".into(), + )), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }), + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(Some(ResultGuarantee::exact("sum"))), + ) } - fn nested_summary() -> Rc { + fn nested_summary() -> Rc { let child = summary(); let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( - "state".into(), - )), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: Some(ResultGuarantee::exact("nested sum")), - }) + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child, + family: family.clone(), + input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( + "state".into(), + )), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }), + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(Some(ResultGuarantee::exact("nested sum"))), + ) } fn batch(predictability: Predictability) -> BatchEntry { @@ -2649,7 +2688,13 @@ mod tests { assert_eq!(plan.raw_recompute_total_cost, Some(Cost(1.0))); assert_eq!(plan.summary_total_cost, None); assert!(plan.deployments.is_empty()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); + // The logical query stays exact; deployment assigns execution timing. + assert!(!plan.root.contains_asap()); + assert!(matches!( + plan.root.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + )); + assert!(plan.root.timing.is_none()); let exported = crate::summary_maintenance_dag_export::export_summary_maintenance_plan(&plan); @@ -2663,7 +2708,7 @@ mod tests { fn whole_candidate_cost_is_evaluated_before_selecting_a_lifecycle() { let root = summary(); let mut deployments = vec![SummaryMaintenanceDeployment { - post_asap_node_id: PostAsapNodeId(0), + post_asap_node_id: 0, summary: Rc::clone(&root), summary_maintenance_lifecycle_guarantee: None, selected_window_framework: None, @@ -2726,7 +2771,7 @@ mod tests { ]; let mut deployments: Vec<_> = (0..13) .map(|summary_index| SummaryMaintenanceDeployment { - post_asap_node_id: PostAsapNodeId(summary_index as u32), + post_asap_node_id: (summary_index as usize), summary: Rc::clone(&root), summary_maintenance_lifecycle_guarantee: None, selected_window_framework: None, @@ -2771,7 +2816,11 @@ mod tests { .unwrap(); assert!(plan.selected_raw_recompute); assert!(plan.raw_recompute_total_cost.is_none()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!plan.root.contains_asap()); + assert!(matches!( + plan.root.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + )); } #[test] @@ -2796,7 +2845,8 @@ mod tests { assert_eq!(plan.raw_recompute_total_cost, Some(Cost(1.0))); assert_eq!(plan.summary_total_cost, None); assert!(plan.deployments.is_empty()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!plan.root.contains_asap()); + assert!(matches!(plan.root.non_asap(), Some(NonASAPOp::Scan { .. }))); } #[test] @@ -2827,15 +2877,30 @@ mod tests { #[test] fn lifecycle_cost_counts_one_shared_summary_node_once() { + // One shared exact accumulator read twice by the same root: a + // query-time `sum + sum` over one finalized state. (`SummaryMerge` + // is reserved in the unified IR, so the sharing is expressed through + // a relational consumer instead.) let shared = summary(); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&shared), Rc::clone(&shared)], - }, - schema: shared.schema.clone(), - guarantee: None, - }); + let finalized = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: Rc::clone(&shared), + })) + .unwrap(), + ); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + }, + return_bool: false, + lhs: Rc::clone(&finalized), + rhs: finalized, + })) + .unwrap(); let workload = workload( vec![batch(Predictability::AdHoc), batch(Predictability::AdHoc)], vec![], @@ -2873,7 +2938,7 @@ mod tests { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -2886,18 +2951,18 @@ mod tests { fn summary_maintenance_capabilities( &self, - summary: &SummaryNode, + summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { UnitCosts.summary_maintenance_capabilities(summary) } fn raw_query_recompute_total_cost( &self, - target: &QueryExpr, + target: &OperatorNode, _expected_reads: f64, ) -> Option { - match target { - QueryExpr::Aggregate { measures, .. } => match measures[..] { + match target.non_asap() { + Some(NonASAPOp::Aggregate { measures, .. }) => match measures[..] { [AggIntent::Quantile { q: 0.5, .. }] => Some(Cost(1.0)), _ => Some(Cost(8.0)), }, @@ -2914,7 +2979,7 @@ mod tests { #[test] fn sharing_class_reverts_when_a_member_selects_elsewhere() { let quantile = |q| { - Rc::new(QueryExpr::Aggregate { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Quantile { col: None, @@ -2926,7 +2991,8 @@ mod tests { filters: vec![], having: None, child: query_root(), - }) + })) + .unwrap() }; let workload = workload(vec![], vec![repeating(), repeating()], at_rest()); // Whether each root selected a summary rather than raw recompute. @@ -3024,7 +3090,7 @@ mod tests { fn choose( candidates: &SummaryMaintenanceLifecycleCandidates<'_>, lifecycle: SummaryMaintenanceLifecycle, - ) -> Vec<(PostAsapNodeId, SummaryMaintenanceLifecycle)> { + ) -> Vec<(PhysicalASAPNodeId, SummaryMaintenanceLifecycle)> { candidates .deployments() .iter() @@ -3134,7 +3200,7 @@ mod tests { use SummaryMaintenanceLifecycleChoiceError as E; let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); - let select = |model: &dyn CostModel, choice: &dyn Fn(PostAsapNodeId) -> Vec<_>| { + let select = |model: &dyn CostModel, choice: &dyn Fn(PhysicalASAPNodeId) -> Vec<_>| { let candidates = continuous_candidates(&workload, &data, model); let id = candidates.deployments()[0].post_asap_node_id; (id, candidates.select(&choice(id)).unwrap_err()) @@ -3176,10 +3242,8 @@ mod tests { vec![(id, continuous.clone()), (id, continuous.clone())] }); assert_eq!(error, E::DuplicateChoice(id)); - let (_, error) = select(&UnitCosts, &|_| { - vec![(PostAsapNodeId(u32::MAX), continuous.clone())] - }); - assert_eq!(error, E::UnknownSummary(PostAsapNodeId(u32::MAX))); + let (_, error) = select(&UnitCosts, &|_| vec![(usize::MAX, continuous.clone())]); + assert_eq!(error, E::UnknownSummary(usize::MAX)); } // Nested states on one maintenance path must share an evaluation schedule. @@ -3219,14 +3283,10 @@ mod tests { #[test] fn enumeration_lists_each_unique_summary_state_once() { let shared = summary(); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&shared), Rc::clone(&shared), summary()], - }, - schema: shared.schema.clone(), - guarantee: None, - }); + let root = test_binary( + test_binary(evaluation(&shared), evaluation(&shared)), + evaluation(&summary()), + ); let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); let candidates = enumerate_summary_maintenance_lifecycles( root, @@ -3250,26 +3310,41 @@ mod tests { .any(|deployment| Rc::ptr_eq(&deployment.summary, &shared))); } - fn readout(state: &Rc) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: Rc::clone(state), - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::QueryTime, - }, - schema: Schema { - closed: true, - unique_keys: vec![], - fields: vec![Field { - table: None, - name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), - nullable: false, - }], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("sum")), - }) + fn evaluation(state: &Rc) -> Rc { + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: Rc::clone(state), + }), + Schema::lifted( + vec![Field::new( + "value", + FieldDataType::Plain(DataType::Float64), + false, + )], + None, + ), + ) + .with_guarantee(Some(ResultGuarantee::exact("sum"))), + ) + } + + fn test_binary(lhs: Rc, rhs: Rc) -> Rc { + let schema = lhs.schema.clone(); + Rc::new(OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::BinaryOp { + lhs, + rhs, + return_bool: false, + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }), + schema, + )) } fn lifecycle_matching( @@ -3287,12 +3362,12 @@ mod tests { /// Bind the lifecycle `choose` picks for every state of `root`, then /// derive the timed DAG. fn timed_dag( - root: Rc, + root: Rc, workload: &QueryWorkload, data: &DataWorkload, horizon: Option, choose: impl Fn(&SummaryMaintenanceDeployment) -> SummaryMaintenanceLifecycle, - ) -> PostAsapDAG { + ) -> PhysicalASAPDAG { let candidates = enumerate_summary_maintenance_lifecycles( root, WorkloadDemand::new_with_data(workload, data, &[0]), @@ -3317,15 +3392,21 @@ mod tests { } /// Operator kinds in node-id order, each paired with its timing. - fn timings(dag: &PostAsapDAG) -> Vec<(&'static str, ExecutionTiming)> { + fn timings(dag: &PhysicalASAPDAG) -> Vec<(&'static str, ExecutionTiming)> { dag.nodes .iter() .map(|node| { let kind = match node.payload { - PostAsapOperatorPayload::Fallback { .. } => "raw", - PostAsapOperatorPayload::SummaryAgg { .. } => "state", - PostAsapOperatorPayload::Value { .. } => "readout", - PostAsapOperatorPayload::Binary { .. } => "binary", + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::BinaryOp { .. }) => "binary", + PhysicalASAPOperatorPayload::NonASAP(_) => "raw", + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { .. }) => "state", + PhysicalASAPOperatorPayload::ASAP(ASAPOp::FinalizeExactAccumulator { + .. + }) + | PhysicalASAPOperatorPayload::ASAP(ASAPOp::EvaluatePopulation { .. }) => { + "evaluation" + } + _ => "other", }; (kind, node.output_state.timing) @@ -3337,7 +3418,7 @@ mod tests { const QUERY: ExecutionTiming = ExecutionTiming::QueryTime; // Every retained lifecycle kind runs its state and inputs at ingestion - // time and its readout at query time. + // time and its evaluation at query time. #[test] fn retained_lifecycles_time_state_and_inputs_at_ingestion() { let mut scheduled = batch(Predictability::Predictable { @@ -3377,7 +3458,7 @@ mod tests { ]; for (workload, data, horizon, kind) in cases { let dag = timed_dag( - readout(&summary()), + evaluation(&summary()), &workload, &data, horizon, @@ -3385,21 +3466,21 @@ mod tests { ); assert_eq!( timings(&dag), - [("raw", INGEST), ("state", INGEST), ("readout", QUERY)] + [("raw", INGEST), ("state", INGEST), ("evaluation", QUERY)] ); } } - // An Ephemeral state, its raw input, and its readout all run at query time. + // An Ephemeral state, its raw input, and its evaluation all run at query time. #[test] fn ephemeral_lifecycle_times_state_and_downstream_at_query() { let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let dag = timed_dag(readout(&summary()), &workload, &at_rest(), None, |_| { + let dag = timed_dag(evaluation(&summary()), &workload, &at_rest(), None, |_| { SummaryMaintenanceLifecycle::Ephemeral }); assert_eq!( timings(&dag), - [("raw", QUERY), ("state", QUERY), ("readout", QUERY)] + [("raw", QUERY), ("state", QUERY), ("evaluation", QUERY)] ); } @@ -3408,25 +3489,9 @@ mod tests { #[test] fn shared_state_is_timed_once_for_all_consumers() { let state = summary(); - let lhs = readout(&state); + let lhs = evaluation(&state); let rhs = Rc::new(lhs.as_ref().clone()); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - lhs, - rhs, - operator: asap_types::post_asap::BinaryOperator { - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - timing: QUERY, - }, - schema: readout(&state).schema.clone(), - guarantee: None, - }); + let root = test_binary(lhs, rhs); let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let dag = timed_dag(root, &workload, &data, Some(Horizon(10.0)), |deployment| { @@ -3438,8 +3503,8 @@ mod tests { [ ("raw", INGEST), ("state", INGEST), - ("readout", QUERY), - ("readout", QUERY), + ("evaluation", QUERY), + ("evaluation", QUERY), ("binary", QUERY), ] ); @@ -3484,7 +3549,7 @@ mod tests { let data = at_rest(); let demand = WorkloadDemand::new_with_data(&workload, &data, &[0]); let plan = plan_summary_maintenance_lifecycles( - readout(&summary()), + evaluation(&summary()), demand, 1_000, None, @@ -3499,7 +3564,7 @@ mod tests { ) ); let raw = plan_summary_maintenance_lifecycles( - crate::replacement::keep_pre_asap(&sum_query()).unwrap(), + crate::replacement::retain_exact(&sum_query()).unwrap(), demand, 1_000, None, @@ -3509,16 +3574,13 @@ mod tests { .unwrap(); assert_eq!( timings(&raw.execution_timed_dag().unwrap()), - [("raw", QUERY)] + [("raw", QUERY), ("raw", QUERY)] ); } /// A strategy-built `sum(a)` over one maintained current-series population. - fn population_readout() -> Rc { - let target = Rc::new(crate::test_support::lower_promql( - "sum(a)", - AccuracyTarget::Exact, - )); + fn population_evaluation() -> Rc { + let target = crate::test_support::lower_promql("sum(a)", AccuracyTarget::Exact); crate::maintained_population::MaintainedPopulationStrategy::new(std::slice::from_ref( &target, )) @@ -3526,24 +3588,21 @@ mod tests { .unwrap() } - fn is_population(node: &SummaryNode) -> bool { + fn is_population(node: &OperatorNode) -> bool { matches!( - node.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { .. }, - .. - } + node.operator, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) ) } - fn population_timings(dag: &PostAsapDAG) -> Vec<(&'static str, ExecutionTiming)> { + fn population_timings(dag: &PhysicalASAPDAG) -> Vec<(&'static str, ExecutionTiming)> { dag.nodes .iter() .zip(timings(dag)) .map(|(node, (kind, timing))| match node.payload { - PostAsapOperatorPayload::Value { - operation: ValueOperation::MaintainPopulation { .. }, - } => ("population", timing), + PhysicalASAPOperatorPayload::ASAP(ASAPOp::MaintainPopulation { .. }) => { + ("population", timing) + } _ => (kind, timing), }) .collect() @@ -3556,7 +3615,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let candidates = enumerate_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), @@ -3594,7 +3653,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let plan = plan_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), @@ -3624,7 +3683,7 @@ mod tests { let workload = workload(vec![], vec![repeating()], data.clone()); let timed = |lifecycle: SummaryMaintenanceLifecycle| { population_timings(&timed_dag( - population_readout(), + population_evaluation(), &workload, &data, Some(Horizon(10.0)), @@ -3633,11 +3692,21 @@ mod tests { }; assert_eq!( timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), - [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + [ + ("raw", INGEST), + ("raw", INGEST), + ("population", INGEST), + ("evaluation", QUERY) + ] ); assert_eq!( timed(SummaryMaintenanceLifecycle::Ephemeral), - [("raw", QUERY), ("population", QUERY), ("readout", QUERY)] + [ + ("raw", QUERY), + ("raw", QUERY), + ("population", QUERY), + ("evaluation", QUERY) + ] ); } @@ -3649,7 +3718,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let plan = plan_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), @@ -3664,7 +3733,12 @@ mod tests { )); assert_eq!( population_timings(&plan.execution_timed_dag().unwrap()), - [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] + [ + ("raw", INGEST), + ("raw", INGEST), + ("population", INGEST), + ("evaluation", QUERY) + ] ); } @@ -3672,39 +3746,38 @@ mod tests { // deployment: the state's lifecycle times it. #[test] fn population_feeding_summary_state_follows_that_state() { - let SummaryExpr::ValueOperation { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child: population, .. - } = &population_readout().expr + }) = &population_evaluation().operator else { unreachable!() }; let state = summary(); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, grouping, .. - } = &state.expr + }) = &state.operator else { unreachable!() }; - let state = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(population), - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: None, - }, - ..state.as_ref().clone() + let mut copy = state.as_ref().clone(); + copy.operator = Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(population), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: None, }); + let state = Rc::new(copy); let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let timed = |lifecycle: SummaryMaintenanceLifecycle| { population_timings(&timed_dag( - readout(&state), + evaluation(&state), &workload, &data, Some(Horizon(10.0)), @@ -3717,19 +3790,21 @@ mod tests { assert_eq!( timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), [ + ("raw", INGEST), ("raw", INGEST), ("population", INGEST), ("state", INGEST), - ("readout", QUERY) + ("evaluation", QUERY) ] ); assert_eq!( timed(SummaryMaintenanceLifecycle::Ephemeral), [ + ("raw", QUERY), ("raw", QUERY), ("population", QUERY), ("state", QUERY), - ("readout", QUERY) + ("evaluation", QUERY) ] ); } @@ -3739,59 +3814,40 @@ mod tests { // timed by the state's lifecycle. #[test] fn shared_population_follows_its_summary_consumer() { - let direct = population_readout(); - let SummaryExpr::ValueOperation { + let direct = population_evaluation(); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child: population, .. - } = &direct.expr + }) = &direct.operator else { unreachable!() }; let state = summary(); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, grouping, .. - } = &state.expr + }) = &state.operator else { unreachable!() }; - let state = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(population), - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: None, - }, - ..state.as_ref().clone() + let mut copy = state.as_ref().clone(); + copy.operator = Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(population), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: None, }); - let binary = |lhs: Rc, rhs: Rc| { - Rc::new(SummaryNode { - schema: lhs.schema.clone(), - expr: SummaryExpr::BinaryOp { - lhs, - rhs, - operator: asap_types::post_asap::BinaryOperator { - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - timing: QUERY, - }, - guarantee: None, - }) - }; + let state = Rc::new(copy); + let binary = test_binary; let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); for root in [ - binary(Rc::clone(&direct), readout(&state)), - binary(readout(&state), Rc::clone(&direct)), + binary(Rc::clone(&direct), evaluation(&state)), + binary(evaluation(&state), Rc::clone(&direct)), ] { for lifecycle in [ SummaryMaintenanceLifecycle::ContinuouslyMaintained, @@ -3830,7 +3886,7 @@ mod tests { let data = continuous(1_000, 60_000); let workload = workload(vec![], vec![repeating()], data.clone()); let mut plan = plan_summary_maintenance_lifecycles( - population_readout(), + population_evaluation(), WorkloadDemand::new_with_data(&workload, &data, &[0]), 1_000, Some(Horizon(10.0)), diff --git a/crates/asap-aware-mapping/src/test_support.rs b/crates/asap-aware-mapping/src/test_support.rs index 612cf7c0d..4147d05eb 100644 --- a/crates/asap-aware-mapping/src/test_support.rs +++ b/crates/asap-aware-mapping/src/test_support.rs @@ -1,11 +1,16 @@ -use asap_types::pre_asap::QueryExpr; +// Shared fixture helpers; not every test module uses every helper. +#![allow(dead_code)] + +use std::rc::Rc; + +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; -pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -35,3 +40,162 @@ pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> QueryExpr { .pop() .unwrap() } + +// ── Shared pre-ASAP fixture builders ───────────────────────────────────── +// +// Every builder returns an `Rc` whose schema is derived by +// `OperatorNode::new_shared`, so a fixture is exactly what a front end +// would hand the planner. Added by the test migration; only add here, never +// rename or remove (several test modules share these). + +use std::time::Duration; + +use asap_types::ir::operator_properties::{GroupKeys, Reduction, Source}; +use asap_types::ir::timing::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; +use asap_types::ir::{NonASAPOp, Predicate, ScalarExpr, TimeRangeKind}; +use asap_types::pre_asap::agg_intent::AggIntent; +use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; + +/// A `TimeSeries("m")` scan over `[ts(0), value(1), labels...]`, time index 0, +/// no unique key. +pub(crate) fn metric_scan(labels: &[&str]) -> Rc { + metric_scan_with_keys(labels, vec![]) +} + +/// [`metric_scan`] with explicit `unique_keys` (a `[[0]]` key makes CSE +/// willing to hoist the scan). +pub(crate) fn metric_scan_with_keys( + labels: &[&str], + unique_keys: Vec>, +) -> Rc { + let mut columns = vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ]; + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); + scan("m", Schema::with_time_index(columns, 0, unique_keys)) +} + +/// A predicate-free `TimeSeries(metric)` scan with the given schema. +pub(crate) fn scan(metric: &str, schema: Schema) -> Rc { + scan_from( + Source::TimeSeries { + metric: metric.into(), + }, + schema, + ) +} + +pub(crate) fn scan_from(source: Source, schema: Schema) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source, + predicates: vec![], + schema, + })) + .unwrap() +} + +/// A general aggregate node. +pub(crate) fn aggregate( + reduction: Reduction, + measures: Vec, + output_names: Vec, + having: Option, + child: Rc, +) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction, + measures, + output_names, + filters: vec![], + having, + child, + })) + .unwrap() +} + +/// `intent by (by)` — a single-measure, `HAVING`-free grouped aggregate. +pub(crate) fn agg( + by: Vec, + intent: AggIntent, + child: Rc, +) -> Rc { + aggregate(Reduction::by(by), vec![intent], vec![], None, child) +} + +/// `intent without (excluded)`. +pub(crate) fn without_agg( + excluded: Vec, + intent: AggIntent, + child: Rc, +) -> Rc { + aggregate( + Reduction::Reduce(GroupKeys::without(excluded)), + vec![intent], + vec![], + None, + child, + ) +} + +/// A per-entity (per-series) single-measure aggregate. +pub(crate) fn agg_per_entity(intent: AggIntent, child: Rc) -> Rc { + aggregate(Reduction::PerEntity, vec![intent], vec![], None, child) +} + +pub(crate) fn filter(pred: ScalarExpr, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(pred), + child, + })) + .unwrap() +} + +pub(crate) fn dedup(cols: Vec, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols, + child, + })) + .unwrap() +} + +/// An explicit range selector `child[range]`. +pub(crate) fn time_range(range: Duration, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + range, + kind: TimeRangeKind::Range, + child, + })) + .unwrap() +} + +/// `root` timed under the default (every summary maintained) lifecycle +/// assignment — the shape export and the post-ASAP validators consume. +pub(crate) fn timed(root: &Rc) -> Rc { + apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .expect("default lifecycle timings apply") +} + +/// Time `root` under the default lifecycle assignment (which runs every +/// data-state / population-contract check) and export it as a physical ASAP DAG. +pub(crate) fn time_and_export( + root: &Rc, +) -> Result< + asap_types::ir::physical_export::PhysicalASAPDAG, + asap_types::post_asap::execution_data_state::ExecutionDataStateError, +> { + let timed = apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + )?; + asap_types::ir::physical_export::compile_physical_asap_dag(&timed) +} diff --git a/crates/asap-aware-mapping/src/topk_reuse.rs b/crates/asap-aware-mapping/src/topk_reuse.rs index 8329569d1..aec95292a 100644 --- a/crates/asap-aware-mapping/src/topk_reuse.rs +++ b/crates/asap-aware-mapping/src/topk_reuse.rs @@ -7,7 +7,7 @@ use std::rc::Rc; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::{NonASAPOp, OperatorNode}; use crate::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, @@ -15,22 +15,23 @@ use crate::replacement::{ /// Derives a smaller top-k result from a compatible larger top-k sibling. pub struct TopKLimitReuseStrategy { - limits: Vec>, + limits: Vec>, } impl TopKLimitReuseStrategy { - pub fn new(limits: &[Rc]) -> Self { + pub fn new(limits: &[Rc]) -> Self { Self { limits: limits.to_vec(), } } - fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { - let QueryExpr::Limit { - n: target_n, + fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + let Some(NonASAPOp::Limit { + n: Some(target_n), offset: 0, child: target_child, - } = target.root.as_ref() + .. + }) = target.root.non_asap() else { return Vec::new(); }; @@ -42,11 +43,12 @@ impl TopKLimitReuseStrategy { if Rc::ptr_eq(candidate, target.root) { return false; } - let QueryExpr::Limit { - n, + let Some(NonASAPOp::Limit { + n: Some(n), offset: 0, child, - } = candidate.as_ref() + .. + }) = candidate.non_asap() else { return false; }; @@ -56,8 +58,8 @@ impl TopKLimitReuseStrategy { .collect(); // Prefer the smallest sufficient materialized top-k when several // larger siblings are available. - sources.sort_by_key(|source| match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + sources.sort_by_key(|source| match source.non_asap() { + Some(NonASAPOp::Limit { n: Some(n), .. }) => *n, _ => unreachable!(), }); sources @@ -70,34 +72,38 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { } fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { - let QueryExpr::Limit { - n: target_n, + let Some(NonASAPOp::Limit { + n: Some(target_n), offset: 0, + partition_by, .. - } = target.root.as_ref() + }) = target.root.non_asap() else { return Vec::new(); }; self.larger_sources(target) .into_iter() - .map(|source| { - let source_n = match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + .filter_map(|source| { + let source_n = match source.non_asap() { + Some(NonASAPOp::Limit { n: Some(n), .. }) => *n, _ => unreachable!(), }; - ReplacementSubDAG { + let rewritten = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(*target_n), + offset: 0, + partition_by: partition_by.clone(), + child: Rc::clone(source), + })) + .ok()?; + Some(ReplacementSubDAG { strategy: "TopKLimitReuseStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Limit { - n: *target_n, - offset: 0, - child: Rc::clone(source), - })), + replacement: Replacement::SubDAG(rewritten), provenance: ReplacementProvenance::LogicalRewrite, rationale: format!( "derives top-{target_n} from the compatible shared top-{source_n} result; both rank the identical input with the same ordering" ), - } + }) }) .collect() } @@ -106,38 +112,39 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::{Schema, Source}; + use crate::test_support::scan; + use asap_types::ir::operator_properties::GroupKeys; + use asap_types::pre_asap::Schema; - fn scan_named(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![], - schema: Schema::with_time_index(vec![], 0, vec![]), - }) + fn scan_named(metric: &str) -> Rc { + scan(metric, Schema::with_time_index(vec![], 0, vec![])) + } + + fn limit(n: usize, offset: usize, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(n), + offset, + partition_by: GroupKeys::none(), + child, + })) + .unwrap() } #[test] fn smaller_limit_reuses_larger_compatible_limit() { let child = scan_named("m"); - let small = Rc::new(QueryExpr::Limit { - n: 5, - offset: 0, - child: Rc::clone(&child), - }); - let large = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child, - }); + let small = limit(5, 0, Rc::clone(&child)); + let large = limit(10, 0, child); let strategy = TopKLimitReuseStrategy::new(&[Rc::clone(&small), Rc::clone(&large)]); let replacements = strategy.replacements(&TargetSubDAG::new(&small)); assert_eq!(replacements.len(), 1); - let Replacement::Rewrite(rewrite) = &replacements[0].replacement else { + let Replacement::SubDAG(rewrite) = &replacements[0].replacement else { panic!() }; - let QueryExpr::Limit { n: 5, child, .. } = rewrite.as_ref() else { + let Some(NonASAPOp::Limit { + n: Some(5), child, .. + }) = rewrite.non_asap() + else { panic!() }; assert!(Rc::ptr_eq(child, &large)); @@ -147,21 +154,9 @@ mod tests { fn offset_or_different_input_is_not_reused() { let a = scan_named("a"); let b = scan_named("b"); - let small = Rc::new(QueryExpr::Limit { - n: 5, - offset: 0, - child: a, - }); - let large = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child: b, - }); - let offset = Rc::new(QueryExpr::Limit { - n: 20, - offset: 1, - child: scan_named("a"), - }); + let small = limit(5, 0, a); + let large = limit(10, 0, b); + let offset = limit(20, 1, scan_named("a")); let strategy = TopKLimitReuseStrategy::new(&[Rc::clone(&small), large, offset]); assert!(!strategy.matches(&TargetSubDAG::new(&small))); } diff --git a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs b/crates/asap-aware-mapping/tests/physical_handoff_cost.rs index a8a9ed42d..141750f61 100644 --- a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs +++ b/crates/asap-aware-mapping/tests/physical_handoff_cost.rs @@ -5,7 +5,7 @@ use asap_aware_mapping::analytical_cost::{ use asap_aware_mapping::physical_operator_statistics::{ ComparisonScope, EdgeStatistics, OperatorStatistics, ScanSelection, UnaryEdgeStatistics, }; -use asap_types::pre_asap::query_expr::Source; +use asap_types::ir::operator_properties::Source; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; diff --git a/crates/asap-aware-mapping/tests/storage_io.rs b/crates/asap-aware-mapping/tests/storage_io.rs index e9290d20a..811328f6b 100644 --- a/crates/asap-aware-mapping/tests/storage_io.rs +++ b/crates/asap-aware-mapping/tests/storage_io.rs @@ -5,7 +5,7 @@ use asap_aware_mapping::analytical_cost::{ use asap_aware_mapping::physical_operator_statistics::{ ComparisonScope, EdgeStatistics, OperatorStatistics, ScanSelection, UnaryEdgeStatistics, }; -use asap_types::pre_asap::query_expr::Source; +use asap_types::ir::operator_properties::Source; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 2d3e7b569..b15154b4a 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -14,7 +14,7 @@ thread pool. Poll multiple root streams concurrently when they share inputs. `operators::Operator` implements native batch sources, scalar values, projection, filtering, grouped exact aggregation, semi-join, grouped Sort and -Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. +Limit, vector-to-scalar conversion, Union, and summary construction/merge/evaluation. Sort followed by Limit implements grouped ranking; no dedicated TopK physical operator is needed. Summary construction updates state batch by batch. End of input means the supplied query range or ingestion window is complete. @@ -47,7 +47,7 @@ assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); # Ok::<(), asap_physical_operators::dag::Error>(()) ``` -`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDAG`) and typed input contracts. +`physical_planner::compile` accepts a logical Post-ASAP DAG (`PhysicalASAPDAG`) and typed input contracts. The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and schema mismatches before starting a source. Implement `PhysicalOperator` for a deployment source, including asynchronous I/O; computation operators remain in @@ -59,7 +59,7 @@ Plain values preserve Planner scalar/collection types and nullability. Numeric arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. Boolean predicates use three-valued logic. Native summary states currently cover exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Binding checks family, -parameters and readout compatibility; source batches also validate state payloads. +parameters and evaluation compatibility; source batches also validate state payloads. Existing accumulator algorithms are reused as kernels behind these operators. This crate is owned by ASAPPlanner. Its `planner-types` dependency is the local @@ -78,9 +78,9 @@ See [the design](../../docs/design_docs/physical-planning-and-deployment.md). - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. - `sources`: raw-source interface, Scan and the memory connector. - `physical_planner`: native operator lowering, typed input contracts and checked instantiation. -- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed readout and update adapters. -- `readout`: readouts over merged exact summary states. -- `capability`: explicit kernel, native-batch and typed readout validation. +- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed evaluation and update adapters. +- `evaluation`: evaluations over merged exact summary states. +- `capability`: explicit kernel, native-batch and typed evaluation validation. The `dag`, `factory`, `traits` and `arithmetic` paths are re-exports. They contain no alternative execution implementations. @@ -104,7 +104,7 @@ There is no spill or partitioned parallel execution in this implementation. ## Physical compilation and deployment inputs -`physical_planner::compile` accepts a Planner `PostAsapDAG`, typed +`physical_planner::compile` accepts a Planner `PhysicalASAPDAG`, typed `InputContract`s and output roots. It returns a reusable `CompiledPhysicalDAG` containing selected native operators and no live readers. Compilation validates schemas, input ordering, sharing and boundedness before deployment source access. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs index 5ed1c5501..8bdb1891e 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/asap-physical-operators/src/capability.rs @@ -2,22 +2,22 @@ //! //! `validate_summary_kernel` checks update kernels, including families without a //! native batch representation. `validate_native_family` and -//! `validate_sketch_readout` / `validate_exact_readout` check native state and readout support. -//! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. +//! `validate_sketch_evaluation` / `validate_exact_evaluation` check native state and evaluation support. +//! Keyed weighted-frequency evaluations are checked by `Operator::keyed_evaluation`. //! A successful kernel check alone does not mean a physical DAG will bind. //! //! Stored-state encodings belong to deployments. Full plan acceptance is //! owned by `binding`, which also validates schemas, expressions and inputs. use crate::Error; use planner_types::post_asap::{ - ExactKind, ExactParams, FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams, - SketchStatistic, SummaryUpdate, + ExactKind, ExactParams, FieldDataType as SummaryFamilyType, GroupingStrategy, SketchAlgorithm, + SketchParams, SketchStatistic, SummaryUpdate, }; /// Check the same contract used by `create_planner_accumulator` before a plan /// is accepted. Execution timing is deliberately not a kernel property. pub fn validate_summary_kernel( - family: &FieldDataType, + family: &SummaryFamilyType, input: &SummaryUpdate, grouping: &GroupingStrategy, ) -> Result<(), String> { @@ -25,7 +25,7 @@ pub fn validate_summary_kernel( return Err("shared summary grouping has no registered kernel".into()); } let keyed = match family { - FieldDataType::ExactAggregate(kind, params) => { + SummaryFamilyType::ExactAggregate(kind, params) => { use ExactKind as K; use ExactParams as P; if !matches!( @@ -41,7 +41,7 @@ pub fn validate_summary_kernel( } input.item.is_some() } - FieldDataType::Sketch(kind, layout) => { + SummaryFamilyType::Sketch(kind, layout) => { if layout != grouping { return Err("Planner family and operator grouping disagree".into()); } @@ -138,9 +138,9 @@ pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::Summar ) } -pub fn validate_native_family(family: &FieldDataType) -> Result<(), Error> { +pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { use planner_types::post_asap::SketchAlgorithm as A; - if let FieldDataType::Sketch(kind, grouping) = family { + if let SummaryFamilyType::Sketch(kind, grouping) = family { // Plain Count-Min is native as stored state only: it merges and reads // its bare count, but the DAG does not build it from rows. if let (A::Cms, SketchParams::Cms { width, depth }) = (kind.algorithm(), kind.params()) { @@ -165,8 +165,8 @@ pub fn validate_native_family(family: &FieldDataType) -> Result<(), Error> { } } match family { - FieldDataType::ExactAggregate(..) => {} - FieldDataType::Sketch(kind, _) + SummaryFamilyType::ExactAggregate(..) => {} + SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} _ => { return Err(Error::Invalid( @@ -184,9 +184,9 @@ pub fn validate_native_family(family: &FieldDataType) -> Result<(), Error> { .map_err(Error::Invalid) } -/// A sketch readout is native only for the families Planner can read directly. -pub fn validate_sketch_readout( - family: &FieldDataType, +/// A sketch evaluation is native only for the families Planner can read directly. +pub fn validate_sketch_evaluation( + family: &SummaryFamilyType, query: &SketchStatistic, ) -> Result<(), Error> { validate_native_family(family)?; @@ -194,12 +194,12 @@ pub fn validate_sketch_readout( // A point count without an item value reads the total count. let bare_count = matches!(query, SketchStatistic::PointCount { value: None, .. }); let supported = match family { - FieldDataType::Sketch(kind, _) => match (kind.algorithm(), query) { + SummaryFamilyType::Sketch(kind, _) => match (kind.algorithm(), query) { (A::Kll, SketchStatistic::Quantile { q }) | (A::DDSketch, SketchStatistic::Quantile { q }) => { if !(0.0..=1.0).contains(q) { return Err(Error::Invalid( - "quantile readout requires quantile in [0,1]".into(), + "quantile evaluation requires quantile in [0,1]".into(), )); } true @@ -208,7 +208,7 @@ pub fn validate_sketch_readout( (A::Hll, SketchStatistic::Cardinality) => true, (A::Hll, _) => bare_count, // Only count intents read a Count-Min bare count, and their - // updates have unit weight; the readout is typed Int64 on that basis. + // updates have unit weight; the evaluation is typed Int64 on that basis. (A::Cms, _) => bare_count, _ => false, }, @@ -216,36 +216,39 @@ pub fn validate_sketch_readout( }; if !supported { return Err(Error::Invalid( - "readout is not implemented for this summary family".into(), + "evaluation is not implemented for this summary family".into(), )); } Ok(()) } -/// An exact readout must match the exact family it reads. -pub fn validate_exact_readout( - family: &FieldDataType, - readout: &crate::summary_kernels::exact::ExactReadout, +/// An exact evaluation must match the exact family it reads. +pub fn validate_exact_evaluation( + family: &SummaryFamilyType, + evaluation: &crate::summary_kernels::exact::ExactEvaluation, ) -> Result<(), Error> { validate_native_family(family)?; use crate::Statistic as S; use planner_types::post_asap::ExactKind as E; let supported = matches!( - (family, readout.statistic), - (FieldDataType::ExactAggregate(E::Sum, _), S::Sum) - | (FieldDataType::ExactAggregate(E::Count, _), S::Count) - | (FieldDataType::ExactAggregate(E::Min, _), S::Min) - | (FieldDataType::ExactAggregate(E::Max, _), S::Max) - | (FieldDataType::ExactAggregate(E::Rate, _), S::Rate) - | (FieldDataType::ExactAggregate(E::Increase, _), S::Increase) + (family, evaluation.statistic), + (SummaryFamilyType::ExactAggregate(E::Sum, _), S::Sum) + | (SummaryFamilyType::ExactAggregate(E::Count, _), S::Count) + | (SummaryFamilyType::ExactAggregate(E::Min, _), S::Min) + | (SummaryFamilyType::ExactAggregate(E::Max, _), S::Max) + | (SummaryFamilyType::ExactAggregate(E::Rate, _), S::Rate) + | ( + SummaryFamilyType::ExactAggregate(E::Increase, _), + S::Increase + ) ); if !supported { return Err(Error::Invalid( - "readout is not implemented for this summary family".into(), + "evaluation is not implemented for this summary family".into(), )); } - if readout.lookback_ms.is_some_and(|lookback| { - lookback <= 0 || !matches!(readout.statistic, S::Rate | S::Increase) + if evaluation.lookback_ms.is_some_and(|lookback| { + lookback <= 0 || !matches!(evaluation.statistic, S::Rate | S::Increase) }) { return Err(Error::Invalid("invalid exact counter lookback".into())); } diff --git a/crates/asap-physical-operators/src/evaluation.rs b/crates/asap-physical-operators/src/evaluation.rs new file mode 100644 index 000000000..843145ce2 --- /dev/null +++ b/crates/asap-physical-operators/src/evaluation.rs @@ -0,0 +1,111 @@ +//! Evaluations over merged exact summary states. +use crate::summary_kernels::exact::ExactAccumulator; +use crate::{AggregateCore, KeyByLabelValues, Statistic}; +use std::sync::Arc; + +fn merge_exact_states( + states: impl IntoIterator>, +) -> Result { + let mut states = states.into_iter(); + let exact = |state: &Arc| { + state + .as_any() + .downcast_ref::() + .cloned() + .ok_or_else(|| "evaluation requires Planner exact state".to_string()) + }; + let mut merged = exact(&states.next().ok_or("empty exact state input")?)?; + for state in states { + merged + .merge_from(&exact(&state)?) + .map_err(|error| error.to_string())?; + } + Ok(merged) +} + +/// PromQL counter evaluations omit a series with fewer than two samples. Other +/// state/type/range failures remain errors rather than empty results. +pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statistic) -> bool { + matches!(statistic, Statistic::Rate | Statistic::Increase) + && state + .as_any() + .downcast_ref::() + .is_some_and(|state| state.insufficient_counter_samples(statistic, &None)) +} + +/// Merge already selected exact panes and read one population. `None` means +/// the population is absent from the result: a counter with too few samples, +/// or an empty MIN/MAX. +pub fn exact_evaluation( + states: impl IntoIterator>, + statistic: Statistic, + range_ms: Option<(i64, i64)>, + key: Option<&KeyByLabelValues>, +) -> Result, String> { + let merged = merge_exact_states(states)?; + if merged.insufficient_counter_samples(statistic, &key.cloned()) { + return Ok(None); + } + merged + .evaluation(statistic, range_ms, key) + .map_err(|error| error.to_string()) +} + +#[cfg(test)] +mod counter_tests { + use super::*; + use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType as SummaryFamilyType}; + + fn counter(kind: ExactKind, params: ExactParams, keyed: bool) -> ExactAccumulator { + ExactAccumulator::new(SummaryFamilyType::ExactAggregate(kind, params), keyed).unwrap() + } + + // A counter population with a single sample is absent, keyed or not. + #[test] + fn planner_counter_population_omits_insufficient_samples() { + for (kind, params, statistic) in [ + (ExactKind::Rate, ExactParams::Rate, Statistic::Rate), + ( + ExactKind::Increase, + ExactParams::Increase, + Statistic::Increase, + ), + ] { + for keyed in [false, true] { + let mut state = counter(kind.clone(), params.clone(), keyed); + let key = keyed.then(|| KeyByLabelValues::new_with_labels(vec!["checkout".into()])); + state.update(key.as_ref(), 10., 10_000); + assert_eq!( + exact_evaluation( + [Arc::new(state) as Arc], + statistic, + None, + key.as_ref() + ) + .unwrap(), + None + ); + } + } + } + + // Two ordered samples read a rate; an inverted range and empty input fail. + #[test] + fn sparse_counter_is_absent_but_invalid_ranges_still_fail() { + let mut state = counter(ExactKind::Rate, ExactParams::Rate, false); + state.update(None, 10., 10_000); + let rate = Statistic::Rate; + let one = [Arc::new(state.clone()) as Arc]; + assert_eq!( + exact_evaluation(one, rate, Some((0, 60_000)), None).unwrap(), + None + ); + state.update(None, 20., 20_000); + let two = || [Arc::new(state.clone()) as Arc]; + assert!(exact_evaluation(two(), rate, Some((0, 60_000)), None) + .unwrap() + .is_some()); + assert!(exact_evaluation(two(), rate, Some((60_000, 0)), None).is_err()); + assert!(exact_evaluation([], rate, Some((0, 60_000)), None).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/expressions/arithmetic.rs b/crates/asap-physical-operators/src/expressions/arithmetic.rs index e0766763d..64dbb8301 100644 --- a/crates/asap-physical-operators/src/expressions/arithmetic.rs +++ b/crates/asap-physical-operators/src/expressions/arithmetic.rs @@ -20,12 +20,13 @@ pub fn evaluate_float64_arithmetic( /// Execute the Planner binary contract after a deployment has resolved matching rows. pub fn evaluate_binary( - operator: &planner_types::post_asap::BinaryOperator, + operator: &crate::expressions::binary::BinaryOperator, left: f64, right: f64, ) -> Result { + use crate::expressions::binary::BinaryOpKind; use crate::{values::Value, Error}; - use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind}; + use planner_types::pre_asap::ArithmeticOpKind; let invalid = || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); if operator.vector_match.is_some() { diff --git a/crates/asap-physical-operators/src/expressions/binary.rs b/crates/asap-physical-operators/src/expressions/binary.rs index c115fd908..81c1c2656 100644 --- a/crates/asap-physical-operators/src/expressions/binary.rs +++ b/crates/asap-physical-operators/src/expressions/binary.rs @@ -1,3 +1,35 @@ -//! Temporary kernel aliases during the unified compiler migration. -pub use planner_types::post_asap::BinaryOperator; -pub use planner_types::pre_asap::BinaryOpKind; +//! Execution configuration for a binary kernel, including comparison evaluation mode. +use planner_types::pre_asap::{ + ArithmeticOpKind, CompareOpKind, PromQLVectorSetOpKind, VectorMatch, +}; +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum BinaryOpKind { + Arithmetic(ArithmeticOpKind), + Compare(CompareOpKind), + CompareBool(CompareOpKind), + Set(PromQLVectorSetOpKind), +} +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub struct BinaryOperator { + pub kind: BinaryOpKind, + pub vector_match: Option, + pub checked_relative_division: bool, + pub checked_finite_division: bool, +} +impl BinaryOperator { + pub fn from_logical(operator: &planner_types::ir::BinaryOperator, return_bool: bool) -> Self { + use planner_types::pre_asap::BinaryOpKind as L; + Self { + kind: match &operator.kind { + L::Arithmetic(op) => BinaryOpKind::Arithmetic(op.clone()), + L::Compare(op) if return_bool => BinaryOpKind::CompareBool(op.clone()), + L::Compare(op) => BinaryOpKind::Compare(op.clone()), + L::CompareBool(op) => BinaryOpKind::CompareBool(op.clone()), + L::Set(op) => BinaryOpKind::Set(op.clone()), + }, + vector_match: operator.vector_match.clone(), + checked_relative_division: operator.checked_relative_division, + checked_finite_division: operator.checked_finite_division, + } + } +} diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs index 5c1a29576..aad035516 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -7,17 +7,15 @@ use planner_types::pre_asap::{ArithmeticOpKind, DataType}; pub mod arithmetic; pub mod binary; mod planner; -pub mod unified_planner; pub use planner::CompiledExpression; #[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub enum Expression { Binary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, left: Box, right: Box, }, Planner(Box), - UnifiedPlanner(Box), Column(usize), ExactFloat64(usize), FiniteFloat64(Box), @@ -56,9 +54,6 @@ pub enum Expression { IsNull(Box), } impl Expression { - pub fn unified_planner(expression: unified_planner::CompiledExpression) -> Self { - Self::UnifiedPlanner(Box::new(expression)) - } pub fn planner(expression: crate::expressions::CompiledExpression) -> Self { Self::Planner(Box::new(expression)) } @@ -70,7 +65,8 @@ impl Expression { left, right, } => { - use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + use crate::expressions::binary::BinaryOpKind; + use planner_types::pre_asap::CompareOpKind; let (a, n) = left.dtype(input)?; let (b, m) = right.dtype(input)?; if a != DataType::Float64 || b != a || operator.vector_match.is_some() { @@ -105,10 +101,6 @@ impl Expression { }; Ok((dtype, n || m)) } - UnifiedPlanner(expression) => { - expression.validate_input(input)?; - Ok(expression.dtype()) - } Planner(expression) => { expression.validate_input(input)?; Ok(expression.dtype()) @@ -296,7 +288,6 @@ impl Expression { } } Planner(expression) => expression.evaluate(row)?, - UnifiedPlanner(expression) => expression.evaluate(row)?, Label { column, name } => { let Value::Map(entries) = &row[*column] else { return Err(invalid("label read requires a map")); diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs index 2044d6e16..99f4c2b89 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -3,20 +3,22 @@ use crate::{ values::{SchemaRef, Value}, Error, }; -use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, ScalarValue}; + +use planner_types::ir::ScalarExpr; use std::{cmp::Ordering, sync::Arc}; pub(super) fn evaluate( - expr: &QueryExpr, + expr: &ScalarExpr, row: &[Value], schema: &planner_types::pre_asap::Schema, ) -> Result { match expr { - QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + ScalarExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( "column {index} outside row width {}", row.len() ))), - QueryExpr::Literal(value) => Ok(match value { + ScalarExpr::Literal(value) => Ok(match value { ScalarValue::Interval { months, days, @@ -32,18 +34,62 @@ pub(super) fn evaluate( ScalarValue::Boolean(value) => Value::Bool(*value), ScalarValue::Null => Value::Null, }), - QueryExpr::Compare { left, op, right } => { + ScalarExpr::Cast { expr, to, .. } => { + let value = evaluate(expr, row, schema)?; + match (value, to) { + (Value::Null, _) => Ok(Value::Null), + (Value::Int64(value), DataType::Float64) => Ok(Value::Float64(value as f64)), + (value, _) + if expr + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0 + == *to => + { + Ok(value) + } + _ => Err(Error::Invalid("unsupported cast".into())), + } + } + ScalarExpr::Negative { expr, .. } => match evaluate(expr, row, schema)? { + Value::Float64(v) => Ok(Value::Float64(-v)), + Value::Int64(v) => v + .checked_neg() + .map(Value::Int64) + .ok_or_else(|| Error::Invalid("integer negation overflow".into())), + Value::Null => Ok(Value::Null), + _ => Err(Error::Invalid("invalid negation input".into())), + }, + ScalarExpr::Compare { + left, op, right, .. + } => { let left = evaluate(left, row, schema)?; let right = evaluate(right, row, schema)?; compare(op, left, right) } - QueryExpr::Arithmetic { op, left, right } => arithmetic( + ScalarExpr::Arithmetic { + op, left, right, .. + } => arithmetic( op, evaluate(left, row, schema)?, evaluate(right, row, schema)?, ), - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { - let and = matches!(expr, QueryExpr::BoolAnd(_)); + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } => { + for (condition, value) in branches { + if matches!(evaluate(condition, row, schema)?, Value::Bool(true)) { + return evaluate(value, row, schema); + } + } + else_expr + .as_ref() + .map_or(Ok(Value::Null), |e| evaluate(e, row, schema)) + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { + let and = matches!(expr, ScalarExpr::BoolAnd(_)); let mut null = false; for part in parts { match evaluate(part, row, schema)? { @@ -55,21 +101,44 @@ pub(super) fn evaluate( } Ok(if null { Value::Null } else { Value::Bool(and) }) } - QueryExpr::Not(value) => match evaluate(value, row, schema)? { + ScalarExpr::Not(value) => match evaluate(value, row, schema)? { Value::Bool(value) => Ok(Value::Bool(!value)), Value::Null => Ok(Value::Null), _ => Err(Error::Invalid("boolean predicate required".into())), }, - QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + ScalarExpr::IsNull(value) => Ok(Value::Bool(matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + ScalarExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::FunctionCall { name, args } => { + ScalarExpr::FunctionCall { name, args } => { use planner_types::pre_asap::scalar_type_rules::MapScalarFunction; + if planner_types::pre_asap::scalar_type_rules::promql_function_arity(name).is_some() { + let values = args + .iter() + .map(|arg| match evaluate(arg, row, schema)? { + Value::Float64(v) => Ok(v), + _ => Err(Error::Invalid("PromQL function requires floats".into())), + }) + .collect::, _>>()?; + return Ok(Value::Float64(promql_function(name, &values)?)); + } + if name == "promql_drop_metric_name" { + let Value::Utf8(encoded) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("series identity must be Utf8".into())); + }; + let mut labels: std::collections::BTreeMap = + serde_json::from_str(&encoded).map_err(|e| Error::Invalid(e.to_string()))?; + labels.remove("__name__"); + return Ok(Value::Utf8( + serde_json::to_string(&labels) + .map_err(|e| Error::Invalid(e.to_string()))? + .into(), + )); + } if name.eq_ignore_ascii_case("asap_struct_field") { expr.scalar_type(schema) .map_err(|error| Error::Invalid(error.to_string()))?; @@ -81,10 +150,10 @@ pub(super) fn evaluate( unreachable!() }; let offset = match &args[1] { - QueryExpr::Literal(ScalarValue::Int64(index)) => { + ScalarExpr::Literal(ScalarValue::Int64(index)) => { usize::try_from(index - 1).ok() } - QueryExpr::Literal(ScalarValue::Utf8(name)) => { + ScalarExpr::Literal(ScalarValue::Utf8(name)) => { fields.iter().position(|field| &field.name == name) } _ => None, @@ -329,31 +398,120 @@ fn cell_cmp(left: &Value, right: &Value) -> Option { } } +fn promql_function(name: &str, args: &[f64]) -> Result { + let x = args[0]; + Ok(match &name[7..] { + "abs" => x.abs(), + "ceil" => x.ceil(), + "floor" => x.floor(), + "exp" => x.exp(), + "ln" => x.ln(), + "log2" => x.log2(), + "log10" => x.log10(), + "sqrt" => x.sqrt(), + "sgn" => { + if x.is_nan() { + f64::NAN + } else if x == 0.0 { + 0.0 + } else { + x.signum() + } + } + "sin" => x.sin(), + "cos" => x.cos(), + "tan" => x.tan(), + "asin" => x.asin(), + "acos" => x.acos(), + "atan" => x.atan(), + "sinh" => x.sinh(), + "cosh" => x.cosh(), + "tanh" => x.tanh(), + "asinh" => x.asinh(), + "acosh" => x.acosh(), + "atanh" => x.atanh(), + "deg" => x.to_degrees(), + "rad" => x.to_radians(), + "round" => { + let inverse = 1.0 / args[1]; + (x * inverse + 0.5).floor() / inverse + } + "clamp_min" => { + if x.is_nan() || args[1].is_nan() { + f64::NAN + } else { + x.max(args[1]) + } + } + "clamp_max" => { + if x.is_nan() || args[1].is_nan() { + f64::NAN + } else { + x.min(args[1]) + } + } + "clamp" => { + if args.iter().any(|x| x.is_nan()) { + f64::NAN + } else { + x.max(args[1]).min(args[2]) + } + } + part => { + use chrono::{Datelike, Timelike}; + if !x.is_finite() || x < i64::MIN as f64 || x >= i64::MAX as f64 { + return Ok(f64::NAN); + } + let Some(date) = chrono::DateTime::from_timestamp(x as i64, 0) else { + return Ok(f64::NAN); + }; + match part { + "minute" => date.minute() as f64, + "hour" => date.hour() as f64, + "day_of_week" => date.weekday().num_days_from_sunday() as f64, + "day_of_month" => date.day() as f64, + "day_of_year" => date.ordinal() as f64, + "month" => date.month() as f64, + "year" => date.year() as f64, + "days_in_month" => { + let year = date.year(); + let leap = year % 4 == 0 && (year % 100 != 0 || year % 400 == 0); + match date.month() { + 2 => { + if leap { + 29.0 + } else { + 28.0 + } + } + 4 | 6 | 9 | 11 => 30.0, + _ => 31.0, + } + } + _ => return Err(Error::Invalid("unregistered PromQL function".into())), + } + } + }) +} + #[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub struct CompiledExpression { - expression: QueryExpr, + expression: ScalarExpr, schema: planner_types::pre_asap::Schema, output: (DataType, bool), } impl CompiledExpression { - pub fn compile(expression: &QueryExpr, input: &SchemaRef) -> Result { - let schema = input - .fields - .iter() - .map(|field| { - let planner_types::post_asap::FieldDataType::Plain(dtype) = &field.dtype else { - return Err(Error::Invalid( - "scalar expression cannot consume opaque summary state".into(), - )); - }; - Ok(planner_types::pre_asap::Field::plain( - field.name.clone(), - dtype.clone(), - field.nullable, - )) - }) - .collect::, Error>>()?; - let schema = planner_types::pre_asap::Schema::new(schema); + pub(crate) fn expression(&self) -> &ScalarExpr { + &self.expression + } + + pub fn compile(expression: &ScalarExpr, input: &SchemaRef) -> Result { + if !input.is_all_plain() { + return Err(Error::Invalid( + "scalar expression cannot consume summary state".into(), + )); + } + let schema = input.as_ref().clone(); validate(expression, &schema)?; let output = expression .scalar_type(&schema) @@ -394,8 +552,7 @@ impl CompiledExpression { if row.len() != self.schema.fields.len() || row.iter().zip(&self.schema.fields).any(|(value, column)| { !column - .dtype - .plain() + .plain_dtype() .is_some_and(|dtype| value.matches(dtype, column.nullable)) }) { @@ -406,13 +563,27 @@ impl CompiledExpression { evaluate(&self.expression, row, &self.schema) } } -fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { +fn validate(expr: &ScalarExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); expr.scalar_type(schema) .map_err(|e| Error::Invalid(e.to_string()))?; match expr { - QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), - QueryExpr::Arithmetic { left, right, .. } => { + ScalarExpr::Column(_) | ScalarExpr::Literal(_) => Ok(()), + ScalarExpr::Cast { expr, to, .. } => { + let source = expr + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0; + if source != *to + && source != DataType::Null + && !(source == DataType::Int64 && *to == DataType::Float64) + { + return Err(invalid()); + } + validate(expr, schema) + } + ScalarExpr::Negative { expr, .. } => validate(expr, schema), + ScalarExpr::Arithmetic { left, right, .. } => { for value in [left, right] { validate(value, schema)?; if !matches!( @@ -427,7 +598,9 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Compare { left, right, op } => { + ScalarExpr::Compare { + left, right, op, .. + } => { if !matches!( op, CompareOpKind::Eq @@ -471,8 +644,10 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::FunctionCall { name, args } => { - if name != "asap_struct_field" + ScalarExpr::FunctionCall { name, args } => { + if name != "promql_drop_metric_name" + && planner_types::pre_asap::scalar_type_rules::promql_function_arity(name).is_none() + && name != "asap_struct_field" && name != "asap_element_access" && planner_types::pre_asap::scalar_type_rules::MapScalarFunction::from_name(name) .is_none() @@ -484,7 +659,29 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } => { + for (condition, value) in branches { + validate(condition, schema)?; + if condition + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0 + != DataType::Bool + { + return Err(invalid()); + } + validate(value, schema)?; + } + if let Some(value) = else_expr { + validate(value, schema)?; + } + Ok(()) + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { for part in parts { validate(part, schema)?; if !matches!( @@ -498,7 +695,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Not(value) => { + ScalarExpr::Not(value) => { validate(value, schema)?; if !matches!( value @@ -511,7 +708,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), + ScalarExpr::IsNull(value) | ScalarExpr::IsNotNull(value) => validate(value, schema), _ => Err(invalid()), } } diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs index 586da2508..ea6b73eaf 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/asap-physical-operators/src/lib.rs @@ -20,7 +20,7 @@ pub use planner_types as planner; pub mod dag; -pub mod readout; +pub mod evaluation; mod error; pub use error::Error; @@ -31,6 +31,3 @@ pub mod plan; pub mod runtime; pub mod sources; pub mod values; - -pub mod unified_physical_planner; -pub mod unified_sources; diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs index 76eccd5ee..7b51b3c66 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -24,7 +24,8 @@ impl Operator { } else { t.clone() }, - false, + !input.has_promql_series_identity() + && (groups.is_empty() || plain(&input, *i)?.1), ) } Reduction::Quantile { column, q } => { @@ -183,7 +184,8 @@ async fn reduce( let mut work = Cooperative::new(context); let mut workspace = Workspace::new(context)?; let mut grouped = BTreeMap::>, Vec>>::new(); - if rows.is_empty() && groups.is_empty() { + // PromQL sums over an empty vector emit no sample. + if rows.is_empty() && groups.is_empty() && !input.has_promql_series_identity() { grouped.insert(vec![], vec![]); } for row in rows { @@ -318,6 +320,9 @@ async fn reduce_one( .ok_or_else(|| invalid("integer aggregate overflow"))?; count += 1; } + if count == 0 && !input.has_promql_series_identity() { + return Ok(Value::Null); + } return if matches!(measure, Reduction::Avg(_)) { Ok(Value::Float64(sum as f64 / count as f64)) } else { @@ -334,6 +339,9 @@ async fn reduce_one( }; floats.push(*v); } + if floats.is_empty() && !input.has_promql_series_identity() { + return Ok(Value::Null); + } Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { promql_avg(&floats) } else { diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs index e40a1fa56..a289c5284 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -308,7 +308,7 @@ mod tests { values::Batch, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, + post_asap::{Field as SummaryField, FieldDataType as SummaryFamilyType, Schema}, pre_asap::DataType, types::AccuracyTarget, }; @@ -318,23 +318,23 @@ mod tests { #[test] fn temporal_windows_execute_in_both_phases_and_count_is_integer() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![ - Field { - table: None, + SummaryField { name: "time".into(), - dtype: FieldDataType::Plain(DataType::Timestamp), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), nullable: false, - }, - Field { table: None, + }, + SummaryField { name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), + dtype: SummaryFamilyType::Plain(DataType::Float64), nullable: false, + table: None, }, ], time_index: Some(0), + unique_keys: vec![], + closed: false, }); for scope in [ Scope::Query { diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs index 188d28d6f..909dcc369 100644 --- a/crates/asap-physical-operators/src/operators/aligned_binary.rs +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -1,6 +1,8 @@ //! Arithmetic on complete, aligned population/window rows used by precomputation. use super::*; -use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; + use std::collections::BTreeSet; impl Operator { @@ -22,11 +24,9 @@ impl Operator { )); } for (input, value) in [(&left, values.0), (&right, values.1)] { - if input - .fields - .get(value) - .is_none_or(|f| f.nullable || f.dtype != FieldDataType::Plain(DataType::Float64)) - { + if input.fields.get(value).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Float64) + }) { return Err(invalid( "aligned arithmetic requires non-null Float64 values", )); diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs index 5ef5023cd..f67314bb1 100644 --- a/crates/asap-physical-operators/src/operators/common.rs +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -2,20 +2,20 @@ use super::*; pub(super) fn invalid(message: &str) -> Error { Error::Invalid(message.into()) } -pub(super) fn schema(fields: Vec) -> SchemaRef { +pub(super) fn schema(fields: Vec) -> SchemaRef { Arc::new(Schema { - closed: true, - unique_keys: vec![], fields, + unique_keys: vec![], + closed: false, time_index: None, }) } -pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> Field { - Field { - table: None, +pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { + SummaryField { name: name.into(), - dtype: FieldDataType::Plain(dtype), + dtype: SummaryFamilyType::Plain(dtype), nullable, + table: None, } } @@ -76,4 +76,3 @@ pub(super) fn key_bytes(key: &[Vec]) -> usize { .map(|part| std::mem::size_of::>() + part.len()) .sum::() } -use planner_types::pre_asap::Schema; diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs index 9dfd20dea..652fbcdcc 100644 --- a/crates/asap-physical-operators/src/operators/joins/mod.rs +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -22,6 +22,15 @@ impl Operator { output: left, }) } + /// Require every candidate key to have an authoritative value at execution. + pub fn certified_semi_join( + left: SchemaRef, + right: SchemaRef, + keys: Vec<(usize, usize)>, + ) -> Result { + Ok(Self::semi_join(left, right, keys)?.require_complete_right()) + } + pub(crate) fn require_complete_right(mut self) -> Self { if let Kind::SemiJoin { require_complete_right, @@ -42,48 +51,18 @@ impl Operator { } } pub fn relational_join( - left: SchemaRef, - right: SchemaRef, - kind: planner_types::pre_asap::JoinKind, - predicate: &planner_types::pre_asap::Predicate, - output: SchemaRef, - ) -> Result { - let mut joined = left.fields.clone(); - joined.extend(right.fields.clone()); - let predicate = Expression::planner(crate::expressions::CompiledExpression::compile( - &predicate.0, - &schema(joined), - )?); - Self::bound_relational_join(left, right, kind, predicate, output) - } - pub fn unified_relational_join( left: SchemaRef, right: SchemaRef, kind: planner_types::pre_asap::JoinKind, predicate: &planner_types::ir::Predicate, output: SchemaRef, - ) -> Result { - let mut joined = left.fields.clone(); - joined.extend(right.fields.clone()); - let predicate = Expression::unified_planner( - crate::expressions::unified_planner::CompiledExpression::compile( - &predicate.0, - &schema(joined), - )?, - ); - Self::bound_relational_join(left, right, kind, predicate, output) - } - pub(crate) fn bound_relational_join( - left: SchemaRef, - right: SchemaRef, - kind: planner_types::pre_asap::JoinKind, - predicate: Expression, - output: SchemaRef, ) -> Result { use planner_types::pre_asap::JoinKind; let mut joined = left.fields.clone(); joined.extend(right.fields.clone()); - if predicate.dtype(&schema(joined.clone()))?.0 != DataType::Bool { + let predicate = + crate::expressions::CompiledExpression::compile(&predicate.0, &schema(joined.clone()))?; + if predicate.dtype().0 != DataType::Bool { return Err(invalid("join predicate must be boolean")); } let fields = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs index 75b313d75..517019118 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -8,7 +8,7 @@ use crate::{ }; use futures::StreamExt; use planner_types::{ - post_asap::{Field, FieldDataType, SummaryUpdate}, + post_asap::{Field as SummaryField, FieldDataType as SummaryFamilyType, Schema, SummaryUpdate}, pre_asap::{ColumnRef, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; @@ -34,7 +34,7 @@ pub(crate) mod vector_window; pub use aggregate::Reduction; pub use series_window::SubquerySteps; pub use sort::SortKey; -pub use summary::ReadoutQuery; +pub use summary::SummaryEvaluation; #[derive(Clone, serde::Serialize, serde::Deserialize)] enum Kind { #[serde(skip)] @@ -58,13 +58,13 @@ enum Kind { column: usize, }, VectorBinary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, return_bool: bool, }, AlignedBinary { keys: Vec<(usize, usize)>, values: (usize, usize), - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, }, RangeWindow { intent: Box>, @@ -89,7 +89,7 @@ enum Kind { unique: bool, }, SeriesBinary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, scalars: [bool; 2], }, SeriesRelabel { @@ -130,21 +130,21 @@ enum Kind { }, Join { kind: planner_types::pre_asap::JoinKind, - predicate: Box, + predicate: Box, }, SummaryBuild { - family: FieldDataType, + family: SummaryFamilyType, value: usize, time: Option, groups: Vec, }, KeyedSummaryBuild { - family: FieldDataType, + family: SummaryFamilyType, value: usize, items: Vec, groups: Vec, }, - KeyedReadout { + KeyedEvaluation { state: usize, k: usize, }, @@ -152,9 +152,9 @@ enum Kind { state: usize, groups: Vec, }, - Readout { + Evaluation { state: usize, - query: ReadoutQuery, + query: SummaryEvaluation, }, } /// A bound operation has a fully checked input/output contract before execution. @@ -175,11 +175,11 @@ impl Operator { } } - pub(crate) fn is_counter_readout(&self) -> bool { + pub(crate) fn is_counter_evaluation(&self) -> bool { matches!( self.kind, - Kind::Readout { - query: ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + Kind::Evaluation { + query: SummaryEvaluation::Exact(crate::summary_kernels::exact::ExactEvaluation { statistic: crate::Statistic::Rate | crate::Statistic::Increase, .. }), @@ -192,21 +192,24 @@ impl Operator { if lookback <= 0 { return Err(invalid("counter lookback must be positive")); } - if let Kind::Readout { - query: ReadoutQuery::Exact(readout), + if let Kind::Evaluation { + query: SummaryEvaluation::Exact(evaluation), .. } = &mut self.kind { - readout.lookback_ms = Some(lookback); + evaluation.lookback_ms = Some(lookback); } Ok(self) } - /// Resolve a counter readout's logical lookback to this run's evaluation range. - pub(super) fn readout_range(&self, context: &RunContext) -> Result, Error> { - let Kind::Readout { + /// Resolve a counter evaluation's logical lookback to this run's evaluation range. + pub(super) fn evaluation_range( + &self, + context: &RunContext, + ) -> Result, Error> { + let Kind::Evaluation { query: - ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + SummaryEvaluation::Exact(crate::summary_kernels::exact::ExactEvaluation { lookback_ms: Some(lookback), .. }), @@ -252,7 +255,7 @@ impl Operator { } if output.time_index.is_some_and(|i| { i >= output.fields.len() - || output.fields[i].dtype != FieldDataType::Plain(DataType::Timestamp) + || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) }) { return Err(invalid("invalid output time column")); } @@ -335,15 +338,15 @@ impl PhysicalOperator for Operator { Kind::SemiJoin { .. } => "SemiJoin", Kind::Join { .. } => "RelationalJoin", Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", - Kind::KeyedReadout { .. } => "SummaryEstimate", + Kind::KeyedEvaluation { .. } => "SummaryEstimate", Kind::SummaryMerge { .. } => "SummaryMerge", - Kind::Readout { .. } => "SummaryReadout", + Kind::Evaluation { .. } => "SummaryEvaluation", } } fn validate_context(&self, context: &RunContext) -> Result<(), Error> { current_series::validate_context(self, context)?; series_window::validate_context(self, context)?; - self.readout_range(context).map(|_| ()) + self.evaluation_range(context).map(|_| ()) } fn input_schemas(&self) -> Vec { self.inputs.clone() @@ -387,9 +390,9 @@ impl PhysicalOperator for Operator { Kind::Join { .. } | Kind::SemiJoin { .. } => joins::execute(self, inputs, context), Kind::SummaryMerge { .. } => summary::execute_merge(self, inputs, context), Kind::SummaryBuild { .. } - | Kind::Readout { .. } + | Kind::Evaluation { .. } | Kind::KeyedSummaryBuild { .. } - | Kind::KeyedReadout { .. } => summary::execute(self, inputs, context), + | Kind::KeyedEvaluation { .. } => summary::execute(self, inputs, context), } } } diff --git a/crates/asap-physical-operators/src/operators/scope_timestamp.rs b/crates/asap-physical-operators/src/operators/scope_timestamp.rs index bb7a4e855..33fbdd1ec 100644 --- a/crates/asap-physical-operators/src/operators/scope_timestamp.rs +++ b/crates/asap-physical-operators/src/operators/scope_timestamp.rs @@ -26,7 +26,7 @@ impl Operator { candidate.dtype == field.dtype && candidate.nullable == field.nullable && (candidate.name == field.name - || !matches!(field.dtype, FieldDataType::Plain(_))) + || !matches!(field.dtype, SummaryFamilyType::Plain(_))) }) .map(|(index, _)| index) .collect(); diff --git a/crates/asap-physical-operators/src/operators/series_labels.rs b/crates/asap-physical-operators/src/operators/series_labels.rs index 2a8c5d1ee..3987c5142 100644 --- a/crates/asap-physical-operators/src/operators/series_labels.rs +++ b/crates/asap-physical-operators/src/operators/series_labels.rs @@ -1,10 +1,9 @@ //! PromQL label-set rewriting and binary operators over rows that carry a //! series identity or plain label columns. use super::*; -use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{schema::PROMQL_SERIES_IDENTITY, BinaryOpKind, VectorMatchKind}, -}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; +use planner_types::pre_asap::{schema::PROMQL_SERIES_IDENTITY, VectorMatchKind}; type Labels = BTreeMap; diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs index 90cc2c724..5dab8109f 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -1,22 +1,22 @@ use super::*; -/// A summary readout: a sketch query, or an exact readout with typed parameters. +/// A summary evaluation: a sketch query, or an exact evaluation with typed parameters. #[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] -pub enum ReadoutQuery { +pub enum SummaryEvaluation { Sketch(planner_types::post_asap::SketchStatistic), - Exact(crate::summary_kernels::exact::ExactReadout), + Exact(crate::summary_kernels::exact::ExactEvaluation), } impl Operator { pub fn keyed_summary_build( input: SchemaRef, - family: FieldDataType, + family: SummaryFamilyType, value: usize, items: Vec, groups: Vec, ) -> Result { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&family)?; - let FieldDataType::Sketch(kind, _) = &family else { + let SummaryFamilyType::Sketch(kind, _) = &family else { return Err(invalid("keyed sketch required")); }; WeightedFrequency::configuration(kind)?; @@ -43,11 +43,11 @@ impl Operator { .iter() .map(|&i| input.fields[i].clone()) .collect::>(); - fields.push(Field { - table: None, + fields.push(SummaryField { name: "state".into(), dtype: family.clone(), nullable: false, + table: None, }); Ok(Self { kind: Kind::KeyedSummaryBuild { @@ -60,7 +60,7 @@ impl Operator { output: schema(fields), }) } - pub fn keyed_readout( + pub fn keyed_evaluation( input: SchemaRef, state: usize, k: usize, @@ -68,31 +68,31 @@ impl Operator { ) -> Result { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&field(&input, state)?.dtype)?; - let FieldDataType::Sketch(kind, _) = &field(&input, state)?.dtype else { - return Err(invalid("keyed readout requires summary state")); + let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + return Err(invalid("keyed evaluation requires summary state")); }; let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; if k > capacity || output.fields.len() <= input.fields.len() { - return Err(invalid("invalid keyed readout shape or capacity")); + return Err(invalid("invalid keyed evaluation shape or capacity")); } if state + 1 != input.fields.len() || output.fields[..state] != input.fields[..state] - || output.fields.last().unwrap().dtype != FieldDataType::Plain(DataType::Float64) + || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) { return Err(invalid( - "keyed readout must preserve partitions and return a Float64 score", + "keyed evaluation must preserve partitions and return a Float64 score", )); } crate::values::validate_schema(&output)?; Ok(Self { - kind: Kind::KeyedReadout { state, k }, + kind: Kind::KeyedEvaluation { state, k }, inputs: vec![input], output, }) } pub fn summary_build( input: SchemaRef, - family: FieldDataType, + family: SummaryFamilyType, value: usize, time: Option, groups: Vec, @@ -110,7 +110,7 @@ impl Operator { if time.is_none() && matches!( family, - FieldDataType::ExactAggregate( + SummaryFamilyType::ExactAggregate( planner_types::post_asap::ExactKind::Rate | planner_types::post_asap::ExactKind::Increase, _ @@ -129,11 +129,11 @@ impl Operator { .iter() .map(|&i| input.fields[i].clone()) .collect::>(); - fields.push(Field { - table: None, + fields.push(SummaryField { name: "state".into(), dtype: family.clone(), nullable: false, + table: None, }); Ok(Self { kind: Kind::SummaryBuild { @@ -153,7 +153,7 @@ impl Operator { ) -> Result { validate_groups(&input, &groups)?; crate::values::validate_family(&field(&input, state)?.dtype)?; - if matches!(field(&input, state)?.dtype, FieldDataType::Plain(_)) { + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { return Err(invalid("summary state required")); } let mut fields = groups @@ -167,21 +167,25 @@ impl Operator { output: schema(fields), }) } - pub fn readout(input: SchemaRef, state: usize, query: ReadoutQuery) -> Result { + pub fn evaluation( + input: SchemaRef, + state: usize, + query: SummaryEvaluation, + ) -> Result { let family = &field(&input, state)?.dtype; crate::values::validate_family(family)?; match &query { - ReadoutQuery::Sketch(query) => { - crate::capability::validate_sketch_readout(family, query)? + SummaryEvaluation::Sketch(query) => { + crate::capability::validate_sketch_evaluation(family, query)? } - ReadoutQuery::Exact(readout) => { - crate::capability::validate_exact_readout(family, readout)? + SummaryEvaluation::Exact(evaluation) => { + crate::capability::validate_exact_evaluation(family, evaluation)? } } let mut fields = input.fields.clone(); let result_type = if matches!( fields[state].dtype, - FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) ) || integral_count(family, &query) { DataType::Int64 @@ -193,7 +197,7 @@ impl Operator { let nullable = fields.len() == 1 && matches!( fields[state].dtype, - FieldDataType::ExactAggregate( + SummaryFamilyType::ExactAggregate( planner_types::post_asap::ExactKind::Min | planner_types::post_asap::ExactKind::Max, _ @@ -201,7 +205,7 @@ impl Operator { ); fields[state] = result_field("value", result_type, nullable); Ok(Self { - kind: Kind::Readout { state, query }, + kind: Kind::Evaluation { state, query }, inputs: vec![input], output: schema(fields), }) @@ -210,12 +214,12 @@ impl Operator { /// The Planner reads a Count-Min bare count only for count intents, whose /// output is Int64 and whose updates have unit weight; execution rejects a /// non-integral total rather than rounding it. -fn integral_count(family: &FieldDataType, query: &ReadoutQuery) -> bool { - matches!(family, FieldDataType::Sketch(kind, _) +fn integral_count(family: &SummaryFamilyType, query: &SummaryEvaluation) -> bool { + matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::Cms) && matches!( query, - ReadoutQuery::Sketch(planner_types::post_asap::SketchStatistic::PointCount { + SummaryEvaluation::Sketch(planner_types::post_asap::SketchStatistic::PointCount { value: None, .. }) @@ -226,7 +230,7 @@ pub(super) fn execute<'a>( mut inputs: Vec>, context: RunContext, ) -> Result, Error> { - let range_ms = operator.readout_range(&context)?; + let range_ms = operator.evaluation_range(&context)?; let output = operator.output.clone(); let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; match &operator.kind { @@ -238,7 +242,7 @@ pub(super) fn execute<'a>( } => Ok(futures::stream::once(async move { Batch::try_new( output, - build_summary(input, family, *value, *time, groups, &context).await?, + build_summary(input, family, *value, *time, groups, !operator.inputs[0].has_promql_series_identity(), &context).await?, ) }) .boxed_local()), @@ -254,7 +258,7 @@ pub(super) fn execute<'a>( ) }) .boxed_local()), - Kind::KeyedReadout { state, k } => Ok(input + Kind::KeyedEvaluation { state, k } => Ok(input .map(move |batch| { let batch = batch?; let mut rows = Vec::new(); @@ -272,7 +276,7 @@ pub(super) fn execute<'a>( // The typed output schema restores epoch-millisecond // timestamp keys from the kernel's Int64 representation. for (value, field) in values.iter_mut().zip(&output.fields) { - if field.dtype == FieldDataType::Plain(DataType::Timestamp) { + if field.dtype == SummaryFamilyType::Plain(DataType::Timestamp) { if let Value::Int64(time) = value { *value = Value::Timestamp(*time); } @@ -284,25 +288,25 @@ pub(super) fn execute<'a>( Batch::try_new(output.clone(), rows) }) .boxed_local()), - Kind::Readout { state, query } => Ok(input + Kind::Evaluation { state, query } => Ok(input .map(move |batch| { let batch = batch?; let mut rows = batch.rows().to_vec(); - if let ReadoutQuery::Exact(readout) = query { + if let SummaryEvaluation::Exact(evaluation) = query { rows.retain(|row| !matches!(&row[*state], Value::Summary { state: summary, .. } - if crate::readout::insufficient_counter_samples(summary.as_ref(), readout.statistic))); + if crate::evaluation::insufficient_counter_samples(summary.as_ref(), evaluation.statistic))); } for row in &mut rows { let Value::Summary { state: summary, .. } = &row[*state] else { return Err(invalid("summary value required")); }; row[*state] = match query { - ReadoutQuery::Sketch(query) => { + SummaryEvaluation::Sketch(query) => { let value = summary .estimate(query) .map_err(|e| Error::Operator(e.to_string()))?; if output.fields[*state].dtype - == FieldDataType::Plain(DataType::Int64) + == SummaryFamilyType::Plain(DataType::Int64) { // Below 2^53 an f64 sum of unit updates is exact. if value.fract() != 0.0 || !(0.0..9.007_199_254_740_992e15).contains(&value) { @@ -315,12 +319,13 @@ pub(super) fn execute<'a>( Value::Float64(value) } } - ReadoutQuery::Exact(readout) => { + SummaryEvaluation::Exact(evaluation) => { let exact = summary .as_any() .downcast_ref::() - .ok_or_else(|| invalid("exact readout requires exact state"))?; - if output.fields[*state].dtype == FieldDataType::Plain(DataType::Int64) { + .ok_or_else(|| invalid("exact evaluation requires exact state"))?; + if output.fields[*state].nullable && exact.is_empty_sum() { Value::Null } + else if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { let count = exact.count().ok_or_else(|| { Error::Operator("exact count state lacks an integer count".into()) })?; @@ -329,7 +334,7 @@ pub(super) fn execute<'a>( })?) } else { match exact - .readout(readout.statistic, range_ms, None) + .evaluation(evaluation.statistic, range_ms, None) .map_err(|e| Error::Operator(e.to_string()))? { Some(value) => Value::Float64(value), @@ -372,10 +377,11 @@ pub(super) fn execute_merge<'a>( async fn build_summary( mut input: Input<'_, Batch>, - family: &FieldDataType, + family: &SummaryFamilyType, value: usize, time: Option, groups: &[usize], + emit_empty_global: bool, context: &RunContext, ) -> Result>, Error> { type State = ( @@ -398,12 +404,13 @@ async fn build_summary( }; let mut work = Cooperative::new(context); let mut states = BTreeMap::>, State>::new(); - if groups.is_empty() { + // PromQL aggregation of an empty vector produces no sample. + if groups.is_empty() && emit_empty_global { states.insert(vec![], create(vec![], 0)?); } let ordered_time = matches!( family, - FieldDataType::ExactAggregate( + SummaryFamilyType::ExactAggregate( planner_types::post_asap::ExactKind::Rate | planner_types::post_asap::ExactKind::Increase, _ @@ -472,7 +479,7 @@ async fn merge_summary( groups: &[usize], context: &RunContext, ) -> Result>, Error> { - type GroupState = (Vec, FieldDataType, Arc); + type GroupState = (Vec, SummaryFamilyType, Arc); let mut states: BTreeMap>, GroupState> = BTreeMap::new(); let mut work = Cooperative::new(context); let mut memory = context.reserve(0)?; @@ -531,14 +538,14 @@ async fn merge_summary( async fn build_keyed_summary( mut input: Input<'_, Batch>, - family: &FieldDataType, + family: &SummaryFamilyType, value: usize, items: &[usize], groups: &[usize], context: &RunContext, ) -> Result>, Error> { use crate::{summary_kernels::weighted_frequency::WeightedFrequency, AggregateCore}; - let FieldDataType::Sketch(kind, _) = family else { + let SummaryFamilyType::Sketch(kind, _) = family else { unreachable!() }; let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs index a7d3bd683..1b5dd7c4f 100644 --- a/crates/asap-physical-operators/src/operators/unchecked.rs +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -134,11 +134,11 @@ impl TryFrom for Operator { operator } } - Kind::Join { kind, predicate } => Operator::bound_relational_join( + Kind::Join { kind, predicate } => Operator::relational_join( input(0)?, input(1)?, kind, - *predicate, + &planner_types::ir::Predicate(predicate.expression().clone()), output.clone(), )?, Kind::SummaryBuild { @@ -153,13 +153,13 @@ impl TryFrom for Operator { items, groups, } => Operator::keyed_summary_build(input(0)?, family, value, items, groups)?, - Kind::KeyedReadout { state, k } => { - Operator::keyed_readout(input(0)?, state, k, output.clone())? + Kind::KeyedEvaluation { state, k } => { + Operator::keyed_evaluation(input(0)?, state, k, output.clone())? } Kind::SummaryMerge { state, groups } => { Operator::summary_merge(input(0)?, state, groups)? } - Kind::Readout { state, query } => Operator::readout(input(0)?, state, query)?, + Kind::Evaluation { state, query } => Operator::evaluation(input(0)?, state, query)?, } .with_output_schema(output)?; if serde_json::to_value(&op.kind).map_err(|error| invalid(&error.to_string()))? diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs index 7027fb9be..0f4af7cf4 100644 --- a/crates/asap-physical-operators/src/operators/vector_binary.rs +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -1,6 +1,7 @@ //! Label matching and scalar broadcasting are physical computation, not source binding. use super::*; -use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; pub(crate) fn value_schema(scalar: bool) -> SchemaRef { let mut fields = Vec::new(); diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs index 6b4d58347..6b21ff6a4 100644 --- a/crates/asap-physical-operators/src/operators/vector_window.rs +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -1,7 +1,6 @@ //! Window bounds are typed input data; aggregation and histogram semantics stay native. use super::*; use planner_types::pre_asap::AggIntent; -use planner_types::pre_asap::Schema; pub(crate) fn matrix_schema() -> SchemaRef { let mut fields = vector_binary::value_schema(false).fields.clone(); @@ -9,9 +8,9 @@ pub(crate) fn matrix_schema() -> SchemaRef { fields.push(result_field("window_start", DataType::Timestamp, false)); fields.push(result_field("window_end", DataType::Timestamp, false)); Arc::new(Schema { - closed: true, - unique_keys: vec![], fields, + unique_keys: vec![], + closed: false, time_index: Some(1), }) } diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index 0e6ee86ba..0815514a7 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -5,8 +5,8 @@ use super::*; /// it during optimization and deployment. Stored outputs have no storage identity. /// Deserialization validates the producer/reader boundary. #[derive(Clone, serde::Serialize, serde::Deserialize)] -#[serde(try_from = "UncheckedPhysicalASAPDAG")] -pub struct PhysicalASAPDAG { +#[serde(try_from = "UncheckedCompiledPhysicalPlan")] +pub struct CompiledPhysicalPlan { pub precompute: Option, pub query: CompiledPhysicalDAG, pub materialized_outputs: BTreeMap, @@ -14,18 +14,18 @@ pub struct PhysicalASAPDAG { /// Compile an explicit materialization frontier selected by Planner maintenance /// search. Operators upstream of that frontier run in precompute, including -/// readouts/reductions; query execution receives their typed output values. +/// evaluations/reductions; query execution receives their typed output values. /// Empty frontiers retain the full computation in the query DAG. /// /// Repeated windows must be instantiated with the same evaluation/population /// contract used to build each output. This API never treats a result from a /// different window or revision as interchangeable merely because types match. pub fn compile_candidate( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[NodeId], frontier: &[NodeId], -) -> Result { +) -> Result { cut_candidate(&compile(dag, inputs, roots)?, frontier) } @@ -36,9 +36,9 @@ pub fn compile_candidate( pub fn cut_candidate( compiled: &CompiledPhysicalDAG, frontier: &[NodeId], -) -> Result { +) -> Result { if frontier.is_empty() { - return Ok(PhysicalASAPDAG { + return Ok(CompiledPhysicalPlan { precompute: None, query: compiled.clone(), materialized_outputs: BTreeMap::new(), @@ -79,7 +79,7 @@ pub fn cut_candidate( "frontier contains an output shadowed by another boundary", )); } - Ok(PhysicalASAPDAG { + Ok(CompiledPhysicalPlan { precompute: Some(precompute), query, materialized_outputs, @@ -93,7 +93,7 @@ pub fn cut_candidate( /// That holds while timing-dependent lowering (an ingestion-time `Binary` /// aligns by value column) has the same timing at compile time as here. /// A query-time node feeding an ingestion-time node has no valid placement. -pub fn frontier_from_timing(dag: &PostAsapDAG) -> Result, Error> { +pub fn frontier_from_timing(dag: &PhysicalASAPDAG) -> Result, Error> { use planner_types::post_asap::ExecutionTiming::IngestionTime; let timing = dag .nodes @@ -101,8 +101,10 @@ pub fn frontier_from_timing(dag: &PostAsapDAG) -> Result, Error> { .map(|node| (node.id, node.output_state.timing)) .collect::>(); let mut frontier = BTreeSet::new(); - if timing.get(&dag.root) == Some(&IngestionTime) { - frontier.insert(u64::from(dag.root.0)); + for root in &dag.roots { + if timing.get(root) == Some(&IngestionTime) { + frontier.insert(*root as u64); + } } for edge in &dag.edges { let (Some(&producer), Some(&consumer)) = @@ -112,7 +114,7 @@ pub fn frontier_from_timing(dag: &PostAsapDAG) -> Result, Error> { }; match (producer == IngestionTime, consumer == IngestionTime) { (true, false) => { - frontier.insert(u64::from(edge.producer.0)); + frontier.insert(edge.producer as u64); } (false, true) => return Err(invalid("query-time node feeds an ingestion-time node")), _ => {} @@ -127,7 +129,7 @@ pub fn frontier_from_timing(dag: &PostAsapDAG) -> Result, Error> { /// and deployment feasibility are evaluated separately before cost selection. /// Exceeding the search budget returns an error, never a partial inventory. pub fn enumerate_frontiers( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: &BTreeMap, roots: &[NodeId], max_candidates: usize, @@ -192,11 +194,11 @@ fn enumerate_compiled_frontiers( /// individual failures visible; do not substitute another computation on error. /// The DAG is lowered once; each frontier is a [`cut_candidate`] of it. pub fn compile_candidates( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[NodeId], frontiers: &[Vec], -) -> Vec> { +) -> Vec> { match compile(dag, inputs, roots) { Ok(compiled) => frontiers .iter() @@ -216,7 +218,7 @@ pub struct CandidateCost { pub total_cost: f64, } -pub struct CandidateSelection { +pub struct CandidateSelection { pub candidate: T, pub candidate_index: usize, pub cost: CandidateCost, @@ -271,14 +273,14 @@ pub fn select_candidate( #[derive(serde::Deserialize)] #[serde(deny_unknown_fields)] -struct UncheckedPhysicalASAPDAG { +struct UncheckedCompiledPhysicalPlan { precompute: Option, query: CompiledPhysicalDAG, materialized_outputs: BTreeMap, } -impl TryFrom for PhysicalASAPDAG { +impl TryFrom for CompiledPhysicalPlan { type Error = Error; - fn try_from(candidate: UncheckedPhysicalASAPDAG) -> Result { + fn try_from(candidate: UncheckedCompiledPhysicalPlan) -> Result { let result = Self { precompute: candidate.precompute, query: candidate.query, @@ -289,7 +291,7 @@ impl TryFrom for PhysicalASAPDAG { } } -impl PhysicalASAPDAG { +impl CompiledPhysicalPlan { /// Validate the physical handoff, including the producer/reader boundary. pub fn validate(&self) -> Result<(), Error> { self.query.validate()?; @@ -331,7 +333,7 @@ mod tests { use super::*; use planner_types::workload::*; - fn grouped_rate() -> (PostAsapDAG, BTreeMap, NodeId) { + fn grouped_rate() -> (PhysicalASAPDAG, BTreeMap, NodeId) { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -361,24 +363,30 @@ mod tests { let root = asap_frontend_promql::lower_promql_workload(&workload, 0) .unwrap() .remove(0); - let root = std::rc::Rc::new(promql_rows::with_series_identity(&root).unwrap()); + let root = promql_rows::with_series_identity(&root).unwrap(); let space = asap_aware_mapping::search_workload(vec![("q", root)]); let selected = space .global_selection(&asap_aware_mapping::cost_model::DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = planner_types::post_asap::compile_post_asap_dag(&selected).unwrap(); + let selected = planner_types::ir::apply_lifecycle_timings( + &selected, + &Default::default(), + &mut Default::default(), + ) + .unwrap(); + let dag = planner_types::ir::physical_export::compile_physical_asap_dag(&selected).unwrap(); let state = dag .nodes .iter() - .find(|node| matches!(node.payload, Payload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, Payload::ASAP(ASAPOp::SummaryAgg { .. }))) .unwrap(); let inputs = BTreeMap::from([( - u64::from(state.id.0), + state.id as u64, InputContract::bounded(Arc::new(state.output_schema.clone())), )]); - (dag.clone(), inputs, u64::from(dag.root.0)) + (dag.clone(), inputs, dag.roots[0] as u64) } /// Enumerating and cutting every frontier lowers each Planner node once. @@ -399,9 +407,9 @@ mod tests { } fn with_timing( - dag: &PostAsapDAG, - timing: impl Fn(&PostAsapDAGNode) -> planner_types::post_asap::ExecutionTiming, - ) -> PostAsapDAG { + dag: &PhysicalASAPDAG, + timing: impl Fn(&PhysicalASAPDAGNode) -> planner_types::post_asap::ExecutionTiming, + ) -> PhysicalASAPDAG { let mut timed = dag.clone(); for node in &mut timed.nodes { node.output_state.timing = timing(node); @@ -413,14 +421,19 @@ mod tests { timed } - fn raw_input(dag: &PostAsapDAG) -> BTreeMap { + fn raw_input(dag: &PhysicalASAPDAG) -> BTreeMap { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, Payload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + Payload::NonASAP(planner_types::ir::NonASAPOp::TimeRange { .. }) + ) + }) .unwrap(); BTreeMap::from([( - u64::from(raw.id.0), + raw.id as u64, InputContract::bounded(Arc::new(raw.output_schema.clone())), )]) } @@ -472,7 +485,7 @@ mod tests { use planner_types::post_asap::ExecutionTiming::{IngestionTime, QueryTime}; let (dag, _, _) = grouped_rate(); let timed = with_timing(&dag, |node| { - if node.id == dag.root { + if node.id == dag.roots[0] { IngestionTime } else { QueryTime diff --git a/crates/asap-physical-operators/src/physical_planner/logical.rs b/crates/asap-physical-operators/src/physical_planner/logical.rs new file mode 100644 index 000000000..e503827dd --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/logical.rs @@ -0,0 +1,48 @@ +//! Reconstruct shared operator references from the transport DAG for native lowering. +use super::*; +use planner_types::ir::{ASAPOp, Operator as LogicalOperator, OperatorNode}; +use std::rc::Rc; + +pub(super) fn restore(dag: &PhysicalASAPDAG) -> Result>, Error> { + dag.validate().map_err(|e| invalid(e.to_string()))?; + let mut done: BTreeMap> = BTreeMap::new(); + let mut remaining: Vec<_> = dag.nodes.iter().collect(); + while !remaining.is_empty() { + let before = remaining.len(); + let mut next = Vec::new(); + for node in remaining { + if node + .payload + .children() + .iter() + .any(|child| !done.contains_key(&(**child as u64))) + { + next.push(node); + continue; + } + if matches!( + node.payload, + LogicalOperator::ASAP( + ASAPOp::SummarySubtract { .. } + | ASAPOp::SummaryDelete { .. } + | ASAPOp::SummaryJoin { .. } + | ASAPOp::Extension { .. } + ) + ) { + return Err(invalid("reserved ASAP operation has no native lowering")); + } + let operator = node + .payload + .map_children(|child| Rc::clone(&done[&(*child as u64)])); + let mut rebuilt = OperatorNode::with_schema(operator, node.output_schema.clone()); + rebuilt.guarantee = node.guarantee.clone(); + rebuilt.timing = Some(node.output_state.timing); + done.insert(node.id as u64, Rc::new(rebuilt)); + } + if next.len() == before { + return Err(invalid("operator DAG is cyclic")); + } + remaining = next; + } + Ok(done) +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index e1490c03c..3deb01473 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -1,23 +1,25 @@ //! Compile logical computation to native operators with typed external inputs. //! Compilation needs no readers; deployment resolves inputs after selection. -use crate::operators::ReadoutQuery; -use crate::summary_kernels::exact::ExactReadout; +use crate::operators::SummaryEvaluation; +use crate::summary_kernels::exact::ExactEvaluation; use crate::{ operators::{Expression, Operator, Reduction, SortKey}, plan::{Boundedness, Emission, NodeId, PhysicalDAG, PhysicalOperator, PlanProperties}, values::{Batch, SchemaRef}, Error, }; +use planner_types::ir::physical_export::{ + PhysicalASAPDAG, PhysicalASAPDAGNode, PhysicalASAPNodeId, + PhysicalASAPOperatorPayload as Payload, +}; +use planner_types::ir::{ASAPOp, NonASAPOp, Operator as LogicalOperator, OperatorNode, ScalarExpr}; use planner_types::{ - post_asap::{ - ExactOperation, FieldDataType, PostAsapDAG, PostAsapDAGNode, - PostAsapOperatorPayload as Payload, SketchStatistic, SummaryInputExpr, ValueOperation, - }, + post_asap::{FieldDataType, SketchStatistic, SummaryInputExpr}, pre_asap::{ - AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, - Reduction as PlannerReduction, + AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, Reduction as PlannerReduction, }, }; +mod logical; use std::{ collections::{BTreeMap, BTreeSet}, sync::Arc, @@ -39,7 +41,8 @@ pub mod promql_values; mod candidates; pub use candidates::{ compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, - frontier_from_timing, select_candidate, CandidateCost, CandidateSelection, PhysicalASAPDAG, + frontier_from_timing, select_candidate, CandidateCost, CandidateSelection, + CompiledPhysicalPlan, }; mod compiled; @@ -50,7 +53,7 @@ mod row_values; /// Compile computation without opening or retaining deployment readers. /// Input contracts identify explicit boundaries selected by maintenance planning. pub fn compile( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[NodeId], ) -> Result { @@ -60,7 +63,7 @@ pub fn compile( /// Convenience for callers that already resolved inputs. Lowering still uses /// only their contracts, and instantiation checks those contracts again. pub fn bind<'a>( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, sources: BTreeMap>, roots: &[NodeId], ) -> Result, Error> { @@ -73,11 +76,12 @@ pub fn bind<'a>( /// Resolve raw scan connectors before invoking the reader-independent compiler. pub fn bind_with_data_sources<'a>( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, mut sources: BTreeMap>, roots: &[NodeId], data_sources: &crate::sources::DataSources, ) -> Result, Error> { + let restored = logical::restore(dag)?; // Only resolve scans reachable below the selected input boundaries. let mut pending = roots.to_vec(); let mut seen = BTreeSet::new(); @@ -85,22 +89,19 @@ pub fn bind_with_data_sources<'a>( if !seen.insert(id) || sources.contains_key(&id) { continue; } - let node = dag + let _node = dag .nodes .iter() - .find(|n| u64::from(n.id.0) == id) + .find(|n| n.id as u64 == id) .ok_or_else(|| invalid(format!("missing node {id}")))?; - if let Payload::Fallback { - expression: expression @ QueryExpr::Scan { .. }, - } = &node.payload - { - sources.insert(id, Box::new(data_sources.bind(expression)?)); + if matches!(restored[&id].non_asap(), Some(NonASAPOp::Scan { .. })) { + sources.insert(id, Box::new(data_sources.bind(&restored[&id])?)); } else { pending.extend( dag.edges .iter() - .filter(|e| u64::from(e.consumer.0) == id) - .map(|e| u64::from(e.producer.0)), + .filter(|e| e.consumer as u64 == id) + .map(|e| e.producer as u64), ); } } @@ -123,64 +124,62 @@ fn helper_id(node: NodeId, index: u64) -> NodeId { } fn compile_internal( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, mut sources: BTreeMap, roots: &[NodeId], ) -> Result { preflight_depth(dag)?; + let restored = logical::restore(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag .nodes .iter() - .map(|node| (u64::from(node.id.0), node)) + .map(|node| (node.id as u64, node)) .collect::>(); let mut dependencies = BTreeMap::>::new(); // Binary input order is semantic; serialized edge order is not. let mut edges = dag.edges.iter().collect::>(); edges.sort_by_key(|edge| { ( - edge.consumer.0, + edge.consumer, match edge.role { - planner_types::post_asap::EdgeRole::Left => 0, - planner_types::post_asap::EdgeRole::Input => 1, - planner_types::post_asap::EdgeRole::Right => 2, + planner_types::ir::physical_export::EdgeRole::Left => 0, + planner_types::ir::physical_export::EdgeRole::Input => 1, + planner_types::ir::physical_export::EdgeRole::Right => 2, + planner_types::ir::physical_export::EdgeRole::ScalarRef => 3, }, ) }); - // Scalar literal operands of query-time arithmetic are folded into the consumer. - let mut literals = BTreeMap::::new(); + let literals = BTreeMap::::new(); for edge in edges { - let consumer = u64::from(edge.consumer.0); - if let ( - Payload::Fallback { expression }, - Some(PostAsapDAGNode { - payload: Payload::Binary { .. }, - .. - }), - ) = ( - &nodes[&u64::from(edge.producer.0)].payload, - nodes.get(&consumer), - ) { - if let Some(value) = row_values::scalar_literal(expression) { - let left = edge.role == planner_types::post_asap::EdgeRole::Left; - if literals.insert(consumer, (value, left)).is_some() { - return Err(invalid("binary with two scalar literals is not folded")); - } - continue; - } - } dependencies - .entry(u64::from(edge.consumer.0)) + .entry(edge.consumer as u64) .or_default() - .push(u64::from(edge.producer.0)); + .push(edge.producer as u64); + } + let mut fallback = BTreeMap::new(); + for (&id, root) in &restored { + let raw_summary_input = matches!(root.non_asap(), Some(NonASAPOp::TimeRange { .. })) + && dag.edges.iter().any(|e| { + e.producer as u64 == id + && matches!( + nodes[&(e.consumer as u64)].payload, + Payload::ASAP(ASAPOp::SummaryAgg { .. }) + ) + }); + if !root.contains_asap() && !raw_summary_input { + if let Ok(lowered) = promql_fallback::lower(root) { + fallback.insert(id, lowered); + } + } } let known = |id: &NodeId| { nodes.contains_key(id) || promql_fallback::raw_series_owner(*id).is_some_and(|owner| { matches!( nodes.get(&owner), - Some(PostAsapDAGNode { - payload: Payload::Fallback { .. }, + Some(PhysicalASAPDAGNode { + payload: Payload::NonASAP(_), .. }) ) @@ -204,7 +203,7 @@ fn compile_internal( return Err(invalid(format!("missing root {id}"))); } pending.push((id, true)); - if !sources.contains_key(&id) { + if !sources.contains_key(&id) && !fallback.contains_key(&id) { for &input in dependencies.get(&id).into_iter().flatten() { pending.push((input, false)); } @@ -229,7 +228,9 @@ fn compile_internal( .iter() .map(|id| Arc::new(nodes[id].output_schema.clone())) .collect::>(); - if matches!(node.payload, Payload::SummaryMerge) && inputs.len() > 1 { + if matches!(node.payload, Payload::ASAP(ASAPOp::SummaryMerge { .. })) + && inputs.len() > 1 + { if schemas.iter().any(|s| s != &schemas[0]) { return Err(invalid("summary merge inputs have different schemas")); } @@ -241,20 +242,11 @@ fn compile_internal( inputs = vec![auxiliary]; schemas.truncate(1); } - // A consumed bare selector supplies raw range rows (e.g. to a - // per-entity summary), not an instant vector, so only its consumer computes. - let raw_rows = matches!( - &node.payload, - Payload::Fallback { - expression: QueryExpr::TimeRange { .. } - } - ) && dag.edges.iter().any(|e| u64::from(e.producer.0) == id); - if let (Payload::Fallback { expression }, false) = (&node.payload, raw_rows) { - let promql_fallback::Lowering { - selectors, - mut steps, - } = promql_fallback::lower(expression) - .map_err(|error| invalid(format!("node {id}: {error}")))?; + if let Some(promql_fallback::Lowering { + selectors, + mut steps, + }) = fallback.remove(&id) + { let mut slots = Vec::new(); for (i, (_, schema)) in selectors.iter().enumerate() { let slot = promql_fallback::raw_series_input(id, i); @@ -300,10 +292,7 @@ fn compile_internal( )?; continue; } - if let Payload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &node.payload - { + if let Payload::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &node.payload { use planner_types::post_asap::maintained_population::PopulationInput; let PopulationInput::CurrentSeries(spec) = &population.input else { return Err(invalid( @@ -336,22 +325,18 @@ fn compile_internal( )?; continue; } - if let Payload::Value { - operation: ValueOperation::ReadPopulation { readout }, - } = &node.payload - { + if let Payload::ASAP(ASAPOp::EvaluatePopulation { evaluation, .. }) = &node.payload { use planner_types::post_asap::maintained_population::{ PopulationInput, PopulationStatistic, }; let [producer] = inputs.as_slice() else { - return Err(invalid("population readout requires one input")); + return Err(invalid("population evaluation requires one input")); }; - let Payload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &nodes[producer].payload + let Payload::ASAP(ASAPOp::MaintainPopulation { population, .. }) = + &nodes[producer].payload else { return Err(invalid( - "population readout requires its declared population", + "population evaluation requires its declared population", )); }; let PopulationInput::CurrentSeries(spec) = &population.input else { @@ -363,9 +348,9 @@ fn compile_internal( )); } let input = schemas[0].clone(); - let PopulationStatistic::TopK { k } = readout else { + let PopulationStatistic::TopK { k } = evaluation else { let mut chain = - row_values::population_aggregate(&input, &spec.grouping, readout)?; + row_values::population_aggregate(&input, &spec.grouping, evaluation)?; let last = chain.pop().expect("nonempty chain"); let mut inputs = inputs; for operator in chain { @@ -404,26 +389,24 @@ fn compile_internal( } // A closed row must include either all source labels or the explicit // complete-label identity. Projected labels alone are insufficient. - if let Payload::SummaryAgg { + if let Payload::ASAP(ASAPOp::SummaryAgg { family, input: update, reduction: PlannerReduction::PerEntity, grouping, filter: None, - } = &node.payload + .. + }) = &node.payload { let [input_id] = inputs.as_slice() else { return Err(invalid("per-entity summary requires one input")); }; - let Payload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, - } = &nodes[input_id].payload - else { + let Some(NonASAPOp::TimeRange { child, .. }) = restored[input_id].non_asap() else { return Err(invalid( "per-entity summary requires a resolved raw time range", )); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let Some(NonASAPOp::Scan { schema, .. }) = child.non_asap() else { return Err(invalid("per-entity summary requires a resolved source")); }; if !schema.closed || update.item.is_some() { @@ -462,7 +445,16 @@ fn compile_internal( )?; continue; } - if let Payload::Binary { operator } = &node.payload { + if let Payload::NonASAP(NonASAPOp::BinaryOp { + operator, + return_bool, + .. + }) = &node.payload + { + let operator = crate::expressions::binary::BinaryOperator::from_logical( + operator, + *return_bool, + ); let query_time = node.output_state.timing == planner_types::post_asap::ExecutionTiming::QueryTime; if let Some(&(value, left)) = literals.get(&id) { @@ -505,19 +497,11 @@ fn compile_internal( // carry the series identity. if let (true, [left, right]) = (query_time, schemas.as_slice()) { if !label_map(left) && !label_map(right) { - // A scalar-valued Fallback operand, such as `scalar(x)`, has no labels. - let scalar = |input: &NodeId| { - matches!( - nodes.get(input).map(|node| &node.payload), - Some(Payload::Fallback { expression }) - if promql_fallback::scalar(expression) - ) - }; let binary = Operator::series_binary( left.clone(), right.clone(), operator.clone(), - [scalar(&inputs[0]), scalar(&inputs[1])], + [false, false], ) .map_err(|error| invalid(format!("node {id}: {error}")))?; physical_dag.add(id, inputs, binary.with_output_schema(output)?)?; @@ -525,14 +509,11 @@ fn compile_internal( } } } - if let Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - } = &node.payload - { + if let Payload::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) = &node.payload { // Exact counts read out as Int64; PromQL declares a Float64 sample. - let readout = bind_operation(node, &schemas) + let evaluation = bind_operation(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; - let actual = readout.schema(); + let actual = evaluation.schema(); let converted = actual.fields.iter().zip(&output.fields).position(|(a, d)| { a.dtype == FieldDataType::Plain(DataType::Int64) && d.dtype == FieldDataType::Plain(DataType::Float64) @@ -555,8 +536,8 @@ fn compile_internal( .collect(); let project = Operator::project(actual, columns)?.with_output_schema(output.clone())?; - physical_dag.add(auxiliary, inputs, readout)?; - if temporal_readout_drops_name(node) { + physical_dag.add(auxiliary, inputs, evaluation)?; + if temporal_evaluation_drops_name(node) { physical_dag.add(auxiliary - 1, vec![auxiliary], project)?; physical_dag.add( id, @@ -572,7 +553,7 @@ fn compile_internal( } let mut operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; - if operator.is_counter_readout() { + if operator.is_counter_evaluation() { let mut pending = vec![id]; let mut visited = BTreeSet::new(); let mut ranges = BTreeSet::new(); @@ -580,9 +561,8 @@ fn compile_internal( if !visited.insert(ancestor) { continue; } - if let Payload::Fallback { - expression: QueryExpr::TimeRange { range, .. }, - } = &nodes[&ancestor].payload + if let Payload::NonASAP(NonASAPOp::TimeRange { range, .. }) = + &nodes[&ancestor].payload { ranges.insert( i64::try_from(range.as_millis()) @@ -593,13 +573,13 @@ fn compile_internal( pending.extend(dependencies.get(&ancestor).into_iter().flatten().copied()); } if ranges.len() > 1 { - return Err(invalid("counter readout has ambiguous logical windows")); + return Err(invalid("counter evaluation has ambiguous logical windows")); } if let Some(lookback) = ranges.into_iter().next() { operator = operator.with_counter_lookback(lookback)?; } } - if temporal_readout_drops_name(node) { + if temporal_evaluation_drops_name(node) { physical_dag.add(auxiliary, inputs, operator)?; physical_dag.add(id, vec![auxiliary], Operator::series_without_name(output)?)?; } else { @@ -611,38 +591,45 @@ fn compile_internal( Ok(physical_dag) } -// Temporal summary readouts produce PromQL vectors, whose range functions drop +// Temporal summary evaluations produce PromQL vectors, whose range functions drop // the metric name before matching/filtering. Stored state retains its full identity. -fn temporal_readout_drops_name(node: &PostAsapDAGNode) -> bool { +fn temporal_evaluation_drops_name(node: &PhysicalASAPDAGNode) -> bool { node.output_schema .fields .iter() .any(|field| field.name == promql_rows::SERIES_IDENTITY_COLUMN) && matches!( &node.payload, - Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } | Payload::SummaryEstimate { - query: SketchStatistic::Quantile { .. } - | SketchStatistic::Cardinality - | SketchStatistic::PointCount { .. } - | SketchStatistic::FrequencyL2 - | SketchStatistic::FrequencyEntropy - } + Payload::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) + | Payload::ASAP(ASAPOp::SummaryEstimate { + query: SketchStatistic::Quantile { .. } + | SketchStatistic::Cardinality + | SketchStatistic::PointCount { .. } + | SketchStatistic::FrequencyL2 + | SketchStatistic::FrequencyEntropy, + .. + }) ) } /// Bind a Planner node against the schemas supplied by its deployment edges. /// This is the same checked path used by complete DAG binding. -pub fn compile_node(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { +pub fn compile_node(node: &PhysicalASAPDAGNode, inputs: &[SchemaRef]) -> Result { for schema in inputs { crate::values::validate_schema(schema)?; } bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) } -fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { - if let Payload::Binary { operator } = &node.payload { +fn bind_operation(node: &PhysicalASAPDAGNode, inputs: &[SchemaRef]) -> Result { + if let Payload::NonASAP(NonASAPOp::BinaryOp { + operator, + return_bool, + .. + }) = &node.payload + { + let operator = + crate::expressions::binary::BinaryOperator::from_logical(operator, *return_bool); let [left, right] = inputs else { return Err(invalid("binary requires two inputs")); }; @@ -688,54 +675,82 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result, Error>>() + }) + .collect::, Error>>()?; + let schema = Arc::new(schema.clone()); + return Operator::source( + schema.clone(), + vec![crate::values::Batch::try_new(schema, rows)?], + ); + } let [input] = inputs else { return Err(invalid( "native Planner binding currently requires a unary operation or an explicit source", )); }; match &node.payload { - Payload::Value { operation, .. } => match operation { - ValueOperation::Project { cols, .. } => Operator::project( + Payload::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) => { + let state = summary_column(input)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let statistic = match &input.fields[state].dtype { + FieldDataType::ExactAggregate(kind, _) => match kind { + E::Sum => S::Sum, + E::Count => S::Count, + E::Min => S::Min, + E::Max => S::Max, + E::Rate => S::Rate, + E::Increase => S::Increase, + _ => return Err(invalid("exact family evaluation is unsupported")), + }, + _ => return Err(invalid("exact finalization requires exact state")), + }; + Operator::evaluation( + input.clone(), + state, + SummaryEvaluation::Exact(ExactEvaluation { + statistic, + lookback_ms: None, + }), + ) + } + + Payload::NonASAP(operator) => match operator { + NonASAPOp::Project { cols, .. } => Operator::project( input.clone(), cols.iter() .enumerate() @@ -748,21 +763,23 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result Expression::Column(*index), + ScalarExpr::Column(index) => Expression::Column(*index), expr => expression(expr, input)?, }, )) }) .collect::>()?, ), - ValueOperation::Filter { pred } => { + NonASAPOp::Filter { pred, .. } => { Operator::filter(input.clone(), expression(&pred.0, input)?) } - ValueOperation::Sort { keys, partition_by } => Operator::sort( + NonASAPOp::Sort { + keys, partition_by, .. + } => Operator::sort( input.clone(), keys.iter() .map(|key| { - let QueryExpr::Column(column) = key.expr else { + let ScalarExpr::Column(column) = key.expr else { return Err(invalid( "sort expression must be projected before sorting", )); @@ -776,23 +793,25 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result>()?, groups(input, partition_by)?, ), - ValueOperation::Limit { + NonASAPOp::Limit { n, offset, partition_by, + .. } => Operator::limit( input.clone(), - *n as u64, + n.unwrap_or(usize::MAX) as u64, *offset as u64, groups(input, partition_by)?, ), - ValueOperation::Exact(ExactOperation::Aggregate { + NonASAPOp::Aggregate { reduction, measures, output_names, filters, having: None, - }) => { + .. + } => { if filters.iter().any(Option::is_some) { return Err(invalid("filtered aggregate has no native implementation")); } @@ -829,40 +848,16 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result>()?; Operator::aggregate(input.clone(), groups(input, keys)?, measures) } - ValueOperation::FinalizeExactAccumulator => { - let state = summary_column(input)?; - use crate::Statistic as S; - use planner_types::post_asap::ExactKind as E; - let statistic = match &input.fields[state].dtype { - FieldDataType::ExactAggregate(kind, _) => match kind { - E::Sum => S::Sum, - E::Count => S::Count, - E::Min => S::Min, - E::Max => S::Max, - E::Rate => S::Rate, - E::Increase => S::Increase, - _ => return Err(invalid("exact family readout is unsupported")), - }, - _ => return Err(invalid("exact finalization requires exact state")), - }; - Operator::readout( - input.clone(), - state, - ReadoutQuery::Exact(ExactReadout { - statistic, - lookback_ms: None, - }), - ) - } _ => Err(invalid("value operation has no native implementation")), }, - Payload::SummaryAgg { + Payload::ASAP(ASAPOp::SummaryAgg { family, input: update, reduction, grouping, filter, - } => { + .. + }) => { if filter.is_some() { return Err(invalid( "filtered summary update has no native implementation", @@ -933,7 +928,7 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { + Payload::ASAP(ASAPOp::SummaryMerge { .. }) => { let state = summary_column(input)?; Operator::summary_merge( input.clone(), @@ -943,19 +938,19 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { + Payload::ASAP(ASAPOp::SummaryEstimate { query, .. }) => { if let SketchStatistic::TopK { k } = query { - return Operator::keyed_readout( + return Operator::keyed_evaluation( input.clone(), summary_column(input)?, *k, Arc::new(node.output_schema.clone()), ); } - Operator::readout( + Operator::evaluation( input.clone(), summary_column(input)?, - ReadoutQuery::Sketch(query.clone()), + SummaryEvaluation::Sketch(query.clone()), ) } _ => Err(invalid( @@ -1009,9 +1004,13 @@ fn groups(input: &SchemaRef, groups: &GroupKeys) -> Result, Error> { } Ok(groups.keys().to_vec()) } -fn expression(expr: &QueryExpr, input: &SchemaRef) -> Result { +fn expression( + expr: &ScalarExpr, + input: &SchemaRef, +) -> Result { + let expr = local_scalar(expr)?; Ok(Expression::planner( - crate::expressions::CompiledExpression::compile(expr, input)?, + crate::expressions::CompiledExpression::compile(&expr, input)?, )) } @@ -1057,7 +1056,7 @@ impl PhysicalOperator for CheckedSource<'_> { } // Bound recursion before invoking the upstream recursive provenance validator. -fn preflight_depth(dag: &PostAsapDAG) -> Result<(), Error> { +fn preflight_depth(dag: &PhysicalASAPDAG) -> Result<(), Error> { let mut remaining = dag .nodes .iter() @@ -1110,23 +1109,24 @@ fn preflight_depth(dag: &PostAsapDAG) -> Result<(), Error> { /// Join predicates address the concatenated left/right schema. fn semi_join_keys( - expr: &QueryExpr, + expr: &ScalarExpr, left: usize, right: usize, keys: &mut Vec<(usize, usize)>, ) -> Result<(), Error> { match expr { - QueryExpr::BoolAnd(parts) => { + ScalarExpr::BoolAnd(parts) => { for part in parts { semi_join_keys(part, left, right, keys)?; } } - QueryExpr::Compare { + ScalarExpr::Compare { left: a, op: CompareOpKind::Eq, right: b, + .. } => { - let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + let (ScalarExpr::Column(a), ScalarExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { return Err(invalid("semi-join requires column equality keys")); }; let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; @@ -1143,7 +1143,7 @@ fn semi_join_keys( /// Resolve equality keys against the Planner join's concatenated input schema. /// Deployments may use these positions to bind their source columns. pub fn equijoin_keys( - pred: &planner_types::pre_asap::Predicate, + pred: &planner_types::ir::Predicate, left: &planner_types::post_asap::Schema, right: &planner_types::post_asap::Schema, ) -> Result, Error> { @@ -1154,3 +1154,12 @@ pub fn equijoin_keys( } Ok(keys) } + +fn local_scalar(expr: &ScalarExpr) -> Result { + if !expr.operator_refs().is_empty() { + return Err(invalid( + "scalar plan reads require explicit execution bindings", + )); + } + Ok(expr.map_operator_refs(&mut |_| unreachable!("no operator references"))) +} diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 10dffdc0c..6ff9196aa 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -1,6 +1,9 @@ //! Compile immutable summary-input computation with explicit population and pane identity. use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; use super::*; +use planner_types::ir::ASAPOp; +use planner_types::ir::NonASAPOp; +use planner_types::post_asap::FieldDataType as SummaryFamilyType; use planner_types::{ post_asap::{ExecutionTiming, GroupingStrategy, Schema}, pre_asap::DataType, @@ -8,35 +11,35 @@ use planner_types::{ /// Physical rows carry the population and pane coordinate alongside the logical value. /// These fields preserve identities which are implicit in a stored summary instance. -pub fn population_schema(family: FieldDataType) -> SchemaRef { +pub fn population_schema(family: SummaryFamilyType) -> SchemaRef { Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![ planner_types::post_asap::Field { - table: None, name: "$population".into(), - dtype: FieldDataType::Plain(DataType::Map { + dtype: SummaryFamilyType::Plain(DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::Utf8), value_nullable: false, }), nullable: false, + table: None, }, planner_types::post_asap::Field { - table: None, name: "$window_end".into(), - dtype: FieldDataType::Plain(DataType::Timestamp), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), nullable: false, + table: None, }, planner_types::post_asap::Field { - table: None, name: "value".into(), dtype: family, nullable: false, + table: None, }, ], time_index: Some(1), + unique_keys: vec![], + closed: false, }) } @@ -48,7 +51,7 @@ pub fn population_schema(family: FieldDataType) -> SchemaRef { /// must be canonical (sorted, unique, no empty values), since they are the /// population identity: build rows with [`raw_sample_row`]. pub fn raw_sample_schema() -> SchemaRef { - let mut schema = (*population_schema(FieldDataType::Plain(DataType::Float64))).clone(); + let mut schema = (*population_schema(SummaryFamilyType::Plain(DataType::Float64))).clone(); schema.fields[1].name = "$timestamp".into(); Arc::new(schema) } @@ -82,20 +85,15 @@ pub fn raw_sample_row( /// Input contract of a precompute boundary: raw sample rows for a raw time /// series scan, otherwise the stored population of its summary state. -pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { - let Payload::Fallback { expression } = &node.payload else { - return source_schema(&node.output_schema); - }; - let scan = match expression { - planner_types::pre_asap::QueryExpr::TimeRange { child, .. } => child.as_ref(), - expression => expression, - }; +pub fn boundary_schema(node: &PhysicalASAPDAGNode) -> Result { if !matches!( - scan, - planner_types::pre_asap::QueryExpr::Scan { - source: planner_types::pre_asap::Source::TimeSeries { .. }, - .. - } + &node.payload, + Payload::NonASAP( + NonASAPOp::Scan { + source: planner_types::pre_asap::Source::TimeSeries { .. }, + .. + } | NonASAPOp::TimeRange { .. } + ) ) { return source_schema(&node.output_schema); } @@ -106,11 +104,11 @@ pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { .iter() .enumerate() .all(|(i, field)| match &field.dtype { - FieldDataType::Plain(DataType::Timestamp) => { + SummaryFamilyType::Plain(DataType::Timestamp) => { Some(i) == logical.time_index && !field.nullable } - FieldDataType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, - FieldDataType::Plain(DataType::Utf8) => true, + SummaryFamilyType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, + SummaryFamilyType::Plain(DataType::Utf8) => true, _ => false, }) && !logical @@ -134,14 +132,14 @@ pub fn source_schema(logical: &Schema) -> Result { let states = logical .fields .iter() - .filter(|f| !matches!(f.dtype, FieldDataType::Plain(_))) + .filter(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) .collect::>(); let [state] = states.as_slice() else { return Err(invalid( "stored population requires one typed summary state", )); }; - if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, FieldDataType::Plain(dtype) + if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, SummaryFamilyType::Plain(dtype) if field.nullable || !matches!(dtype, DataType::Utf8) && !(Some(i) == logical.time_index && *dtype == DataType::Timestamp))) { return Err(invalid("stored population metadata cannot reconstruct extra value columns")); } @@ -161,7 +159,7 @@ pub fn is_population_schema(schema: &SchemaRef) -> bool { /// Compile a complete selected precompute sub-DAG. Inputs are already-computed /// state boundaries; the deployment supplies groups, panes and states, never operations. pub fn compile( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, frontiers: &[NodeId], roots: &[NodeId], ) -> Result { @@ -170,7 +168,7 @@ pub fn compile( let nodes = dag .nodes .iter() - .map(|n| (u64::from(n.id.0), n)) + .map(|n| (n.id as u64, n)) .collect::>(); let frontier = frontiers.iter().copied().collect::>(); if frontier.len() != frontiers.len() || roots.iter().any(|r| frontier.contains(r)) { @@ -182,19 +180,20 @@ pub fn compile( let mut edges = dag.edges.iter().collect::>(); edges.sort_by_key(|edge| { ( - edge.consumer.0, + edge.consumer, match edge.role { - planner_types::post_asap::EdgeRole::Left => 0, - planner_types::post_asap::EdgeRole::Input => 1, - planner_types::post_asap::EdgeRole::Right => 2, + planner_types::ir::physical_export::EdgeRole::Left => 0, + planner_types::ir::physical_export::EdgeRole::Input => 1, + planner_types::ir::physical_export::EdgeRole::Right => 2, + planner_types::ir::physical_export::EdgeRole::ScalarRef => 3, }, ) }); for edge in edges { dependencies - .entry(u64::from(edge.consumer.0)) + .entry(edge.consumer as u64) .or_default() - .push(u64::from(edge.producer.0)); + .push(edge.producer as u64); } let mut ordered = Vec::new(); let mut seen = BTreeSet::new(); @@ -261,7 +260,7 @@ pub fn compile( CompiledPhysicalDAG::compose(sources, fragments, roots.to_vec()) } -fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { +fn validate_value_output(node: &PhysicalASAPDAGNode) -> Result<(), Error> { let schema = &node.output_schema; // Physical population rows already carry the complete identity in `$population`. // Typed logical plans may expose its opaque series-identity column as metadata. @@ -274,7 +273,7 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { if identities.len() > 1 || identities .iter() - .any(|field| field.nullable || field.dtype != FieldDataType::Plain(DataType::Utf8)) + .any(|field| field.nullable || field.dtype != SummaryFamilyType::Plain(DataType::Utf8)) { return Err(invalid( "precompute series identity requires one non-null Utf8 column", @@ -286,12 +285,11 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { .enumerate() .filter(|(i, field)| Some(*i) != schema.time_index && field.name != identity) .collect::>(); - if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == FieldDataType::Plain(DataType::Float64)) + if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == SummaryFamilyType::Plain(DataType::Float64)) || schema.time_index.is_some_and(|i| { - schema - .fields - .get(i) - .is_none_or(|f| f.nullable || f.dtype != FieldDataType::Plain(DataType::Timestamp)) + schema.fields.get(i).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) }) { return Err(invalid( @@ -302,9 +300,9 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { } fn fragment( - node: &PostAsapDAGNode, + node: &PhysicalASAPDAGNode, schemas: &[SchemaRef], - parents: &[&PostAsapDAGNode], + parents: &[&PhysicalASAPDAGNode], ) -> Result { let sources = schemas .iter() @@ -320,7 +318,13 @@ fn fragment( Ok(id) }; let root = match &node.payload { - Payload::Binary { operator } => { + Payload::NonASAP(NonASAPOp::BinaryOp { + operator, + return_bool, + .. + }) => { + let operator = + crate::expressions::binary::BinaryOperator::from_logical(operator, *return_bool); validate_value_output(node)?; if node.output_schema.time_index.is_none() || parents.iter().any(|p| p.output_schema.time_index.is_none()) @@ -343,30 +347,29 @@ fn fragment( )?, )? } - Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - } => { + Payload::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) => { let [input] = schemas else { return Err(invalid("finalize requires one state input")); }; validate_value_output(node)?; let statistic = match &input.fields[2].dtype { - FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { crate::Statistic::Sum } - FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) => { - crate::Statistic::Count - } + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Count, + _, + ) => crate::Statistic::Count, _ => { return Err(invalid( "precompute finalization requires explicit Sum or Count semantics", )) } }; - let read = Operator::readout( + let read = Operator::evaluation( input.clone(), 2, - ReadoutQuery::Exact(ExactReadout { + SummaryEvaluation::Exact(ExactEvaluation { statistic, lookback_ms: None, }), @@ -384,16 +387,19 @@ fn fragment( ), ], )? - .with_output_schema(population_schema(FieldDataType::Plain(DataType::Float64)))?; + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; add(vec![read], project)? } - Payload::SummaryAgg { + Payload::ASAP(ASAPOp::SummaryAgg { family, input: update, reduction, grouping, filter, - } => { + .. + }) => { if filter.is_some() { return Err(invalid( "filtered summary update has no native implementation", @@ -403,12 +409,12 @@ fn fragment( return Err(invalid("summary update requires one input")); }; // Item identities resolve against the complete label set of raw - // samples; finalized readouts carry no such identity. + // samples; finalized evaluations carry no such identity. let raw = *input == raw_sample_schema(); // A unit-frequency summary (HLL) observes each raw sample value. let unit_frequency = raw && crate::capability::is_unit_sample_frequency(update) - && matches!(family, FieldDataType::Sketch(kind, _) if !matches!( + && matches!(family, SummaryFamilyType::Sketch(kind, _) if !matches!( kind.algorithm(), planner_types::post_asap::SketchAlgorithm::Cms | planner_types::post_asap::SketchAlgorithm::CountSketch @@ -436,7 +442,7 @@ fn fragment( )); } if keyed - && matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) && !matches!( update.weight_domain, planner_types::post_asap::WeightDomain::NonNegative { .. } @@ -461,7 +467,7 @@ fn fragment( .filter(|field| { (raw || !field.nullable) && field.name != SERIES_IDENTITY - && field.dtype == FieldDataType::Plain(DataType::Utf8) + && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) }) .map(|f| f.name.clone()) .ok_or_else(|| { @@ -481,7 +487,7 @@ fn fragment( SummaryInputExpr::Column(ColumnRef::SampleValue) => Expression::Column(2), SummaryInputExpr::Column(ColumnRef::Named(name)) if parents[0].output_schema.fields.iter().any(|f| { - f.name == *name && f.dtype == FieldDataType::Plain(DataType::Float64) + f.name == *name && f.dtype == SummaryFamilyType::Plain(DataType::Float64) }) => { Expression::Column(2) @@ -497,7 +503,7 @@ fn fragment( ("$window_end".into(), Expression::Column(1)), ("value".into(), Expression::FiniteFloat64(Box::new(weight))), ]; - let mut fields = population_schema(FieldDataType::Plain(DataType::Float64)) + let mut fields = population_schema(SummaryFamilyType::Plain(DataType::Float64)) .fields .clone(); if keyed { @@ -510,10 +516,10 @@ fn fragment( for (index, (expression, dtype)) in items.into_iter().enumerate() { let name = format!("$item{index}"); fields.push(planner_types::post_asap::Field { - table: None, name: name.clone(), - dtype: FieldDataType::Plain(dtype), + dtype: SummaryFamilyType::Plain(dtype), nullable: false, + table: None, }); columns.push((name, expression)); } @@ -521,9 +527,9 @@ fn fragment( let item_columns = (3..fields.len()).collect::>(); let project = Operator::project(input.clone(), columns)?.with_output_schema( Arc::new(Schema { - closed: true, - unique_keys: vec![], fields, + unique_keys: vec![], + closed: false, time_index: Some(1), }), )?; @@ -541,7 +547,7 @@ fn fragment( Operator::scope_timestamp(built, population_schema(family.clone()))?, )? } - Payload::SummaryMerge => { + Payload::ASAP(ASAPOp::SummaryMerge { .. }) => { let Some(input) = schemas.first() else { return Err(invalid("summary merge requires inputs")); }; @@ -583,7 +589,7 @@ fn raw_items( ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } if !name.starts_with('$') && scan.fields.iter().all(|f| { - &f.name != name || f.dtype == FieldDataType::Plain(DataType::Utf8) + &f.name != name || f.dtype == SummaryFamilyType::Plain(DataType::Utf8) }) => { Some(name.clone()) diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index 237504c98..fcd1382ab 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -19,14 +19,14 @@ pub(super) fn raw_series_owner(slot: NodeId) -> Option { } /// A selector expression and its raw-series row schema. -pub type Selector = (QueryExpr, SchemaRef); +pub type Selector = (OperatorNode, SchemaRef); /// The selectors a Fallback expression reads, left to right, and the row /// schema of the raw series the deployment supplies for each at /// [`raw_series_input`]. The rows must cover the selector's window at every /// evaluation instant `T`, or at its `@` time: `(T - offset - range, T - offset]`; /// under a subquery `[R:S] offset O` that is `(T - O - R - offset - range, T - O - offset]`. -pub fn raw_series(expression: &QueryExpr) -> Result, Error> { +pub fn raw_series(expression: &OperatorNode) -> Result, Error> { Ok(lower(expression)?.selectors) } @@ -43,16 +43,47 @@ pub(super) struct Lowering { pub steps: Vec<(Operator, Vec)>, } -pub(super) fn lower(expression: &QueryExpr) -> Result { +pub(super) fn lower(expression: &OperatorNode) -> Result { let mut lowering = Lowering::default(); lowering.value(expression)?; Ok(lowering) } -fn declared(expression: &QueryExpr) -> Result { - let schema = expression - .output_schema() - .map_err(|error| invalid(error.to_string()))?; +/// Compile a standalone scalar expression and expose its real series dependencies. +/// Input slots use root 0; no logical wrapper node is introduced. +pub fn compile_scalar_root( + expr: &ScalarExpr, +) -> Result<(CompiledPhysicalDAG, Vec), Error> { + let mut lowering = Lowering::default(); + lowering.scalar_value(expr)?; + let mut inputs = BTreeMap::new(); + for (i, (_, schema)) in lowering.selectors.iter().enumerate() { + inputs.insert( + raw_series_input(0, i), + InputContract::bounded(schema.clone()), + ); + } + let last = lowering.steps.len() - 1; + let mut operators = BTreeMap::new(); + for (i, (operator, dependencies)) in lowering.steps.into_iter().enumerate() { + let id = if i == last { 0 } else { i as u64 + 1 }; + let dependencies = dependencies + .into_iter() + .map(|input| match input { + Input::Raw(i) => raw_series_input(0, i), + Input::Step(i) => i as u64 + 1, + }) + .collect(); + operators.insert(id, (dependencies, operator)); + } + Ok(( + CompiledPhysicalDAG::from_operators(inputs, operators, vec![0])?, + lowering.selectors, + )) +} + +fn declared(expression: &OperatorNode) -> Result { + let schema = expression.schema.clone(); Ok(Arc::new(lift_plain(&schema))) } @@ -69,10 +100,10 @@ fn at(shift: &planner_types::pre_asap::TimeShift) -> Result, Error> } } -fn range_anchor(expression: &QueryExpr) -> Option { - match expression { - QueryExpr::TimeRange { child, .. } => range_anchor(child), - QueryExpr::TimeShift { shift, .. } => shift +fn range_anchor(expression: &OperatorNode) -> Option { + match expression.expect_non_asap() { + NonASAPOp::TimeRange { child, .. } => range_anchor(child), + NonASAPOp::TimeShift { shift, .. } => shift .at .filter(|at| matches!(at, AtModifier::Start | AtModifier::End)), _ => None, @@ -80,32 +111,22 @@ fn range_anchor(expression: &QueryExpr) -> Option { } /// `TimeRange { range, [TimeShift { offset, @ }], Scan }`: range, offset, `@`. -fn selector(expression: &QueryExpr) -> Result<(i64, i64, Option), Error> { - let QueryExpr::TimeRange { range, child } = expression else { +fn selector(expression: &OperatorNode) -> Result<(i64, i64, Option), Error> { + let NonASAPOp::TimeRange { range, child, .. } = expression.expect_non_asap() else { return Err(invalid("PromQL operand must be a series selector")); }; - let (offset, at, scan) = match child.as_ref() { - QueryExpr::TimeShift { shift, child } => (shift.offset_ms, at(shift)?, child.as_ref()), + let (offset, at, scan) = match child.expect_non_asap() { + NonASAPOp::TimeShift { shift, child } => { + (shift.offset_ms, at(shift)?, child.expect_non_asap()) + } scan => (0, None, scan), }; - if !matches!(scan, QueryExpr::Scan { .. }) { + if !matches!(scan, NonASAPOp::Scan { .. }) { return Err(invalid("PromQL selector must read one scan")); } Ok((millis(range)?, offset, at)) } -/// PromQL scalar-valued expressions have no labels to match. A binary -/// operator is scalar-valued when both operands are. -pub(super) fn scalar(expression: &QueryExpr) -> bool { - match expression { - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp => true, - QueryExpr::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), - _ => false, - } -} - impl Lowering { fn schema(&self, input: &Input) -> SchemaRef { match input { @@ -124,12 +145,12 @@ impl Lowering { &mut self, operator: Operator, inputs: Vec, - logical: &QueryExpr, + logical: &OperatorNode, ) -> Result { Ok(self.add(operator.with_output_schema(declared(logical)?)?, inputs)) } - fn read(&mut self, selector: &QueryExpr) -> Result { + fn read(&mut self, selector: &OperatorNode) -> Result { let schema = declared(selector)?; if !schema .fields @@ -145,12 +166,12 @@ impl Lowering { } /// An instant vector, or a scalar for scalar-valued expressions. - fn value(&mut self, expression: &QueryExpr) -> Result { - match expression { - QueryExpr::Concat { children, .. } => { - if !children.iter().all(|branch| matches!(branch, - QueryExpr::PromqlRelabel { child, .. } if matches!(child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { + fn value(&mut self, expression: &OperatorNode) -> Result { + match expression.expect_non_asap() { + NonASAPOp::Concat { children, .. } => { + if !children.iter().all(|branch| matches!(branch.expect_non_asap(), + NonASAPOp::PromqlRelabel { child, .. } if matches!(child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { return Err(invalid("PromQL concatenation requires classic histogram quantile branches")); } let inputs = children @@ -171,15 +192,17 @@ impl Lowering { expression, ) } - QueryExpr::PromqlRelabel { dst, value, child } => { + NonASAPOp::PromqlRelabel { dst, value, child } => { let step = self.value(child)?; let input = self.schema(&step); - let (replacement, source_regex) = match value.as_ref() { - QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(value)) => { + let (replacement, source_regex) = match value { + ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(value)) => { (value.clone(), None) } - QueryExpr::FunctionCall { name, args } if name == "label_replace" => { - let [QueryExpr::Column(source), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(pattern)), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( + ScalarExpr::FunctionCall { name, args } if name == "label_replace" => { + let [ScalarExpr::Column(source), ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( + pattern, + )), ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( replacement, ))] = args.as_slice() else { @@ -204,7 +227,7 @@ impl Lowering { )?; self.push(operator, vec![step], expression) } - QueryExpr::TimeRange { .. } => { + NonASAPOp::TimeRange { .. } => { let (range, offset, at) = selector(expression)?; let input = self.read(expression)?; let schema = self.schema(&input); @@ -215,7 +238,7 @@ impl Lowering { expression, ) } - QueryExpr::Aggregate { + NonASAPOp::Aggregate { reduction: planner_types::pre_asap::Reduction::PerEntity, measures, having: None, @@ -234,7 +257,7 @@ impl Lowering { let input = self.schema(&step); Ok(self.add(Operator::series_without_name(input)?, vec![step])) } - QueryExpr::Aggregate { + NonASAPOp::Aggregate { reduction: planner_types::pre_asap::Reduction::Reduce(keys), measures, having: None, @@ -248,7 +271,74 @@ impl Lowering { let input = self.value(child)?; self.aggregate(input, measure, keys, expression) } - QueryExpr::Sort { + NonASAPOp::Project { + cols, + child, + qualifier, + } => { + let value = planner_types::pre_asap::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + ) + .map_err(|e| invalid(e.to_string()))?; + let sample = cols + .iter() + .find(|col| { + col.alias.as_deref() == Some(child.schema.fields[value].name.as_str()) + }) + .ok_or_else(|| invalid("missing sample projection"))?; + let keep_name = matches!(sample.expr, ScalarExpr::Negative { .. }); + let fields: Vec<_> = child + .schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| keep_name || field.name != "__name__") + .collect(); + if qualifier.is_some() || cols.len() != fields.len() { + return Err(invalid("unsupported temporal projection shape")); + } + let mut computed = None; + for (col, (index, field)) in cols.iter().zip(fields) { + if col.alias.as_deref() != Some(field.name.as_str()) { + return Err(invalid("unsupported temporal projection alias")); + } + if index == value { + computed = Some(col); + } else { + let expected = if !keep_name + && field.name == planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY + { + ScalarExpr::FunctionCall { + name: "promql_drop_metric_name".into(), + args: vec![ScalarExpr::Column(index)], + } + } else { + ScalarExpr::Column(index) + }; + if col.expr != expected { + return Err(invalid("unsupported temporal projection expression")); + } + } + } + let computed = computed.ok_or_else(|| invalid("no computed sample"))?; + if matches!( + computed.expr, + ScalarExpr::Negative { .. } | ScalarExpr::FunctionCall { .. } + ) { + return self.pointwise_projection(cols, child, value, expression, keep_name); + } + self.sample_scalar_operation(&computed.expr, child, value, expression) + } + NonASAPOp::Filter { pred, child } => { + let value = planner_types::pre_asap::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + ) + .map_err(|e| invalid(e.to_string()))?; + self.sample_scalar_operation(&pred.0, child, value, expression) + } + NonASAPOp::Sort { keys, partition_by, child, @@ -258,7 +348,7 @@ impl Lowering { let keys = keys .iter() .map(|key| match key.expr { - QueryExpr::Column(column) => Ok(SortKey { + ScalarExpr::Column(column) => Ok(SortKey { column, descending: !key.ascending, nulls_first: key.nulls_first, @@ -269,70 +359,223 @@ impl Lowering { let groups = groups(&input, partition_by)?; self.push(Operator::sort(input, keys, groups)?, vec![step], expression) } - QueryExpr::Limit { n, offset, child } => { + NonASAPOp::Limit { + n, offset, child, .. + } => { let step = self.value(child)?; let input = self.schema(&step); // `topk by (...)` partitions through the Sort it limits. - let groups = match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => groups(&input, partition_by)?, + let groups = match child.expect_non_asap() { + NonASAPOp::Sort { partition_by, .. } => groups(&input, partition_by)?, _ => vec![], }; self.push( - Operator::limit(input, *n as u64, *offset as u64, groups)?, + Operator::limit( + input, + n.unwrap_or(usize::MAX) as u64, + *offset as u64, + groups, + )?, vec![step], expression, ) } - QueryExpr::BinaryOp { - op, + NonASAPOp::BinaryOp { + operator, lhs, rhs, - vector_match, + return_bool, } => { let sides = vec![self.value(lhs)?, self.value(rhs)?]; - let operator = planner_types::post_asap::BinaryOperator { - kind: op.clone(), - vector_match: vector_match.clone(), - checked_relative_division: false, - checked_finite_division: false, - }; + let operator = crate::expressions::binary::BinaryOperator::from_logical( + operator, + *return_bool, + ); let binary = Operator::series_binary( self.schema(&sides[0]), self.schema(&sides[1]), operator, - [scalar(lhs), scalar(rhs)], + [false, false], )?; self.push(binary, sides, expression) } - QueryExpr::PromqlScalarFromVector(child) => { - let step = self.value(child)?; - let input = self.schema(&step); - let value = named_column(&input, &ColumnRef::SampleValue)?; - self.push( - Operator::vector_to_scalar(input, value)?, - vec![step], - expression, - ) - } - QueryExpr::PromqlVectorFromScalar(child) => { - let step = self.value(child)?; + NonASAPOp::PromqlVectorFromScalar(expr) => { + let step = self.scalar_value(expr)?; let input = self.schema(&step); Ok(self.add( Operator::scope_timestamp(input, declared(expression)?)?, vec![step], )) } - QueryExpr::EvalTimestamp => self.push(Operator::evaluation_time(), vec![], expression), - QueryExpr::PromqlScalarBridge(_) => { - let value = row_values::scalar_literal(expression) - .ok_or_else(|| invalid("PromQL scalar must be a literal"))?; - self.push( - Operator::scalar(crate::values::Value::Float64(value), DataType::Float64)?, + _ => Err(invalid("PromQL expression has no native fallback lowering")), + } + } + + fn pointwise_projection( + &mut self, + cols: &[planner_types::ir::ProjectItem], + child: &OperatorNode, + value: usize, + output: &OperatorNode, + keep_name: bool, + ) -> Result { + let mut input = self.value(child)?; + let mut projected = cols.to_vec(); + for col in &mut projected { + if col.alias.as_deref() != Some(child.schema.fields[value].name.as_str()) { + continue; + } + if let ScalarExpr::FunctionCall { name, args } = &mut col.expr { + if planner_types::pre_asap::scalar_type_rules::promql_function_arity(name).is_none() + || args.first() != Some(&ScalarExpr::Column(value)) + { + return Err(invalid("unsupported pointwise function")); + } + for arg in args.iter_mut().skip(1) { + let scalar = self.scalar_value(arg)?; + let left = self.schema(&input); + let right = self.schema(&scalar); + let index = left.fields.len(); + let mut schema = (*left).clone(); + schema.fields.extend(right.fields.clone()); + let join = Operator::relational_join( + left, + right, + planner_types::pre_asap::JoinKind::Inner, + &planner_types::ir::Predicate(ScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Boolean(true), + )), + Arc::new(schema), + )?; + input = self.add(join, vec![input, scalar]); + *arg = ScalarExpr::Column(index); + } + if name == "promql_clamp" { + let predicate = ScalarExpr::Not(Box::new(ScalarExpr::Compare { + left: Box::new(args[1].clone()), + right: Box::new(args[2].clone()), + op: planner_types::pre_asap::CompareOpKind::Gt, + semantics: planner_types::ir::ExprSemantics::Promql, + })); + let schema = self.schema(&input); + let predicate = + crate::expressions::CompiledExpression::compile(&predicate, &schema)?; + input = self.add( + Operator::filter( + schema, + crate::expressions::Expression::planner(predicate), + )?, + vec![input], + ); + } + } + } + let schema = self.schema(&input); + let columns = projected + .iter() + .map(|col| { + Ok(( + col.alias.clone().unwrap(), + crate::expressions::Expression::planner( + crate::expressions::CompiledExpression::compile(&col.expr, &schema)?, + ), + )) + }) + .collect::, Error>>()?; + let project = Operator::project(schema, columns)?; + let result = self.push(project, vec![input], output)?; + if keep_name { + Ok(result) + } else { + self.push( + Operator::series_without_name(self.schema(&result))?, + vec![result], + output, + ) + } + } + + fn sample_scalar_operation( + &mut self, + expr: &ScalarExpr, + child: &OperatorNode, + value: usize, + output: &OperatorNode, + ) -> Result { + let (left, right, kind) = scalar_binary(expr)?; + let (scalar, scalar_left) = match (left, right) { + (ScalarExpr::Column(i), scalar) if *i == value => (scalar, false), + (scalar, ScalarExpr::Column(i)) if *i == value => (scalar, true), + _ => { + return Err(invalid( + "sample projection requires one vector sample and one scalar", + )) + } + }; + let vector = self.value(child)?; + let scalar = self.scalar_value(scalar)?; + let sides = if scalar_left { + vec![scalar, vector] + } else { + vec![vector, scalar] + }; + let operator = Operator::series_binary( + self.schema(&sides[0]), + self.schema(&sides[1]), + kernel(kind), + [scalar_left, !scalar_left], + )?; + self.push(operator, sides, output) + } + + fn scalar_value(&mut self, expr: &ScalarExpr) -> Result { + match expr { + ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Float64(value)) => Ok(self + .add( + Operator::scalar(crate::values::Value::Float64(*value), DataType::Float64)?, vec![], - expression, - ) + )), + ScalarExpr::EvalTimestamp => Ok(self.add(Operator::evaluation_time(), vec![])), + ScalarExpr::PromqlScalarFromVector(child) => { + let step = self.value(child)?; + let input = self.schema(&step); + let values: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| f.dtype == FieldDataType::Plain(DataType::Float64)) + .map(|(i, _)| i) + .collect(); + let [value] = values.as_slice() else { + return Err(invalid("scalar() requires one float sample column")); + }; + let value = *value; + Ok(self.add(Operator::vector_to_scalar(input, value)?, vec![step])) + } + ScalarExpr::Negative { expr, .. } => { + let value = self.scalar_value(expr)?; + let minus = self.scalar_value(&ScalarExpr::literal_f64(-1.0))?; + let op = Operator::series_binary( + self.schema(&value), + self.schema(&minus), + kernel(crate::expressions::binary::BinaryOpKind::Arithmetic( + planner_types::pre_asap::ArithmeticOpKind::Mul, + )), + [true, true], + )?; + Ok(self.add(op, vec![value, minus])) + } + _ => { + let (left, right, kind) = scalar_binary(expr)?; + let sides = vec![self.scalar_value(left)?, self.scalar_value(right)?]; + let op = Operator::series_binary( + self.schema(&sides[0]), + self.schema(&sides[1]), + kernel(kind), + [true, true], + )?; + Ok(self.add(op, sides)) } - _ => Err(invalid("PromQL expression has no native fallback lowering")), } } @@ -340,19 +583,19 @@ impl Lowering { fn range_function( &mut self, function: &AggIntent, - matrix: &QueryExpr, - logical: &QueryExpr, + matrix: &OperatorNode, + logical: &OperatorNode, ) -> Result { let function = unbound(function)?; - let (subquery, offset, at_ms) = match matrix { - QueryExpr::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), - other => (other, 0, None), + let (subquery, offset, at_ms) = match matrix.expect_non_asap() { + NonASAPOp::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), + _ => (matrix, 0, None), }; - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range: outer, resolution, child, - } = subquery + } = subquery.expect_non_asap() else { let (range, offset, at) = selector(matrix)?; let input = self.read(matrix)?; @@ -374,8 +617,8 @@ impl Lowering { at_ms, }; // Each step evaluates a per-series selection or range function. - let (inner, selected) = match child.as_ref() { - QueryExpr::Aggregate { + let (inner, selected) = match child.expect_non_asap() { + NonASAPOp::Aggregate { reduction: planner_types::pre_asap::Reduction::PerEntity, measures, having: None, @@ -385,7 +628,7 @@ impl Lowering { [inner] => (Some(unbound(inner)?), selected.as_ref()), _ => return Err(invalid("range function requires one measure")), }, - selected => (None, selected), + _ => (None, child.as_ref()), }; let (range, inner_offset, inner_at) = selector(selected)?; let raw = self.read(selected)?; @@ -425,7 +668,7 @@ impl Lowering { mut step: Input, measure: &AggIntent, keys: &GroupKeys, - logical: &QueryExpr, + logical: &OperatorNode, ) -> Result { let mut input = self.schema(&step); if let AggIntent::HistogramQuantile { q, le } = measure { @@ -447,10 +690,10 @@ impl Lowering { }; let value = *value; let reduction = match measure { - AggIntent::Sum { col: None } => Reduction::Sum(value), - AggIntent::Avg { col: None } => Reduction::Avg(value), - AggIntent::Min { col: None } => Reduction::Min(value), - AggIntent::Max { col: None } => Reduction::Max(value), + AggIntent::Sum { .. } => Reduction::Sum(value), + AggIntent::Avg { .. } => Reduction::Avg(value), + AggIntent::Min { .. } => Reduction::Min(value), + AggIntent::Max { .. } => Reduction::Max(value), AggIntent::Count { .. } => Reduction::Count, _ => return Err(invalid("vector aggregate has no native lowering")), }; @@ -527,10 +770,10 @@ fn unbound(intent: &AggIntent) -> Result, Error> { AggIntent::Count { accuracy } => AggIntent::Count { accuracy: accuracy.clone(), }, - AggIntent::Sum { col: None } => AggIntent::Sum { col: None }, - AggIntent::Avg { col: None } => AggIntent::Avg { col: None }, - AggIntent::Min { col: None } => AggIntent::Min { col: None }, - AggIntent::Max { col: None } => AggIntent::Max { col: None }, + AggIntent::Sum { .. } => AggIntent::Sum { col: None }, + AggIntent::Avg { .. } => AggIntent::Avg { col: None }, + AggIntent::Min { .. } => AggIntent::Min { col: None }, + AggIntent::Max { .. } => AggIntent::Max { col: None }, AggIntent::IRate => AggIntent::IRate, AggIntent::IDelta => AggIntent::IDelta, AggIntent::Changes => AggIntent::Changes, @@ -548,3 +791,65 @@ fn unbound(intent: &AggIntent) -> Result, Error> { _ => return Err(invalid("unsupported PromQL range function")), }) } + +fn kernel( + kind: crate::expressions::binary::BinaryOpKind, +) -> crate::expressions::binary::BinaryOperator { + crate::expressions::binary::BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + } +} + +fn scalar_binary( + expr: &ScalarExpr, +) -> Result< + ( + &ScalarExpr, + &ScalarExpr, + crate::expressions::binary::BinaryOpKind, + ), + Error, +> { + use crate::expressions::binary::BinaryOpKind as K; + match expr { + ScalarExpr::Arithmetic { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + } => Ok((left, right, K::Arithmetic(op.clone()))), + ScalarExpr::Compare { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + } => Ok((left, right, K::Compare(op.clone()))), + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } if matches!(else_expr.as_deref(), Some(ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Float64(v))) if *v == 0.0) => + { + let [( + ScalarExpr::Compare { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + }, + ScalarExpr::Literal(planner_types::pre_asap::ScalarValue::Float64(v)), + )] = branches.as_slice() + else { + return Err(invalid("unsupported scalar case")); + }; + if *v != 1.0 { + return Err(invalid("unsupported scalar case result")); + } + Ok((left, right, K::CompareBool(op.clone()))) + } + _ => Err(invalid("scalar expression has no native temporal lowering")), + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index f3082927b..81b790b62 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -1,6 +1,12 @@ //! A bounded PromQL source row carries the entire label set, not just labels //! mentioned by the query. The source adapter owns this lossless encoding. use super::*; +use planner_types::ir::physical_export::{ + compile_physical_asap_dag, compile_physical_asap_dag_with_node_ids, +}; +use planner_types::ir::ASAPOp; +use planner_types::ir::NonASAPOp; +use planner_types::post_asap::FieldDataType as SummaryFamilyType; use planner_types::pre_asap::DataType; use std::rc::Rc; @@ -23,9 +29,9 @@ pub fn decode_series_identity(encoded: &str) -> Result, } /// Resolve the row representation before candidate search; see -/// [`planner_types::pre_asap::schema::with_promql_series_identity`]. -pub fn with_series_identity(root: &QueryExpr) -> Result { - planner_types::pre_asap::schema::with_promql_series_identity(root).map_err(invalid) +/// [`planner_types::ir::schema_support::with_promql_series_identity`]. +pub fn with_series_identity(root: &Rc) -> Result, Error> { + planner_types::ir::schema_support::with_promql_series_identity(root).map_err(invalid) } /// Construct source rows only from full identities. The named label columns @@ -45,7 +51,10 @@ pub fn series_row( .enumerate() .map(|(index, field)| { if field.name == SERIES_IDENTITY_COLUMN { - if field.dtype != FieldDataType::Plain(DataType::Utf8) || field.nullable || found { + if field.dtype != SummaryFamilyType::Plain(DataType::Utf8) + || field.nullable + || found + { return Err(invalid("invalid series identity column")); } found = true; @@ -53,10 +62,10 @@ pub fn series_row( } else if Some(index) == schema.time_index { Ok(Value::Timestamp(timestamp)) } else if field.name == "value" - && field.dtype == FieldDataType::Plain(DataType::Float64) + && field.dtype == SummaryFamilyType::Plain(DataType::Float64) { Ok(Value::Float64(value)) - } else if field.dtype == FieldDataType::Plain(DataType::Utf8) { + } else if field.dtype == SummaryFamilyType::Plain(DataType::Utf8) { Ok(labels.get(&field.name).map_or_else( || Value::Utf8("".into()), |value| Value::Utf8(value.clone().into()), @@ -75,17 +84,24 @@ pub fn series_row( /// Compile the selected TopK computation above an existing maintained-population /// source. The boundary supplies the complete eligible vector, not a truncated /// TopK result; ranking remains a native physical operator. -pub fn compile_current_series_readout( - selected: &Rc, +pub fn compile_current_series_evaluation( + selected: &Rc, ) -> Result { use planner_types::post_asap::{ - compile_post_asap_dag, maintained_population::PopulationStatistic, Field, + maintained_population::PopulationStatistic, Field as SummaryField, }; - let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; + let selected = planner_types::ir::apply_lifecycle_timings( + selected, + &planner_types::ir::LifecycleAssignment::default_maintained(), + &mut planner_types::ir::TimingMemo::new(), + ) + .map_err(|e| invalid(e.to_string()))?; + let mut dag = + compile_physical_asap_dag(&selected).map_err(|error| invalid(error.to_string()))?; // Typed snapshot candidates already carry full identity throughout the DAG. - // Cut at the population output, preserving all selected heap/readout nodes. + // Cut at the population output, preserving all selected heap/evaluation nodes. let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, - Payload::Value { operation: ValueOperation::MaintainPopulation { population } } + Payload::ASAP(ASAPOp::MaintainPopulation { population, .. }) if matches!(population.input, planner_types::post_asap::maintained_population::PopulationInput::CurrentSeries(_)) )).collect::>(); if let [population] = populations.as_slice() { @@ -98,48 +114,34 @@ pub fn compile_current_series_readout( return compile( &dag, BTreeMap::from([( - u64::from(population.id.0), + population.id as u64, InputContract::bounded(Arc::new(population.output_schema.clone())), )]), - &[u64::from(dag.root.0)], + &dag.roots.iter().map(|r| *r as u64).collect::>(), ); } } - if dag.nodes.len() != 3 - || !dag.nodes.iter().any(|node| { - node.id == dag.root - && matches!( - node.payload, - Payload::Value { - operation: ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { .. } - } - } - ) - }) - { - return Err(invalid( - "expected one selected current-series TopK computation", - )); - } let mut frontier = None; for node in &mut dag.nodes { match &mut node.payload { - Payload::Fallback { expression } => { - *expression = with_series_identity(expression)?; + Payload::NonASAP(operator) => { + if let NonASAPOp::Scan { schema, .. } = operator { + schema.fields.push(SummaryField::new( + SERIES_IDENTITY_COLUMN, + SummaryFamilyType::Plain(DataType::Utf8), + false, + )); + schema.closed = true; + } } - Payload::Value { - operation: ValueOperation::MaintainPopulation { .. }, - } => { - frontier = Some(u64::from(node.id.0)); + Payload::ASAP(ASAPOp::MaintainPopulation { .. }) => { + frontier = Some(node.id as u64); } - Payload::Value { - operation: - ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { .. }, - }, - } => {} - _ => return Err(invalid("unsupported current-series readout dependency")), + Payload::ASAP(ASAPOp::EvaluatePopulation { + evaluation: PopulationStatistic::TopK { .. }, + .. + }) => {} + _ => return Err(invalid("unsupported current-series evaluation dependency")), } if node .output_schema @@ -151,11 +153,11 @@ pub fn compile_current_series_readout( "current-series input already has a physical identity column", )); } - node.output_schema.fields.push(Field { - table: None, + node.output_schema.fields.push(SummaryField { name: SERIES_IDENTITY_COLUMN.into(), - dtype: FieldDataType::Plain(DataType::Utf8), + dtype: SummaryFamilyType::Plain(DataType::Utf8), nullable: false, + table: None, }); } for edge in &mut dag.edges { @@ -171,7 +173,7 @@ pub fn compile_current_series_readout( let schema = Arc::new( dag.nodes .iter() - .find(|node| u64::from(node.id.0) == frontier) + .find(|node| node.id as u64 == frontier) .unwrap() .output_schema .clone(), @@ -179,48 +181,36 @@ pub fn compile_current_series_readout( compile( &dag, BTreeMap::from([(frontier, InputContract::bounded(schema))]), - &[u64::from(dag.root.0)], + &dag.roots.iter().map(|r| *r as u64).collect::>(), ) } /// Compile selected ranking or aggregation above an exact per-series Rate -/// readout. Deployments bind complete window readouts at this boundary; +/// evaluation. Deployments bind complete window evaluations at this boundary; /// the heap is rebuilt independently for each evaluation. This does not move /// that frontier to ingestion time or authorize combining finalized rates. pub fn compile_rate_ranking( - selected: &Rc, -) -> Result< - ( - Rc, - CompiledPhysicalDAG, - ), - Error, -> { - use planner_types::post_asap::{ - compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, - }; - fn frontier(node: &Rc) -> Option> { - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: planner_types::post_asap::ExecutionTiming::QueryTime, - } if matches!(&child.expr, SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Rate, _), - reduction: planner_types::pre_asap::Reduction::PerEntity, - child: raw, .. - } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => - { - Some(Rc::clone(node)) - } - SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { - frontier(child) - } - SummaryExpr::SummaryEstimate { summary_input, .. } => frontier(summary_input), - _ => None, + selected: &Rc, +) -> Result<(Rc, CompiledPhysicalDAG), Error> { + use planner_types::post_asap::ExactKind; + fn frontier(node: &Rc) -> Option> { + if matches!(&node.operator, LogicalOperator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!(&child.operator, LogicalOperator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, child: raw, .. + }) if matches!(raw.non_asap(), Some(NonASAPOp::TimeRange { .. })))) + { + return Some(Rc::clone(node)); } + node.children().into_iter().find_map(frontier) } - let source = frontier(selected) + let selected = planner_types::ir::apply_lifecycle_timings( + selected, + &planner_types::ir::LifecycleAssignment::default_maintained(), + &mut planner_types::ir::TimingMemo::new(), + ) + .map_err(|e| invalid(e.to_string()))?; + let source = frontier(&selected) .ok_or_else(|| invalid("ranking requires one exact per-series Rate frontier"))?; if !source .schema @@ -230,29 +220,31 @@ pub fn compile_rate_ranking( { return Err(invalid("Rate ranking requires complete series identity")); } - let compiled = compile_post_asap_dag_with_node_ids(selected) + let compiled = compile_physical_asap_dag_with_node_ids(&selected) .map_err(|error| invalid(error.to_string()))?; - let id = u64::from( - compiled - .node_ids - .node_id(&source) - .ok_or_else(|| invalid("missing Rate frontier"))? - .0, - ); + let id = compiled + .node_ids + .node_id(&source) + .ok_or_else(|| invalid("missing Rate frontier"))? as u64; let program = compile( &compiled.dag, BTreeMap::from([(id, InputContract::bounded(Arc::new(source.schema.clone())))]), - &[u64::from(compiled.dag.root.0)], + &compiled + .dag + .roots + .iter() + .map(|r| *r as u64) + .collect::>(), )?; Ok((source, program)) } /// Compile a lifecycle-timed DAG whose heap or grouped Sum over per-series -/// Rate readouts runs at ingestion time: fresh aggregate state per closed +/// Rate evaluations runs at ingestion time: fresh aggregate state per closed /// window. The input is the complete collection of per-series counter states. pub fn compile_fixed_window_rate_aggregation( - dag: &planner_types::post_asap::PostAsapDAG, -) -> Result { + dag: &planner_types::ir::physical_export::PhysicalASAPDAG, +) -> Result { use planner_types::post_asap::{ExactKind, ExecutionTiming, SketchAlgorithm}; let sources = dag .nodes @@ -260,11 +252,11 @@ pub fn compile_fixed_window_rate_aggregation( .filter(|n| { matches!( &n.payload, - Payload::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Rate, _), + Payload::ASAP(ASAPOp::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), reduction: planner_types::pre_asap::Reduction::PerEntity, .. - } + }) ) }) .collect::>(); @@ -274,17 +266,17 @@ pub fn compile_fixed_window_rate_aggregation( .filter(|n| { n.output_state.timing == ExecutionTiming::IngestionTime && match &n.payload { - Payload::SummaryAgg { - family: FieldDataType::Sketch(kind, _), + Payload::ASAP(ASAPOp::SummaryAgg { + family: SummaryFamilyType::Sketch(kind, _), .. - } => matches!( + }) => matches!( kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap ), - Payload::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Sum, _), + Payload::ASAP(ASAPOp::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. - } => true, + }) => true, _ => false, } }) @@ -307,10 +299,10 @@ pub fn compile_fixed_window_rate_aggregation( compile_candidate( dag, BTreeMap::from([( - u64::from(source.id.0), + source.id as u64, InputContract::bounded(Arc::new(source.output_schema.clone())), )]), - &[u64::from(dag.root.0)], - &[u64::from(heap.id.0)], + &dag.roots.iter().map(|r| *r as u64).collect::>(), + &[heap.id as u64], ) } diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 880916ed2..a8a091249 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -1,5 +1,6 @@ //! Physical scalar/vector contracts preserve complete label sets across native computation. use super::*; +use planner_types::post_asap::FieldDataType as SummaryFamilyType; pub fn scalar_schema() -> SchemaRef { crate::operators::vector_binary::value_schema(true) @@ -63,7 +64,7 @@ pub fn compile_histogram_quantile() -> Result { /// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. pub fn compile_binary( - operator: &planner_types::post_asap::BinaryOperator, + operator: &crate::expressions::binary::BinaryOperator, return_bool: bool, left_scalar: bool, right_scalar: bool, @@ -209,8 +210,8 @@ pub fn compile_vector_to_scalar() -> Result { /// A stored exact-state input retains the complete population identity. The /// deployment supplies eligible panes; merging and finalization are computation. -pub fn exact_state_schema(family: FieldDataType) -> Result { - if !matches!(family, FieldDataType::ExactAggregate(..)) { +pub fn exact_state_schema(family: SummaryFamilyType) -> Result { + if !matches!(family, SummaryFamilyType::ExactAggregate(..)) { return Err(invalid("exact-state input requires an exact family")); } crate::values::validate_family(&family)?; @@ -219,31 +220,33 @@ pub fn exact_state_schema(family: FieldDataType) -> Result { Ok(Arc::new(schema)) } -/// Retain exact readout semantics before any deployment state is opened. -pub fn compile_exact_readout( - family: FieldDataType, +/// Retain exact evaluation semantics before any deployment state is opened. +pub fn compile_exact_evaluation( + family: SummaryFamilyType, lookback_ms: u64, preserve_metric_name: bool, ) -> Result { use planner_types::post_asap::ExactKind; let statistic = match &family { - FieldDataType::ExactAggregate(kind, _) => match kind { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { ExactKind::Sum => crate::Statistic::Sum, ExactKind::Count => crate::Statistic::Count, ExactKind::Min => crate::Statistic::Min, ExactKind::Max => crate::Statistic::Max, ExactKind::Rate => crate::Statistic::Rate, ExactKind::Increase => crate::Statistic::Increase, - ExactKind::IRate => return Err(invalid("instant-rate state readout is not supported")), + ExactKind::IRate => { + return Err(invalid("instant-rate state evaluation is not supported")) + } }, - _ => return Err(invalid("exact readout requires an exact family")), + _ => return Err(invalid("exact evaluation requires an exact family")), }; let input = exact_state_schema(family)?; let merge = Operator::summary_merge(input.clone(), 1, vec![0])?; - let mut readout = Operator::readout( + let mut evaluation = Operator::evaluation( merge.schema(), 1, - ReadoutQuery::Exact(ExactReadout { + SummaryEvaluation::Exact(ExactEvaluation { statistic, lookback_ms: None, }), @@ -252,12 +255,12 @@ pub fn compile_exact_readout( statistic, crate::Statistic::Rate | crate::Statistic::Increase ) { - readout = readout.with_counter_lookback( + evaluation = evaluation.with_counter_lookback( i64::try_from(lookback_ms).map_err(|_| invalid("counter lookback exceeds Int64"))?, )?; } let project = Operator::project( - readout.schema(), + evaluation.schema(), vec![ ( "labels".into(), @@ -274,5 +277,5 @@ pub fn compile_exact_readout( ("value".into(), Expression::ExactFloat64(1)), ], )?; - unary(vec![merge, readout, project], input) + unary(vec![merge, evaluation, project], input) } diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/asap-physical-operators/src/physical_planner/row_values.rs index 3396df685..14f3d4813 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/row_values.rs @@ -1,29 +1,20 @@ //! Query-time PromQL value computation over logical row schemas. use super::*; use planner_types::post_asap::maintained_population::PopulationStatistic; -use planner_types::pre_asap::{DataType, ScalarValue}; +use planner_types::pre_asap::DataType; -/// A PromQL number literal has no row schema; its consumer folds it in. -pub(super) fn scalar_literal(expression: &QueryExpr) -> Option { - match expression { - QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), - QueryExpr::Literal(ScalarValue::Float64(value)) => Some(*value), - _ => None, - } -} - -/// Aggregate readouts of a maintained current-series population, as a chain. +/// Aggregate evaluations of a maintained current-series population, as a chain. pub(super) fn population_aggregate( input: &SchemaRef, grouping: &[String], - readout: &PopulationStatistic, + evaluation: &PopulationStatistic, ) -> Result, Error> { let groups = grouping .iter() .map(|name| named_column(input, &ColumnRef::Named(name.clone()))) .collect::, _>>()?; let value = named_column(input, &ColumnRef::SampleValue)?; - let reduction = match readout { + let reduction = match evaluation { PopulationStatistic::Sum => Reduction::Sum(value), PopulationStatistic::Count => Reduction::Count, PopulationStatistic::Average => Reduction::Avg(value), @@ -33,7 +24,7 @@ pub(super) fn population_aggregate( }, PopulationStatistic::TopK { .. } => { return Err(invalid( - "TopK population readout ranks; it does not aggregate", + "TopK population evaluation ranks; it does not aggregate", )) } }; diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index ca59877cb..733741022 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -88,7 +88,7 @@ mod tests { values::Value, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, + post_asap::{Field as SummaryField, FieldDataType as SummaryFamilyType, Schema}, pre_asap::DataType, }; use std::sync::Arc; @@ -97,15 +97,15 @@ mod tests { #[test] fn same_native_chain_inside_query_and_ingestion_execution() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], - fields: vec![Field { - table: None, + fields: vec![SummaryField { name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), + dtype: SummaryFamilyType::Plain(DataType::Float64), nullable: false, + table: None, }], time_index: None, + unique_keys: vec![], + closed: false, }); for scope in [ Scope::Query { @@ -143,10 +143,10 @@ mod tests { #[test] fn in_memory_source_drives_cooperative_yields() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); let source = Operator::source(schema, vec![batch; 65]).unwrap(); @@ -165,10 +165,10 @@ mod tests { #[test] fn returned_batches_keep_their_resource_reservation() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); let bytes = batch.bytes(); @@ -196,10 +196,10 @@ mod tests { #[test] fn cancellation_is_not_bypassed_by_in_memory_execution() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema, vec![vec![]]).unwrap(); let context = RunContext::new( diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/asap-physical-operators/src/sources/memory.rs index 1055888de..856c73cb0 100644 --- a/crates/asap-physical-operators/src/sources/memory.rs +++ b/crates/asap-physical-operators/src/sources/memory.rs @@ -11,7 +11,7 @@ impl MemorySource { if schema .fields .iter() - .any(|f| !matches!(f.dtype, FieldDataType::Plain(_))) + .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) { return Err(Error::Invalid( "raw source cannot contain summary states".into(), diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs index 4b401d666..9713c2a6f 100644 --- a/crates/asap-physical-operators/src/sources/mod.rs +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -7,9 +7,10 @@ use crate::{ Error, }; use futures::{stream, StreamExt}; +use planner_types::ir::{NonASAPOp, OperatorNode}; use planner_types::{ - post_asap::{FieldDataType, Schema}, - pre_asap::{DataType, QueryExpr, Source}, + post_asap::FieldDataType as SummaryFamilyType, + pre_asap::{DataType, Source}, }; use std::sync::Arc; @@ -40,18 +41,18 @@ impl DataSources { self.sources.push((identity, source)); Ok(()) } - pub fn bind(&self, expression: &QueryExpr) -> Result { - let QueryExpr::Scan { + pub fn bind(&self, expression: &OperatorNode) -> Result { + let Some(NonASAPOp::Scan { source, predicates, schema, - } = expression + }) = expression.non_asap() else { return Err(Error::Invalid( "raw Scan requires a Planner Scan leaf".into(), )); }; - let output = Arc::new(Schema::lifted(schema.fields.clone(), schema.time_index)); + let output = Arc::new(schema.clone()); crate::values::validate_schema(&output)?; let reader = self .sources diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs index d4754686e..d5375f9bf 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -2,7 +2,7 @@ use super::increase::IncreaseAccumulator; use crate::Statistic; use crate::{AggregateCore, KeyByLabelValues, Measurement}; -use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType}; +use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType as SummaryFamilyType}; use serde::{Deserialize, Serialize}; use std::collections::HashMap; @@ -10,7 +10,11 @@ type Error = Box; #[derive(Debug, Clone, Serialize, Deserialize)] enum ScalarState { - Sum { sum: f64, compensation: f64 }, + Sum { + sum: f64, + compensation: f64, + seen: bool, + }, Count(u64), Min(Option), Max(Option), @@ -18,7 +22,7 @@ enum ScalarState { } /// Both the family and population layout survive persistence. Sharing counter -/// arithmetic never authorizes a Rate state to answer an Increase readout. +/// arithmetic never authorizes a Rate state to answer an Increase evaluation. /// /// Deserialization validates the payload against its declared family, so /// deployments can persist this state with any serde format without mirroring @@ -26,14 +30,14 @@ enum ScalarState { #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(try_from = "ExactPayload")] pub struct ExactAccumulator { - family: FieldDataType, + family: SummaryFamilyType, scalar: ScalarState, keyed: Option>, } #[derive(Deserialize)] struct ExactPayload { - family: FieldDataType, + family: SummaryFamilyType, scalar: ScalarState, keyed: Option>, } @@ -63,10 +67,10 @@ impl TryFrom for ExactAccumulator { } } -/// Planned readout of an exact summary. `lookback_ms` is the logical PromQL +/// Planned evaluation of an exact summary. `lookback_ms` is the logical PromQL /// counter window; the evaluation range is resolved from it at run time. #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -pub struct ExactReadout { +pub struct ExactEvaluation { pub statistic: Statistic, #[serde(default, skip_serializing_if = "Option::is_none")] pub lookback_ms: Option, @@ -75,22 +79,24 @@ pub struct ExactReadout { impl ExactAccumulator { /// Read one population. An empty MIN/MAX population reads as `None`. /// `range_ms` extrapolates a counter Rate/Increase to that evaluation range. - pub fn readout( + pub fn evaluation( &self, statistic: Statistic, range_ms: Option<(i64, i64)>, key: Option<&KeyByLabelValues>, ) -> Result, Error> { if statistic != self.statistic() { - return Err("readout differs from Planner exact family".into()); + return Err("evaluation differs from Planner exact family".into()); } let state = match (&self.keyed, key) { (Some(states), Some(key)) => states.get(key).ok_or("unknown exact population")?, (None, None) => &self.scalar, - _ => return Err("readout population differs from installed layout".into()), + _ => return Err("evaluation population differs from installed layout".into()), }; match state { - ScalarState::Sum { sum, compensation } => Ok(Some(sum + compensation)), + ScalarState::Sum { + sum, compensation, .. + } => Ok(Some(sum + compensation)), ScalarState::Count(count) => Ok(Some(*count as f64)), ScalarState::Min(value) | ScalarState::Max(value) => Ok(*value), ScalarState::Counter(Some(counter)) => counter @@ -100,6 +106,11 @@ impl ExactAccumulator { } } + /// SQL SUM distinguishes an empty/all-NULL input from an observed zero. + pub(crate) fn is_empty_sum(&self) -> bool { + self.keyed.is_none() && matches!(self.scalar, ScalarState::Sum { seen: false, .. }) + } + /// Exact integer count of an unkeyed Count state. pub fn count(&self) -> Option { match (&self.keyed, &self.scalar) { @@ -128,19 +139,22 @@ impl ExactAccumulator { Ok(()) } - pub fn new(family: FieldDataType, keyed: bool) -> Result { + pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { use ExactKind as K; use ExactParams as P; let scalar = match &family { - FieldDataType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum { + SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, - FieldDataType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), - FieldDataType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), - FieldDataType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), - FieldDataType::ExactAggregate(K::Rate, P::Rate) - | FieldDataType::ExactAggregate(K::Increase, P::Increase) => ScalarState::Counter(None), + SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), + SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), + SummaryFamilyType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), + SummaryFamilyType::ExactAggregate(K::Rate, P::Rate) + | SummaryFamilyType::ExactAggregate(K::Increase, P::Increase) => { + ScalarState::Counter(None) + } _ => return Err(format!("unsupported exact Planner family: {family:?}")), }; Ok(Self { @@ -150,7 +164,7 @@ impl ExactAccumulator { }) } - pub fn family(&self) -> &FieldDataType { + pub fn family(&self) -> &SummaryFamilyType { &self.family } pub(crate) fn insufficient_counter_samples( @@ -188,7 +202,14 @@ impl ExactAccumulator { _ => panic!("exact update population layout differs from installed DAG"), }; match state { - ScalarState::Sum { sum, compensation } => compensated_add(sum, compensation, value), + ScalarState::Sum { + sum, + compensation, + seen, + } => { + compensated_add(sum, compensation, value); + *seen = true; + } ScalarState::Count(count) => { *count = count.checked_add(1).expect("exact count overflow") } @@ -214,12 +235,12 @@ impl ExactAccumulator { fn statistic(&self) -> Statistic { match self.family { - FieldDataType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, - FieldDataType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, - FieldDataType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, - FieldDataType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, - FieldDataType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, - FieldDataType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, + SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, + SummaryFamilyType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, + SummaryFamilyType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, + SummaryFamilyType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, + SummaryFamilyType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, + SummaryFamilyType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, _ => unreachable!("validated exact family"), } } @@ -244,16 +265,22 @@ fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result { let (mut sum, mut compensation) = (*a, *ac); compensated_add(&mut sum, &mut compensation, *b); compensated_add(&mut sum, &mut compensation, *bc); - ScalarState::Sum { sum, compensation } + ScalarState::Sum { + sum, + compensation, + seen: *a_seen || *b_seen, + } } (ScalarState::Count(a), ScalarState::Count(b)) => { ScalarState::Count(a.checked_add(*b).ok_or("exact count overflow")?) @@ -311,7 +338,7 @@ mod tests { #[derive(Serialize)] struct Payload { - family: FieldDataType, + family: SummaryFamilyType, scalar: ScalarState, keyed: Option>, } @@ -320,8 +347,8 @@ mod tests { rmp_serde::from_slice(&rmp_serde::to_vec_named(payload).unwrap()) } - fn sum() -> FieldDataType { - FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + fn sum() -> SummaryFamilyType { + SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) } // Stored Sum preserves low-order increments across updates, persistence and pane merge. @@ -337,7 +364,7 @@ mod tests { negative.update(None, -1e16, 1); restored.merge_from(&negative).unwrap(); assert_eq!( - restored.readout(Statistic::Sum, None, None).unwrap(), + restored.evaluation(Statistic::Sum, None, None).unwrap(), Some(1.0) ); } @@ -349,18 +376,18 @@ mod tests { state.update(None, f64::INFINITY, 0); state.update(None, 1.0, 0); assert_eq!( - state.readout(Statistic::Sum, None, None).unwrap(), + state.evaluation(Statistic::Sum, None, None).unwrap(), Some(f64::INFINITY) ); state.update(None, f64::NEG_INFINITY, 0); assert!(state - .readout(Statistic::Sum, None, None) + .evaluation(Statistic::Sum, None, None) .unwrap() .unwrap() .is_nan()); } - // A persisted exact state decodes back to the same family, layout and readout. + // A persisted exact state decodes back to the same family, layout and evaluation. #[test] fn serialized_state_round_trips() { let mut state = ExactAccumulator::new(sum(), true).unwrap(); @@ -370,7 +397,9 @@ mod tests { let restored: ExactAccumulator = rmp_serde::from_slice(&bytes).unwrap(); assert_eq!(restored.family(), &sum()); assert_eq!( - restored.readout(Statistic::Sum, None, Some(&key)).unwrap(), + restored + .evaluation(Statistic::Sum, None, Some(&key)) + .unwrap(), Some(2.5) ); } @@ -390,6 +419,7 @@ mod tests { scalar: ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, keyed: Some(HashMap::from([(key, ScalarState::Max(Some(1.0)))])), }; @@ -400,10 +430,11 @@ mod tests { #[test] fn decode_rejects_unsupported_family() { let payload = Payload { - family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Count), + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Count), scalar: ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, keyed: None, }; diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs index 4be126d6e..be819f8a7 100644 --- a/crates/asap-physical-operators/src/summary_kernels/factory.rs +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -6,7 +6,7 @@ use crate::summary_kernels::{ HydraKllSketchAccumulator, }; use crate::{AggregateCore, KeyByLabelValues}; -use planner_types::post_asap::{FieldDataType, SketchAlgorithm, SketchParams}; +use planner_types::post_asap::{FieldDataType as SummaryFamilyType, SketchAlgorithm, SketchParams}; /// Generate the clone-based `AccumulatorUpdater` methods for updaters whose /// inner `acc` field implements `Clone + AggregateCore`. @@ -527,7 +527,7 @@ fn cms_heap_dims(params: &SketchParams) -> (usize, usize, usize) { /// Construct the kernel declared by a Planner SummaryAgg. No deployment config /// tags participate in this dispatch and unsupported payloads are errors. pub fn create_planner_accumulator( - family: &FieldDataType, + family: &SummaryFamilyType, input: &planner_types::post_asap::SummaryUpdate, grouping: &planner_types::post_asap::GroupingStrategy, ) -> Result, String> { @@ -548,7 +548,7 @@ pub fn create_planner_accumulator( if grouping != &GroupingStrategy::PerSubpopulationInstance { return Err("shared summary grouping requires a supported Planner Hydra kernel".into()); } - if matches!(family, FieldDataType::ExactAggregate(..)) { + if matches!(family, SummaryFamilyType::ExactAggregate(..)) { return Ok(Box::new(PlannerExactUpdater { acc: crate::summary_kernels::exact::ExactAccumulator::new( family.clone(), @@ -556,7 +556,7 @@ pub fn create_planner_accumulator( )?, })); } - let FieldDataType::Sketch(kind, family_grouping) = family else { + let SummaryFamilyType::Sketch(kind, family_grouping) = family else { return Err(format!("unsupported Planner summary family {family:?}")); }; if family_grouping != grouping { @@ -748,7 +748,7 @@ mod planner_parameter_regression { }, ), ] { - let family = FieldDataType::Sketch( + let family = SummaryFamilyType::Sketch( SketchKind::new(algorithm.clone(), params), Default::default(), ); diff --git a/crates/asap-physical-operators/src/summary_kernels/traits.rs b/crates/asap-physical-operators/src/summary_kernels/traits.rs index 9c5d028fb..46e2c7f86 100644 --- a/crates/asap-physical-operators/src/summary_kernels/traits.rs +++ b/crates/asap-physical-operators/src/summary_kernels/traits.rs @@ -5,7 +5,7 @@ pub type KernelError = Box; /// In-memory state of one population's summary. /// /// Kernels adapt `asap_sketchlib` structures (or exact Planner state) to the -/// operations physical operators need: merge, typed readout and memory +/// operations physical operators need: merge, typed evaluation and memory /// accounting. Grouping belongs to operators; byte encodings belong to /// `asap_sketchlib` and deployments. pub trait AggregateCore: Send + Sync { @@ -20,8 +20,8 @@ pub trait AggregateCore: Send + Sync { /// Merge with a state of the same family and shape, leaving both inputs unchanged. fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError>; - /// Answer a sketch readout. Exact states are read through - /// [`ExactAccumulator::readout`](super::exact::ExactAccumulator::readout). + /// Answer a sketch evaluation. Exact states are read through + /// [`ExactAccumulator::evaluation`](super::exact::ExactAccumulator::evaluation). fn estimate(&self, query: &SketchStatistic) -> Result { Err(format!("{query:?} is not supported by this summary").into()) } diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs index 9340b1c0e..23114ab1b 100644 --- a/crates/asap-physical-operators/src/summary_kernels/univmon.rs +++ b/crates/asap-physical-operators/src/summary_kernels/univmon.rs @@ -1,4 +1,4 @@ -//! One frequency state shared by count, distinct, L2 and entropy readouts. +//! One frequency state shared by count, distinct, L2 and entropy evaluations. use crate::AggregateCore; use asap_sketchlib::{DataInput, UnivMon}; @@ -169,10 +169,10 @@ mod tests { } } - // Count, distinct, L2 and entropy readouts count each non-NaN sample once; + // Count, distinct, L2 and entropy evaluations count each non-NaN sample once; // signed zero is one identity. #[test] - fn frequency_readouts() { + fn frequency_evaluations() { let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); for value in [0.0, -0.0, 2.0, 2.0, f64::NAN] { state.insert_sample(value).unwrap(); @@ -187,9 +187,9 @@ mod tests { .is_err()); } - // A sketch taken out and adopted back answers the same readouts. + // A sketch taken out and adopted back answers the same evaluations. #[test] - fn adopted_sketch_keeps_readouts() { + fn adopted_sketch_keeps_evaluations() { let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); for value in [1.0, 2.0, 2.0] { state.insert_sample(value).unwrap(); diff --git a/crates/asap-physical-operators/src/unified_physical_planner/mod.rs b/crates/asap-physical-operators/src/unified_physical_planner/mod.rs index ad036e1a7..251d5af3f 100644 --- a/crates/asap-physical-operators/src/unified_physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/unified_physical_planner/mod.rs @@ -10,8 +10,7 @@ use crate::{ }; use planner_types::ir::physical_export::{ PhysicalASAPDAG, PhysicalASAPDAGNode, PhysicalASAPNodeId, - PhysicalASAPOperatorPayload as Payload, -}; + PhysicalASAPOperatorPayload as Payload}; use planner_types::ir::{ASAPOp, NonASAPOp, Operator as LogicalOperator, OperatorNode, ScalarExpr}; use planner_types::{ post_asap::{FieldDataType, SketchStatistic, SummaryInputExpr}, diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs index b1186c171..41640b7b6 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/asap-physical-operators/src/values.rs @@ -2,7 +2,7 @@ use crate::AggregateCore; use crate::Error; use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, + post_asap::{Field as SummaryField, FieldDataType as SummaryFamilyType, Schema}, pre_asap::DataType, }; use std::{cmp::Ordering, sync::Arc}; @@ -28,7 +28,7 @@ pub enum Value { Map(Arc<[(Value, Value)]>), #[serde(skip)] Summary { - family: FieldDataType, + family: SummaryFamilyType, state: Arc, }, } @@ -214,7 +214,9 @@ impl Batch { } for (value, field) in row.iter().zip(&schema.fields) { let matches = match (&field.dtype, value) { - (FieldDataType::Plain(dtype), value) => value.matches(dtype, field.nullable), + (SummaryFamilyType::Plain(dtype), value) => { + value.matches(dtype, field.nullable) + } (expected, Value::Summary { family, state }) => { expected == family && validate_state(family, state.as_ref()).is_ok() } @@ -260,7 +262,7 @@ pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result> pub(crate) use crate::capability::validate_native_family as validate_family; -fn validate_state(family: &FieldDataType, state: &dyn AggregateCore) -> Result<(), Error> { +fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { use crate::summary_kernels::{ count_min_sketch::CountMinSketchAccumulator, datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, @@ -268,7 +270,7 @@ fn validate_state(family: &FieldDataType, state: &dyn AggregateCore) -> Result<( use planner_types::post_asap::SketchParams; validate_family(family)?; let valid = match family { - FieldDataType::Sketch(kind, _) + SummaryFamilyType::Sketch(kind, _) if matches!( kind.params(), SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } @@ -284,11 +286,11 @@ fn validate_state(family: &FieldDataType, state: &dyn AggregateCore) -> Result<( }) } - FieldDataType::ExactAggregate(..) => state + SummaryFamilyType::ExactAggregate(..) => state .as_any() .downcast_ref::() .is_some_and(|s| s.family() == family && !s.is_keyed()), - FieldDataType::Sketch(kind, _) => match kind.params() { + SummaryFamilyType::Sketch(kind, _) => match kind.params() { SketchParams::Kll { k } => state .as_any() .downcast_ref::() @@ -325,14 +327,14 @@ pub(crate) fn validate_schema(schema: &SchemaRef) -> Result<(), Error> { schema .fields .get(index) - .is_none_or(|field| field.dtype != FieldDataType::Plain(DataType::Timestamp)) + .is_none_or(|field| field.dtype != SummaryFamilyType::Plain(DataType::Timestamp)) }) { return Err(Error::Invalid( "time index must name a Timestamp column".into(), )); } for field in &schema.fields { - if !matches!(field.dtype, FieldDataType::Plain(_)) { + if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { validate_family(&field.dtype)?; if field.nullable { return Err(Error::Invalid( @@ -344,7 +346,7 @@ pub(crate) fn validate_schema(schema: &SchemaRef) -> Result<(), Error> { Ok(()) } -pub(crate) fn field(schema: &SchemaRef, column: usize) -> Result<&Field, Error> { +pub(crate) fn field(schema: &SchemaRef, column: usize) -> Result<&SummaryField, Error> { schema .fields .get(column) @@ -352,7 +354,7 @@ pub(crate) fn field(schema: &SchemaRef, column: usize) -> Result<&Field, Error> } pub(crate) fn plain(schema: &SchemaRef, column: usize) -> Result<(&DataType, bool), Error> { let f = field(schema, column)?; - let FieldDataType::Plain(dtype) = &f.dtype else { + let SummaryFamilyType::Plain(dtype) = &f.dtype else { return Err(Error::Invalid("plain value required".into())); }; Ok((dtype, f.nullable)) @@ -367,7 +369,7 @@ mod weighted_state_tests { // A state cannot acquire a different family or shape merely by relabeling its batch. #[test] fn weighted_state_family_and_shape_must_match() { - let cms = FieldDataType::Sketch( + let cms = SummaryFamilyType::Sketch( SketchKind::new( SketchAlgorithm::CmsWithHeap, SketchParams::CmsWithHeap { @@ -378,7 +380,7 @@ mod weighted_state_tests { ), Default::default(), ); - let cs = FieldDataType::Sketch( + let cs = SummaryFamilyType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { @@ -395,7 +397,7 @@ mod weighted_state_tests { let wrong_shape = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 5, 8).unwrap(); assert!(validate_state(&cs, &wrong_shape).is_err()); - let even_depth = FieldDataType::Sketch( + let even_depth = SummaryFamilyType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 37f780812..66702afdb 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -7,16 +7,18 @@ use asap_physical_operators::{ Error, }; use futures::{executor::block_on, FutureExt, StreamExt}; +use planner_types::ir::Predicate; +use planner_types::ir::ScalarExpr as QueryExpr; use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, + post_asap::{Field, FieldDataType}, + pre_asap::{DataType, JoinKind, ScalarValue}, }; use std::sync::Arc; fn schema(width: usize) -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: (0..width) .map(|i| Field { table: None, @@ -60,9 +62,7 @@ fn cross_join() -> Operator { schema(1), schema(1), JoinKind::Cross, - &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))), + &Predicate(QueryExpr::Literal(ScalarValue::Boolean(true))), schema(2), ) .unwrap() @@ -189,9 +189,10 @@ fn cooperative_sort_preserves_ties_across_chunks() { #[test] fn weighted_summary_build_yields_within_a_batch() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - let input = Arc::new(Schema { - closed: true, + + let input = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, diff --git a/crates/asap-physical-operators/tests/unified_common/mod.rs b/crates/asap-physical-operators/tests/common/mod.rs similarity index 100% rename from crates/asap-physical-operators/tests/unified_common/mod.rs rename to crates/asap-physical-operators/tests/common/mod.rs diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 079c2bd69..1f5d5ab2d 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -1,4 +1,5 @@ //! Spatial heap weights come from a fresh instant vector, never sample history. +mod common; use asap_physical_operators::{ operators::Operator, physical_planner::{ @@ -8,15 +9,16 @@ use asap_physical_operators::{ runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_physical_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; +use planner_types::ir::physical_export::PhysicalASAPOperatorPayload; use planner_types::{post_asap::*, pre_asap::DataType}; use std::{collections::BTreeMap, sync::Arc}; fn schema() -> Arc { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: [ ("ts", DataType::Timestamp), ("value", DataType::Float64), @@ -169,9 +171,9 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { }; let family = FieldDataType::Sketch(SketchKind::new(algorithm, params), Default::default()); let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); - let output = Arc::new(Schema { - closed: true, + let output = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![ schema().fields[2].clone(), schema().fields[3].clone(), @@ -179,7 +181,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { ], time_index: None, }); - let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); + let read = Operator::keyed_evaluation(build.schema(), 1, 1, output).unwrap(); let plan = CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema()))]), BTreeMap::from([ @@ -331,7 +333,7 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { .candidate(&open_root) .unwrap(); let snapshot_program = - asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + asap_physical_operators::physical_planner::promql_rows::compile_current_series_evaluation( &open_selected, ) .unwrap(); @@ -348,20 +350,24 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { ) .candidate(&root) .unwrap(); - let logical = compile_post_asap_dag(&selected).unwrap(); + let logical = compile_physical_asap_dag(&selected).unwrap(); let raw = logical .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP( + planner_types::ir::NonASAPOp::TimeRange { .. } + ) + ) + }) .unwrap(); let raw_schema = Arc::new(raw.output_schema.clone()); let physical = compile( &logical, - BTreeMap::from([( - u64::from(raw.id.0), - InputContract::bounded(raw_schema.clone()), - )]), - &[u64::from(logical.root.0)], + BTreeMap::from([(raw.id as u64, InputContract::bounded(raw_schema.clone()))]), + &[logical.roots[0] as u64], ) .unwrap(); let bytes = String::from_utf8(serde_json::to_vec(&physical).unwrap()).unwrap(); diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/asap-physical-operators/tests/deployment.rs index 2857a81cb..72e0503f1 100644 --- a/crates/asap-physical-operators/tests/deployment.rs +++ b/crates/asap-physical-operators/tests/deployment.rs @@ -35,7 +35,7 @@ fn read(state: &dyn AggregateCore) -> f64 { } // The same kernels work when every build is query-time, when only a prefix -// was precomputed, and when all state was precomputed before the readout. +// was precomputed, and when all state was precomputed before the evaluation. #[test] fn raw_partial_and_fully_precomputed_use_the_same_kernels() { let raw: Vec = (0..128).map(f64::from).collect(); diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index e20fc27ee..2f8f98b17 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -1,21 +1,24 @@ //! Planner-selected PromQL computation compiles from the timed DAG alone; //! the deployment supplies only raw rows at the ingestion frontier. +mod common; use asap_physical_operators::{ operators::Operator, physical_planner::{compile, promql_rows, CompiledPhysicalDAG, InputContract, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_physical_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; -use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; +use planner_types::ir::physical_export::{PhysicalASAPDAG, PhysicalASAPOperatorPayload}; +use planner_types::ir::ASAPOp; +use planner_types::{post_asap::*, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { lower_with(query, AccuracyTarget::Exact) } -fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn lower_with(query: &str, accuracy: AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -46,54 +49,55 @@ fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { } /// The first exact summary candidate, as Planner selection would hand it over. -fn exact_dag(query: &str) -> PostAsapDAG { - use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; +fn exact_dag(query: &str) -> PhysicalASAPDAG { let expression = lower(query); - let root = Rc::new(promql_rows::with_series_identity(&expression).unwrap_or(expression)); - asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) - .replacements(&TargetSubDAG::new(&root)) - .into_iter() - .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; - dag.nodes - .iter() - .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, FieldDataType::ExactAggregate(..)))) - .then_some(dag) - } - _ => None, - }) + let root = promql_rows::with_series_identity(&expression).unwrap_or(expression); + let space = asap_aware_mapping::search_workload(vec![("q", root)]); + let selected = space + .global_selection(&asap_aware_mapping::DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) .unwrap() + .unwrap(); + compile_physical_asap_dag(&selected).unwrap() } -fn population_dag(query: &str) -> PostAsapDAG { - let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); +fn population_dag(query: &str) -> PhysicalASAPDAG { + let root = promql_rows::with_series_identity(&lower(query)).unwrap(); let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&root), ) .candidate(&root) .unwrap(); - compile_post_asap_dag(&selected).unwrap() + compile_physical_asap_dag(&selected).unwrap() } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &PhysicalASAPDAG) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, - } => match child.as_ref() { - QueryExpr::Scan { - source: planner_types::pre_asap::Source::TimeSeries { metric }, - .. - } => Some(( - u64::from(node.id.0), - Arc::new(node.output_schema.clone()), - metric.clone(), - )), - _ => None, - }, + PhysicalASAPOperatorPayload::NonASAP(planner_types::ir::NonASAPOp::TimeRange { + .. + }) => { + let mut id = node.id; + loop { + let n = dag.nodes.iter().find(|n| n.id == id)?; + if let PhysicalASAPOperatorPayload::NonASAP( + planner_types::ir::NonASAPOp::Scan { + source: planner_types::pre_asap::Source::TimeSeries { metric }, + .. + }, + ) = &n.payload + { + return Some(( + node.id as u64, + Arc::new(node.output_schema.clone()), + metric.clone(), + )); + } + id = dag.edges.iter().find(|e| e.consumer == id)?.producer; + } + } _ => None, }) .collect() @@ -104,7 +108,7 @@ type Sample = (&'static str, &'static str, &'static str, i64, f64); /// Compile, round-trip, bind raw `(metric, job, instance, ts, value)` samples, /// and return the root's batches. fn execute( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, samples: &[Sample], end: i64, ) -> Result>, String> { @@ -114,7 +118,7 @@ fn execute( /// [`execute`], supplying samples of each instance in `relabel` under its /// `(__name__, instance)` instead. fn execute_relabeled( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, samples: &[Sample], end: i64, relabel: &BTreeMap<&str, (&str, &str)>, @@ -126,7 +130,7 @@ fn execute_relabeled( .iter() .map(|(id, schema, _)| (*id, InputContract::bounded(schema.clone()))) .collect(), - &[u64::from(dag.root.0)], + &[dag.roots[0] as u64], ) .map_err(|e| e.to_string())?; let program: CompiledPhysicalDAG = @@ -195,7 +199,11 @@ fn execute_relabeled( } /// [`execute`], returning `(job, value)` rows of the root. -fn run(dag: &PostAsapDAG, samples: &[Sample], end: i64) -> Result, String> { +fn run( + dag: &PhysicalASAPDAG, + samples: &[Sample], + end: i64, +) -> Result, String> { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let job = batch.schema().fields.iter().position(|f| f.name == "job"); @@ -250,7 +258,7 @@ fn population_aggregates_match_current_series_reference() { } } -// A global readout of an empty population is an empty vector, as in PromQL. +// A global evaluation of an empty population is an empty vector, as in PromQL. #[test] fn global_population_aggregate_of_no_members_is_empty() { // Latest values are [1, 2, 5, 7] at 60s; every member has expired by 1000s. @@ -313,40 +321,7 @@ fn grouped_vector_arithmetic_matches_labels() { // then roll up per job: api has 2 + 1 + 1 samples in 5m, db has 1. #[test] fn exact_count_finalizes_to_declared_float_value() { - let mut dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); - let finalize = dag - .nodes - .iter() - .find(|node| { - matches!( - node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } - ) - }) - .unwrap() - .clone(); - let root = dag.nodes.iter().find(|n| n.id == dag.root).unwrap().clone(); - let mut edge = dag - .edges - .iter() - .find(|e| e.producer == finalize.id) - .unwrap() - .clone(); - // Read the rolled-up exact state the same way the query path does. - let mut read = finalize.clone(); - read.id = PostAsapNodeId(root.id.0 + 1); - read.output_schema = root.output_schema.clone(); - read.output_schema.fields.last_mut().unwrap().dtype = - FieldDataType::Plain(planner_types::pre_asap::DataType::Float64); - edge.producer = root.id; - edge.consumer = read.id; - edge.intermediate_schema = root.output_schema.clone(); - edge.data_state = root.output_state; - dag.root = read.id; - dag.nodes.push(read); - dag.edges.push(edge); + let dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); assert_eq!( run(&dag, SAMPLES, 60_000).unwrap(), reference(&[("api", 4.), ("db", 1.)]) @@ -354,10 +329,20 @@ fn exact_count_finalizes_to_declared_float_value() { } /// `dag` with its Binary operator replaced by `kind`. -fn with_kind(mut dag: PostAsapDAG, kind: planner_types::pre_asap::BinaryOpKind) -> PostAsapDAG { +fn with_kind( + mut dag: PhysicalASAPDAG, + kind: planner_types::pre_asap::BinaryOpKind, + bool_result: bool, +) -> PhysicalASAPDAG { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + if let PhysicalASAPOperatorPayload::NonASAP(planner_types::ir::NonASAPOp::BinaryOp { + operator, + return_bool, + .. + }) = &mut node.payload + { operator.kind = kind.clone(); + *return_bool = bool_result; } } dag @@ -367,30 +352,30 @@ fn with_kind(mut dag: PostAsapDAG, kind: planner_types::pre_asap::BinaryOpKind) // holds, with their value, on either side of the literal; `bool` yields 1 or 0. #[test] fn grouped_comparisons_filter_or_return_bool() { - use planner_types::pre_asap::{BinaryOpKind::*, CompareOpKind::Gt}; - // sum_over_time over 5m per job: api = 14, db = 5. - let right = exact_dag("sum by (job) (sum_over_time(m[5m])) * 10"); - let left = exact_dag("10 - sum by (job) (sum_over_time(m[5m]))"); - for (dag, expected) in [ + for (query, expected) in [ ( - with_kind(right.clone(), Compare(Gt)), + "sum by(job)(sum_over_time(m[5m])) > 10", reference(&[("api", 14.)]), ), ( - with_kind(right, CompareBool(Gt)), + "sum by(job)(sum_over_time(m[5m])) > bool 10", reference(&[("api", 1.), ("db", 0.)]), ), - (with_kind(left, Compare(Gt)), reference(&[("db", 5.)])), + ( + "10 > sum by(job)(sum_over_time(m[5m]))", + reference(&[("db", 5.)]), + ), ] { - assert_eq!(run(&dag, SAMPLES, 60_000).unwrap(), expected); + assert_eq!(run(&exact_dag(query), SAMPLES, 60_000).unwrap(), expected); } } -// A `bool` comparison Binary over per-series readouts matches one-to-one and +// A `bool` comparison Binary over per-series evaluations matches one-to-one and // drops the metric name; a filter keeps the surviving left value. #[test] fn per_series_comparisons_filter_or_return_bool() { use planner_types::pre_asap::{BinaryOpKind::*, CompareOpKind::*}; + let samples = counter("a", "api", 10., 10.) .chain(counter("a", "db", 10., 10.)) .chain(counter("b", "api", 5., 5.)) @@ -399,18 +384,23 @@ fn per_series_comparisons_filter_or_return_bool() { // rate: a{api} = a{db} = 50/300, b{api} = 25/300, b{db} = 100/300. let dag = exact_dag("rate(a[5m]) / rate(b[5m])"); assert_eq!( - run_series(&with_kind(dag.clone(), Compare(Gt)), &samples, 300_000).unwrap(), + run_series( + &with_kind(dag.clone(), Compare(Gt), false), + &samples, + 300_000 + ) + .unwrap(), series(&[("api", "x", 50. / 300.)]) ); assert_eq!( - run_series(&with_kind(dag, CompareBool(Lt)), &samples, 300_000).unwrap(), + run_series(&with_kind(dag, Compare(Lt), true), &samples, 300_000).unwrap(), series(&[("api", "x", 0.), ("db", "x", 1.)]) ); } /// [`execute`], returning per-series `(identity, value)` rows of the root, /// with NaN-aware formatting for comparison. -fn run_series(dag: &PostAsapDAG, samples: &[Sample], end: i64) -> Result { +fn run_series(dag: &PhysicalASAPDAG, samples: &[Sample], end: i64) -> Result { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let schema = batch.schema(); @@ -509,13 +499,13 @@ fn per_series_rate_ratio_matches_prometheus() { // A literal operand applies to every stored per-series value, on either side, // and drops the metric name. #[test] -fn per_series_scalar_arithmetic_applies_to_stored_readouts() { +fn per_series_scalar_arithmetic_applies_to_stored_evaluations() { let samples = counter("m", "api", 10., 10.).collect::>(); // rate = 40 * 1.25 / 300 = 1/6. for (query, expected) in [ ("rate(m[5m]) * 2", 50. / 300. * 2.), ("1 - rate(m[5m])", 1. - 50. / 300.), - // The stored sum readout keeps `__name__`; the arithmetic drops it. + // The stored sum evaluation keeps `__name__`; the arithmetic drops it. ("sum_over_time(m[5m]) * 2", 150. * 2.), ] { assert_eq!( @@ -545,12 +535,16 @@ fn per_series_scalar_arithmetic_rejects_label_sets_equal_without_the_name() { } fn with_vector_match( - mut dag: PostAsapDAG, + mut dag: PhysicalASAPDAG, kind: planner_types::pre_asap::VectorMatchKind, labels: &[&str], -) -> PostAsapDAG { +) -> PhysicalASAPDAG { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + if let PhysicalASAPOperatorPayload::NonASAP(planner_types::ir::NonASAPOp::BinaryOp { + operator, + .. + }) = &mut node.payload + { operator.vector_match = Some(planner_types::pre_asap::VectorMatch { kind: kind.clone(), labels: labels.iter().map(|l| l.to_string()).collect(), @@ -621,47 +615,52 @@ fn population_sums_and_averages_are_compensated() { } } -// A bare count over stored Count-Min state compiles to a Planner readout that +// A bare count over stored Count-Min state compiles to a Planner evaluation that // returns the sketch's total update weight, including colliding items. #[test] -fn stored_count_min_bare_count_compiles_to_a_readout() { +fn stored_count_min_bare_count_compiles_to_a_evaluation() { use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; use asap_physical_operators::summary_kernels::CountMinSketchAccumulator; - let root = Rc::new(lower_with("count(up)", AccuracyTarget::Epsilon(0.02))); - let dag = - asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) - .replacements(&TargetSubDAG::new(&root)) - .into_iter() - .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; - let bare_count = dag.nodes.iter().any(|n| { - matches!( - &n.payload, - PostAsapOperatorPayload::SummaryEstimate { - query: SketchStatistic::PointCount { value: None, .. } - } - ) - }); - let count_min = dag.nodes.iter().any(|n| { - matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { + let root = lower_with("count(up)", AccuracyTarget::Epsilon(0.02)); + let dag = asap_aware_mapping::ASAPStrategies::new(&asap_aware_mapping::DefaultCostModel) + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::SubDAG(node) => { + let dag = compile_physical_asap_dag(&node).ok()?; + let bare_count = dag.nodes.iter().any(|n| { + matches!( + &n.payload, + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryEstimate { + query: SketchStatistic::PointCount { value: None, .. }, + .. + }) + ) + }); + let count_min = dag.nodes.iter().any(|n| { + matches!(&n.payload, PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } if kind.algorithm() == &SketchAlgorithm::Cms) - }); - (bare_count && count_min).then_some(dag) - } - _ => None, - }) - .expect("Planner lists a Count-Min candidate for count(up)"); + }) if kind.algorithm() == &SketchAlgorithm::Cms) + }); + (bare_count && count_min).then_some(dag) + } + _ => None, + }) + .expect("Planner lists a Count-Min candidate for count(up)"); let state = dag .nodes .iter() - .find(|n| matches!(n.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|n| { + matches!( + n.payload, + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { .. }) + ) + }) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &state.payload + }) = &state.payload else { unreachable!() }; @@ -671,11 +670,8 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { let schema = Arc::new(state.output_schema.clone()); let program = compile( &dag, - BTreeMap::from([( - u64::from(state.id.0), - InputContract::bounded(schema.clone()), - )]), - &[u64::from(dag.root.0)], + BTreeMap::from([(state.id as u64, InputContract::bounded(schema.clone()))]), + &[dag.roots[0] as u64], ) .unwrap(); let program: CompiledPhysicalDAG = @@ -697,7 +693,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { let batch = Batch::try_new(schema.clone(), vec![row]).unwrap(); let physical_dag = program .instantiate(BTreeMap::from([( - u64::from(state.id.0), + state.id as u64, Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, )])) .unwrap(); diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index e2ad735d6..ff121162c 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -8,15 +8,23 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::physical_export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::ASAPOp; +use planner_types::ir::BinaryOperator; +use planner_types::ir::NonASAPOp; +use planner_types::ir::ScalarExpr as QueryExpr; use planner_types::{ - post_asap::{ExactKind, ExactParams, Field, FieldDataType, Schema}, + post_asap::{ExactKind, ExactParams, Field, FieldDataType}, pre_asap::DataType, }; use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: fields .iter() .map(|(name, dtype, nullable)| Field { @@ -117,7 +125,7 @@ fn grouped_sort_limit_across_batches() { // The same computation runs in either engine scope with fresh per-run state. #[test] -fn summary_construction_merge_and_readout_at_both_phases() { +fn summary_construction_merge_and_evaluation_at_both_phases() { let schema = schema(&[("v", DataType::Float64, false)]); let batches = (1..=20) .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) @@ -140,11 +148,11 @@ fn summary_construction_merge_and_readout_at_both_phases() { dag.add( 4, vec![3], - Operator::readout( + Operator::evaluation( state, 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Sum, lookback_ms: None, }, @@ -295,10 +303,10 @@ fn binding_rejects_unsupported_operations() { vec![], ) .unwrap(); - assert!(Operator::readout( + assert!(Operator::evaluation( sum.schema(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( + asap_physical_operators::operators::SummaryEvaluation::Sketch( planner_types::post_asap::SketchStatistic::Quantile { q: 0.5 } ) ) @@ -310,6 +318,7 @@ fn binding_rejects_unsupported_operations() { #[test] fn kll_raw_partial_and_precomputed_are_native_dags() { use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("value", DataType::Float64, false)]); let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), @@ -396,10 +405,10 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { dag.add( 5, vec![4], - Operator::readout( + Operator::evaluation( state.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( + asap_physical_operators::operators::SummaryEvaluation::Sketch( planner_types::post_asap::SketchStatistic::Quantile { q: 0.5 }, ), ) @@ -423,9 +432,9 @@ fn exact_state_and_family_validation() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "state".into(), @@ -452,11 +461,11 @@ fn exact_state_and_family_validation() { dag.add( 1, vec![0], - Operator::readout( + Operator::evaluation( schema.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Sum, lookback_ms: None, }, @@ -486,57 +495,58 @@ fn exact_state_and_family_validation() { fn bind_post_asap_before_execution() { use asap_physical_operators::dag::planner::bind; use planner_types::{ - post_asap::{ - EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDAG, PostAsapDAGEdge, - PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, - WindowEdgeCompatibility, - }, - pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, + post_asap::ExecutionDataState, + pre_asap::{ArithmeticOpKind, ScalarValue}, }; - use std::{collections::BTreeMap, rc::Rc}; + use std::collections::BTreeMap; let schema = schema(&[("value", DataType::Float64, false)]); - let node = |id, payload| PostAsapDAGNode { - id: PostAsapNodeId(id), + let node = |id, payload| PhysicalASAPDAGNode { + id, payload, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, }; - let mut dag = PostAsapDAG { + let mut dag = PhysicalASAPDAG { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Values { + rows: vec![vec![planner_types::ir::ScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(1.), + )]], + schema: (*schema).clone(), + }), ), node( 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Project { - cols: vec![ProjectItem { - alias: None, - expr: QueryExpr::Arithmetic { - op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(0)), - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), - }, - }], - qualifier: None, - }, - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Project { + cols: vec![planner_types::ir::ProjectItem { + alias: None, + expr: planner_types::ir::ScalarExpr::Arithmetic { + semantics: planner_types::ir::ExprSemantics::Sql, + op: ArithmeticOpKind::Add, + left: Box::new(planner_types::ir::ScalarExpr::Column(0)), + right: Box::new(planner_types::ir::ScalarExpr::Literal( + ScalarValue::Float64(2.), + )), + }, + }], + qualifier: None, + child: 0, + }), ), ], - edges: vec![PostAsapDAGEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + edges: vec![PhysicalASAPDAGEdge { + producer: 0, + consumer: 1, role: EdgeRole::Input, intermediate_schema: (*schema).clone(), data_state: ExecutionDataState::QUERY_ROWS, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }], - root: PostAsapNodeId(1), + roots: vec![1], }; let sources = || -> BTreeMap> { BTreeMap::from([( @@ -555,17 +565,16 @@ fn bind_post_asap_before_execution() { // A literal Fallback needs no deployment input. let literal = bind(&dag, BTreeMap::new(), &[1]).unwrap(); assert_eq!(floats(&run(&literal, 1, query()), 0), vec![3.]); - dag.nodes[1].payload = PostAsapOperatorPayload::Value { - operation: ValueOperation::Extension { - name: "unknown".into(), - }, - }; + dag.nodes[1].payload = PhysicalASAPOperatorPayload::ASAP(ASAPOp::Extension { + child: 0, + name: "unsupported".into(), + }); assert!(bind(&dag, sources(), &[1]).is_err()); } // A completed empty population has an exact zero count, with integer output. #[test] -fn empty_exact_count_is_an_integer_state_readout() { +fn empty_exact_count_is_an_integer_state_evaluation() { let input = schema(&[("value", DataType::Float64, false)]); let build = Operator::summary_build( input.clone(), @@ -575,11 +584,11 @@ fn empty_exact_count_is_an_integer_state_readout() { vec![], ) .unwrap(); - let read = Operator::readout( + let read = Operator::evaluation( build.schema(), 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Count, lookback_ms: None, }, @@ -598,13 +607,7 @@ fn empty_exact_count_is_an_integer_state_readout() { #[test] fn source_batches_must_match_the_bound_schema() { use asap_physical_operators::dag::{self, PhysicalOperator}; - use planner_types::{ - post_asap::{ - ExecutionDataState, PostAsapDAG, PostAsapDAGNode, PostAsapNodeId, - PostAsapOperatorPayload, - }, - pre_asap::QueryExpr, - }; + use planner_types::post_asap::ExecutionDataState; use std::{cell::Cell, collections::BTreeMap, rc::Rc}; struct WrongSource { schema: SchemaRef, @@ -637,18 +640,21 @@ fn source_batches_must_match_the_bound_schema() { } let expected = schema(&[("value", DataType::Float64, false)]); let starts = Rc::new(Cell::new(0)); - let plan = PostAsapDAG { - nodes: vec![PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), - }, + let plan = PhysicalASAPDAG { + nodes: vec![PhysicalASAPDAGNode { + id: 0, + payload: PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Values { + rows: vec![vec![planner_types::ir::ScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(1.), + )]], + schema: (*expected).clone(), + }), output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*expected).clone(), guarantee: None, }], edges: vec![], - root: PostAsapNodeId(0), + roots: vec![0], }; let source = Box::new(WrongSource { schema: expected, @@ -709,26 +715,27 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { use asap_physical_operators::dag::planner::{bind, Source}; use planner_types::{ post_asap::*, - pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, + pre_asap::{CompareOpKind, GroupKeys, JoinKind}, }; - use std::{collections::BTreeMap, rc::Rc}; + use std::collections::BTreeMap; let rows_schema = schema(&[ ("group", DataType::Utf8, false), ("key", DataType::Utf8, false), ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &asap_physical_operators::values::SchemaRef| PostAsapDAGNode { - id: PostAsapNodeId(id), - payload, - output_schema: (**schema).clone(), - output_state: ExecutionDataState::QUERY_ROWS, - guarantee: None, - }; + let node = + |id, payload, schema: &asap_physical_operators::values::SchemaRef| PhysicalASAPDAGNode { + id, + payload, + output_schema: (**schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + }; let edge = |producer, consumer, role, schema: &asap_physical_operators::values::SchemaRef| { - PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + PhysicalASAPDAGEdge { + producer, + consumer, role, intermediate_schema: (**schema).clone(), data_state: ExecutionDataState::QUERY_ROWS, @@ -737,58 +744,64 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { } }; let groups = GroupKeys::by(vec![0]); - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Values { + rows: vec![vec![planner_types::ir::ScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(0.), + )]], + schema: (*rows_schema).clone(), + }), &rows_schema, ), node( 1, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Values { + rows: vec![vec![planner_types::ir::ScalarExpr::Literal( + planner_types::pre_asap::ScalarValue::Float64(0.), + )]], + schema: (*keys_schema).clone(), + }), &keys_schema, ), node( 2, - PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pruning: None, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(1)), + PhysicalASAPOperatorPayload::NonASAP(planner_types::ir::NonASAPOp::Join { + kind: JoinKind::Semi, + pred: planner_types::ir::Predicate(planner_types::ir::ScalarExpr::Compare { + left: Box::new(planner_types::ir::ScalarExpr::Column(1)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(3)), - })), - }, + right: Box::new(planner_types::ir::ScalarExpr::Column(3)), + semantics: planner_types::ir::ExprSemantics::Sql, + }), + left: 0, + right: 1, + }), &rows_schema, ), node( 3, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Sort { - keys: vec![SortKey { - expr: QueryExpr::Column(2), - ascending: false, - nulls_first: false, - }], - partition_by: groups.clone(), - }, - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Sort { + keys: vec![planner_types::ir::SortKey { + expr: planner_types::ir::ScalarExpr::Column(2), + ascending: false, + nulls_first: false, + }], + partition_by: groups.clone(), + child: 2, + }), &rows_schema, ), node( 4, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, - offset: 0, - partition_by: groups, - }, - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Limit { + n: Some(1), + offset: 0, + partition_by: groups, + child: 3, + }), &rows_schema, ), ], @@ -799,7 +812,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { edge(2, 3, EdgeRole::Input, &rows_schema), edge(3, 4, EdgeRole::Input, &rows_schema), ], - root: PostAsapNodeId(4), + roots: vec![4], }; let text = |v: &str| Value::Utf8(v.into()); for (phase, scope) in [ @@ -862,8 +875,8 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { #[test] fn planner_expressions_preserve_collection_and_nullable_types() { use asap_physical_operators::dag::expressions::CompiledExpression; - use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; - use std::rc::Rc; + use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let input_schema = schema(&[( "items", DataType::Map { @@ -911,9 +924,10 @@ fn planner_expressions_preserve_collection_and_nullable_types() { let projected = project.schema(); dag.add(1, vec![0], project).unwrap(); let predicate = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Ge, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Box::new(QueryExpr::Literal(ScalarValue::Int64(1))), }; dag.add( 2, @@ -937,14 +951,16 @@ fn planner_expressions_preserve_collection_and_nullable_types() { // Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. #[test] fn native_relational_join_kinds_preserve_unmatched_rows() { - use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; - use std::rc::Rc; + use planner_types::ir::Predicate; + use planner_types::pre_asap::{CompareOpKind, JoinKind}; + let input = schema(&[("key", DataType::Int64, true)]); - let predicate = Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let predicate = Predicate(QueryExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })); + right: Box::new(QueryExpr::Column(1)), + }); for (kind, count) in [ (JoinKind::Inner, 1), (JoinKind::Left, 3), @@ -1018,6 +1034,7 @@ fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scop } fn assert_weighted_rate_topk(count_sketch: bool) { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let raw = schema(&[ ("service", DataType::Utf8, false), ("job", DataType::Utf8, false), @@ -1084,7 +1101,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { ("service", DataType::Utf8, false), ("score", DataType::Float64, false), ]); - let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); + let evaluation = Operator::keyed_evaluation(build.schema(), 1, 8, output.clone()).unwrap(); let mut dag = PhysicalDAG::default(); dag.add( 0, @@ -1094,7 +1111,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { .unwrap(); dag.add(1, vec![0], rates).unwrap(); dag.add(2, vec![1], build).unwrap(); - dag.add(3, vec![2], readout).unwrap(); + dag.add(3, vec![2], evaluation).unwrap(); dag.add( 4, vec![3], @@ -1141,17 +1158,16 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { use asap_physical_operators::physical_planner::{ compile_node, CompiledPhysicalDAG, InputContract, Source, }; - use planner_types::post_asap::{ - ExecutionDataState, PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, - ValueOperation, - }; + use planner_types::ir::physical_export::{PhysicalASAPDAGNode, PhysicalASAPOperatorPayload}; + use planner_types::post_asap::ExecutionDataState; + use planner_types::pre_asap::{ - aggregate_output_schema, AggIntent, Field, GroupKeys, QueryExpr, Reduction as IrReduction, - Schema as IrSchema, + aggregate_output_schema, AggIntent, GroupKeys, Reduction as IrReduction, Schema as IrSchema, }; + let grouped = IrSchema::new(vec![ - Field::plain("job", DataType::Utf8, false), - Field::plain("sum", DataType::Float64, false), + planner_types::pre_asap::Field::plain("job", DataType::Utf8, false), + planner_types::pre_asap::Field::plain("sum", DataType::Float64, false), ]); let output = aggregate_output_schema( &grouped, @@ -1167,15 +1183,15 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { .map(|c| { ( c.name.as_str(), - c.dtype.plain().unwrap().clone(), + c.plain_dtype().unwrap().clone(), c.nullable, ) }) .collect::>(), ); - let node = |id, operation| PostAsapDAGNode { - id: PostAsapNodeId(id), - payload: PostAsapOperatorPayload::Value { operation }, + let node = |id, operation| PhysicalASAPDAGNode { + id, + payload: PhysicalASAPOperatorPayload::NonASAP(operation), output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*input).clone(), guarantee: None, @@ -1183,13 +1199,14 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { let sort = compile_node( &node( 1, - ValueOperation::Sort { - keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(1), + NonASAPOp::Sort { + keys: vec![planner_types::ir::SortKey { + expr: planner_types::ir::ScalarExpr::Column(1), ascending: false, nulls_first: false, }], partition_by: GroupKeys::none(), + child: 0, }, ), std::slice::from_ref(&input), @@ -1198,10 +1215,11 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { let limit = compile_node( &node( 2, - ValueOperation::Limit { - n: 1, + NonASAPOp::Limit { + n: Some(1), offset: 0, partition_by: GroupKeys::none(), + child: 1, }, ), std::slice::from_ref(&input), @@ -1253,32 +1271,27 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { }; use planner_types::{ post_asap::*, - pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}, + pre_asap::{CompareOpKind, JoinKind}, }; - use std::{collections::BTreeMap, rc::Rc}; + use std::collections::BTreeMap; let schema = schema(&[("key", DataType::Utf8, false)]); for certified in [false, true] { - let node = PostAsapDAGNode { - id: PostAsapNodeId(2), + let node = PhysicalASAPDAGNode { + id: 2, output_schema: (*schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + payload: PhysicalASAPOperatorPayload::NonASAP(planner_types::ir::NonASAPOp::Join { + kind: JoinKind::Semi, + pred: planner_types::ir::Predicate(planner_types::ir::ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(planner_types::ir::ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })), - pruning: certified.then_some(CandidateCompleteness::Certified { - guarantee: ResultGuarantee { - metric: ErrorMetric::TopKMembership, - bound: BoundExpr::Zero, - failure_probability: ProbabilityExpr::Constant { value: 0.01 }, - provenance: vec![], - }, + right: Box::new(planner_types::ir::ScalarExpr::Column(1)), }), - }, + left: 0, + right: 1, + }), }; let dag = CompiledPhysicalDAG::from_operators( [ @@ -1290,7 +1303,12 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { 2, ( vec![0, 1], - compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(), + if certified { + Operator::certified_semi_join(schema.clone(), schema.clone(), vec![(0, 0)]) + .unwrap() + } else { + compile_node(&node, &[schema.clone(), schema.clone()]).unwrap() + }, ), )] .into(), @@ -1373,19 +1391,22 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ("time", DataType::Timestamp, false), ("value", DataType::Float64, false), ]); - let node = PostAsapDAGNode { - id: PostAsapNodeId(2), + let node = PhysicalASAPDAGNode { + id: 2, output_schema: (*input).clone(), output_state: ExecutionDataState::INGESTION_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::Binary { + payload: PhysicalASAPOperatorPayload::NonASAP(planner_types::ir::NonASAPOp::BinaryOp { operator: BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), vector_match: None, checked_relative_division: false, checked_finite_division: false, }, - }, + return_bool: false, + lhs: 0, + rhs: 1, + }), }; let program = CompiledPhysicalDAG::from_operators( [ diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs index fda1b502a..e755090e6 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -5,15 +5,15 @@ use asap_physical_operators::{ physical_planner::{CompiledPhysicalDAG, InputContract}, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, + post_asap::{Field, FieldDataType}, pre_asap::DataType, }; use std::{collections::BTreeMap, sync::Arc}; fn sorted() -> CompiledPhysicalDAG { - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -74,7 +74,7 @@ fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { #[test] fn candidate_recovery_preserves_materialization_boundary() { - use asap_physical_operators::physical_planner::PhysicalASAPDAG; + use asap_physical_operators::physical_planner::CompiledPhysicalPlan; let precompute = sorted(); let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); let query = CompiledPhysicalDAG::from_operators( @@ -89,13 +89,13 @@ fn candidate_recovery_preserves_materialization_boundary() { vec![2], ) .unwrap(); - let candidate = PhysicalASAPDAG { + let candidate = CompiledPhysicalPlan { precompute: Some(precompute), query, materialized_outputs: BTreeMap::from([(1, output)]), }; let bytes = serde_json::to_vec(&candidate).unwrap(); - let restored = serde_json::from_slice::(&bytes).unwrap(); + let restored = serde_json::from_slice::(&bytes).unwrap(); assert_eq!(restored.precompute.as_ref().unwrap().roots(), &[1]); assert_eq!(restored.query.roots(), &[2]); assert_eq!(serde_json::to_vec(&restored).unwrap(), bytes); @@ -103,6 +103,7 @@ fn candidate_recovery_preserves_materialization_boundary() { wire["materialized_outputs"]["1"]["schema"]["fields"][0]["dtype"] = serde_json::json!({"Plain":"utf8"}); assert!( - serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()).is_err() + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) + .is_err() ); } diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index 56b62590b..b5ffbc8ae 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -9,16 +9,20 @@ use asap_physical_operators::{ values::{Batch, SchemaRef, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::physical_export::{PhysicalASAPDAGNode, PhysicalASAPOperatorPayload}; +use planner_types::ir::NonASAPOp; +use planner_types::ir::Predicate; +use planner_types::ir::ScalarExpr as QueryExpr; use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, + post_asap::{Field, FieldDataType}, + pre_asap::{CompareOpKind, DataType, JoinKind}, }; -use std::{rc::Rc, sync::Arc}; +use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: fields .iter() .map(|(name, dtype, nullable)| Field { @@ -74,11 +78,12 @@ fn keys(rows: &[Vec]) -> Vec>> { .collect() } fn eq_predicate() -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + Predicate(QueryExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })) + right: Box::new(QueryExpr::Column(1)), + }) } fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { let input = schema(&[("key", DataType::Float64, true)]); @@ -345,7 +350,7 @@ fn global_extrema_bind_with_planner_derived_schema() { use asap_physical_operators::physical_planner::compile_node; use planner_types::{ post_asap::*, - pre_asap::{AggIntent, Field, GroupKeys, Reduction as PlanReduction}, + pre_asap::{AggIntent, GroupKeys, Reduction as PlanReduction}, }; let input = schema(&[("v", DataType::Int64, false)]); for measure in [ @@ -353,8 +358,12 @@ fn global_extrema_bind_with_planner_derived_schema() { AggIntent::Max { col: Some(0) }, ] { let planner_input = - planner_types::pre_asap::Schema::new(vec![Field::plain("v", DataType::Int64, false)]); - let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( + planner_types::pre_asap::Schema::new(vec![planner_types::pre_asap::Field::plain( + "v", + DataType::Int64, + false, + )]); + let derived = planner_types::pre_asap::aggregate_output_schema( &planner_input, &PlanReduction::Reduce(GroupKeys::by(vec![])), std::slice::from_ref(&measure), @@ -364,20 +373,19 @@ fn global_extrema_bind_with_planner_derived_schema() { let result = derived.fields[0].clone(); let output = schema(&[( &result.name, - result.dtype.plain().unwrap().clone(), + result.plain_dtype().unwrap().clone(), result.nullable, )]); - let node = PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::Exact(ExactOperation::Aggregate { - reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), - measures: vec![measure], - output_names: vec![result.name], - filters: vec![], - having: None, - }), - }, + let node = PhysicalASAPDAGNode { + id: 1, + payload: PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Aggregate { + reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), + measures: vec![measure], + output_names: vec![result.name], + filters: vec![], + having: None, + child: 0, + }), output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*output).clone(), guarantee: None, @@ -408,9 +416,10 @@ fn planner_comparisons_handle_nan_without_execution_errors() { CompareOpKind::Ge, ] { let expression = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: op.clone(), - right: Rc::new(QueryExpr::Column(1)), + right: Box::new(QueryExpr::Column(1)), }; let compiled = CompiledExpression::compile(&expression, &input).unwrap(); for row in [ @@ -475,9 +484,10 @@ fn mixed_numeric_comparisons_preserve_large_integer_precision() { ("b", DataType::Float64, false), ]); let expr = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(QueryExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Column(1)), + right: Box::new(QueryExpr::Column(1)), }; let compiled = CompiledExpression::compile(&expr, &input).unwrap(); for (a, b, expected) in [ @@ -543,8 +553,9 @@ fn boolean_truth_tables_agree_between_expression_paths() { // Partial/final execution must agree with one build for an uncompacted KLL population. #[test] -fn kll_partial_merge_and_multiple_readouts_preserve_population() { +fn kll_partial_merge_and_multiple_evaluations_preserve_population() { use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("v", DataType::Float64, false)]); let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), @@ -588,10 +599,10 @@ fn kll_partial_merge_and_multiple_readouts_preserve_population() { dag.add( id, vec![build], - Operator::readout( + Operator::evaluation( state.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( + asap_physical_operators::operators::SummaryEvaluation::Sketch( planner_types::post_asap::SketchStatistic::Quantile { q }, ), ) @@ -662,6 +673,7 @@ fn zero_column_output_obeys_memory_limit() { fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { use asap_physical_operators::Statistic; use planner_types::post_asap::{ExactKind, ExactParams}; + let input = schema(&[("v", DataType::Float64, false)]); for (kind, params, statistic) in [ (ExactKind::Min, ExactParams::Min, Statistic::Min), @@ -683,11 +695,11 @@ fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { dag.add( 2, vec![1], - Operator::readout( + Operator::evaluation( state, 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_physical_operators::operators::SummaryEvaluation::Exact( + asap_physical_operators::summary_kernels::exact::ExactEvaluation { statistic, lookback_ms: None, }, diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs index e3dd7b5fc..ab488162d 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -8,8 +8,8 @@ use asap_physical_operators::{ Error, }; use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::{DataType, QueryExpr, Source}, + post_asap::{Field, FieldDataType}, + pre_asap::{DataType, Schema, Source}, }; use std::sync::{ atomic::{AtomicUsize, Ordering}, @@ -35,9 +35,9 @@ impl RawSource for DeclaredSource { // A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. #[test] fn blocking_inputs_require_an_explicit_finite_source() { - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "v".into(), @@ -67,11 +67,20 @@ fn blocking_inputs_require_an_explicit_finite_source() { ) .unwrap(); let scan = registry - .bind(&QueryExpr::Scan { - source: identity, - schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), - predicates: vec![], - }) + .bind( + &planner_types::ir::OperatorNode::new_shared(planner_types::ir::Operator::NonASAP( + planner_types::ir::NonASAPOp::Scan { + source: identity, + schema: Schema::new(vec![planner_types::pre_asap::Field::plain( + "v", + DataType::Int64, + false, + )]), + predicates: vec![], + }, + )) + .unwrap(), + ) .unwrap(); let mut dag = PhysicalDAG::default(); dag.add(0, vec![], scan).unwrap(); @@ -112,11 +121,11 @@ fn blocking_inputs_require_an_explicit_finite_source() { } } -// Kernel support must not be mistaken for executable native state/readout support. +// Kernel support must not be mistaken for executable native state/evaluation support. #[test] fn summary_capability_levels_are_distinct() { use asap_physical_operators::{ - capability::{validate_native_family, validate_sketch_readout, validate_summary_kernel}, + capability::{validate_native_family, validate_sketch_evaluation, validate_summary_kernel}, planner::post_asap::SketchStatistic, }; use planner_types::{ @@ -148,8 +157,8 @@ fn summary_capability_levels_are_distinct() { key: ColumnRef::SampleValue, value: None, }; - assert!(validate_sketch_readout(&cms, &bare_count).is_ok()); - assert!(validate_sketch_readout( + assert!(validate_sketch_evaluation(&cms, &bare_count).is_ok()); + assert!(validate_sketch_evaluation( &cms, &SketchStatistic::PointCount { key: ColumnRef::Named("host".into()), @@ -162,7 +171,7 @@ fn summary_capability_levels_are_distinct() { grouping, ); assert!(validate_native_family(&kll).is_ok()); - assert!(validate_sketch_readout(&kll, &SketchStatistic::Quantile { q: 1.5 }).is_err()); - assert!(validate_sketch_readout(&kll, &SketchStatistic::Cardinality).is_err()); - assert!(validate_sketch_readout(&kll, &SketchStatistic::Quantile { q: 0.5 }).is_ok()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Quantile { q: 1.5 }).is_err()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Cardinality).is_err()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Quantile { q: 0.5 }).is_ok()); } diff --git a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs index 1f03b61c4..228332cbc 100644 --- a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs +++ b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs @@ -2,6 +2,10 @@ //! Planner's search space: `enumerate_candidate_dags_for_root` lists //! current-series TopK heaps without a caller-side series-identity pass, cost //! ranking, or workload Cartesian expansion. Placement variants are not listed. +mod common; +use common::compile_physical_asap_dag; +use planner_types::ir::OperatorNode as QueryExpr; + use asap_aware_mapping::{ accuracy::{AccuracyEvidenceProvider, DefaultAccuracyModel, PropagationStats}, cost_model::DefaultCostModel, @@ -9,11 +13,10 @@ use asap_aware_mapping::{ search_workload_with_targets, Proposals, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_physical_operators::physical_planner::promql_rows::{ - compile_current_series_readout, SERIES_IDENTITY_COLUMN, + compile_current_series_evaluation, SERIES_IDENTITY_COLUMN, }; use planner_types::{ post_asap::*, - pre_asap::QueryExpr, types::AccuracyTarget, workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, @@ -89,14 +92,12 @@ fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { ..Default::default() }), }; - Rc::new( - asap_frontend_promql::lower_promql_workload(&workload, 0) - .unwrap() - .remove(0), - ) + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0) } -type InventoryDAG = Vec<(usize, Rc)>; +type InventoryDAG = Vec<(usize, Rc)>; /// Candidate DAGs for query 1 of a two-query workload, with and without /// whole-root proposals. Query 0 is a bystander that must not multiply them. @@ -126,7 +127,7 @@ fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec fn carries_identity(dag: &InventoryDAG) -> bool { dag.iter().any(|(_, root)| { - compile_post_asap_dag(root) + compile_physical_asap_dag(root) .unwrap() .nodes .iter() @@ -140,7 +141,10 @@ fn carries_identity(dag: &InventoryDAG) -> bool { } /// Shared acceptance checks; returns the added identity-carrying alternatives. -fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { +fn added_alternatives( + query: &str, + accuracy: AccuracyTarget, +) -> Vec> { let (full, logical) = inventories(query, accuracy); for (index, dag) in full.iter().enumerate() { assert_eq!(dag.len(), 1, "one root per candidate, no workload product"); @@ -159,15 +163,18 @@ fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec asap_aware_mapping::CandidateLogicalASAPDAGs<&'static str> { let workload = PlanningWorkload { @@ -40,26 +44,22 @@ fn grouped_rate_space() -> asap_aware_mapping::CandidateLogicalASAPDAGs<&'static ..Default::default() }), }; - let root = Rc::new( - asap_frontend_promql::lower_promql_workload(&workload, 0) - .unwrap() - .remove(0), - ); - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) - .unwrap(), - ); + let root = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let root = asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) + .unwrap(); search_workload(vec![("grouped-rate", root)]) } -fn grouped_rate() -> PostAsapDAG { +fn grouped_rate() -> PhysicalASAPDAG { let space = grouped_rate_space(); let selected = space .global_selection(&DefaultCostModel) .assemble_selected_query(&space.roots[0].1) .unwrap() .unwrap(); - compile_post_asap_dag(&selected).unwrap() + compile_physical_asap_dag(&selected).unwrap() } fn run(plan: &CompiledPhysicalDAG, inputs: BTreeMap, scope: Scope) -> Vec { let sources = inputs @@ -81,7 +81,7 @@ fn run(plan: &CompiledPhysicalDAG, inputs: BTreeMap, scope: Scope) - }) } -/// Rate readouts and grouped Sum can run together during bounded precompute; +/// Rate evaluations and grouped Sum can run together during bounded precompute; /// storing per-series rates instead leaves the same Sum in the query DAG. #[test] fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { @@ -92,22 +92,20 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. - } + }) ) }) .unwrap(); - let readout = dag + let evaluation = dag .nodes .iter() .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } + PhysicalASAPOperatorPayload::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) ) && dag .edges .iter() @@ -116,12 +114,12 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .unwrap(); let input_schema = Arc::new(state.output_schema.clone()); let (family, update, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family, input, grouping, .. - } => (family, input, grouping), + }) => (family, input, grouping), _ => unreachable!(), }; let range_ms = Some((-58_000, 2_000)); @@ -139,7 +137,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .as_any() .downcast_ref::() .unwrap() - .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .evaluation(asap_physical_operators::Statistic::Rate, range_ms, None) .unwrap() .unwrap(); let summary = Value::Summary { @@ -169,9 +167,9 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { }) .collect(); let batch = Batch::try_new(input_schema.clone(), rows).unwrap(); - let root = u64::from(dag.root.0); - let state_id = u64::from(state.id.0); - let rate_id = u64::from(readout.id.0); + let root = dag.roots[0] as u64; + let state_id = state.id as u64; + let rate_id = evaluation.id as u64; let frontiers = asap_physical_operators::physical_planner::enumerate_frontiers( &dag, &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), @@ -367,7 +365,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .as_any() .downcast_ref::() .unwrap() - .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .evaluation(asap_physical_operators::Statistic::Rate, range_ms, None) .unwrap() .unwrap(); assert_ne!( @@ -376,7 +374,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { ); } -/// Enumerated frontiers include both grouped-result and per-series readout +/// Enumerated frontiers include both grouped-result and per-series evaluation /// persistence; an explicit Rate-state input retains its original semantics. #[test] fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { @@ -388,18 +386,18 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. - } + }) ) }) .unwrap(); let inputs = BTreeMap::from([( - u64::from(state.id.0), + state.id as u64, InputContract::bounded(Arc::new(state.output_schema.clone())), )]); - let roots = [u64::from(dag.root.0)]; + let roots = [dag.roots[0] as u64]; let frontiers = enumerate_frontiers(&dag, &inputs, &roots, 4096).unwrap(); let candidates = compile_candidates(&dag, inputs.clone(), &roots, &frontiers) .into_iter() @@ -422,14 +420,14 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { let mut executed = 0; for forest in inventory.candidates { let root = &forest[0].1; - let dag = compile_post_asap_dag(root).unwrap(); + let dag = compile_physical_asap_dag(root).unwrap(); let Some(state) = dag.nodes.iter().find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. - } + }) ) }) else { continue; @@ -440,30 +438,30 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. - } + }) ) }) - .map(|node| u64::from(node.id.0)) - .unwrap_or(u64::from(dag.root.0)); + .map(|node| node.id as u64) + .unwrap_or(dag.roots[0] as u64); let physical_asap_dags = compile_candidates( &dag, BTreeMap::from([( - u64::from(state.id.0), + state.id as u64, InputContract::bounded(Arc::new(state.output_schema.clone())), )]), - &[u64::from(dag.root.0)], + &[dag.roots[0] as u64], &[vec![], vec![boundary]], ); let (family, input, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family, input, grouping, .. - } => (family, input, grouping), + }) => (family, input, grouping), _ => unreachable!(), }; let schema = Arc::new(state.output_schema.clone()); @@ -558,14 +556,14 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { /// The per-frontier lowering used before compile-once cuts: each boundary /// choice lowers the precompute and query DAGs from the logical DAG again. fn recompiled_candidate( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: &BTreeMap, roots: &[u64], frontier: &[u64], -) -> Result { +) -> Result { use asap_physical_operators::plan::Emission; if frontier.is_empty() { - return Ok(PhysicalASAPDAG { + return Ok(CompiledPhysicalPlan { precompute: None, query: compile(dag, inputs.clone(), roots)?, materialized_outputs: BTreeMap::new(), @@ -580,7 +578,7 @@ fn recompiled_candidate( } let mut query_inputs = inputs.clone(); query_inputs.extend(materialized_outputs.clone()); - Ok(PhysicalASAPDAG { + Ok(CompiledPhysicalPlan { precompute: Some(precompute), query: compile(dag, query_inputs, roots)?, materialized_outputs, @@ -588,7 +586,7 @@ fn recompiled_candidate( } fn assert_cuts_match_recompilation( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[u64], min_frontiers: usize, @@ -615,13 +613,18 @@ fn grouped_rate_cuts_equal_per_frontier_compilation() { let state = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { .. }) + ) + }) .unwrap(); let inputs = BTreeMap::from([( - u64::from(state.id.0), + state.id as u64, InputContract::bounded(Arc::new(state.output_schema.clone())), )]); - assert_cuts_match_recompilation(&dag, inputs, &[u64::from(dag.root.0)], 3); + assert_cuts_match_recompilation(&dag, inputs, &[dag.roots[0] as u64], 3); } /// Cuts of a DAG whose nodes lower to helper operators (current-series @@ -655,26 +658,32 @@ fn population_topk_cuts_equal_per_frontier_compilation() { let original = asap_frontend_promql::lower_promql_workload(&workload, 0) .unwrap() .remove(0); - let root = Rc::new( + let root = asap_physical_operators::physical_planner::promql_rows::with_series_identity(&original) - .unwrap(), - ); + .unwrap(); let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&root), ) .candidate(&root) .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let dag = compile_physical_asap_dag(&selected).unwrap(); let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP( + planner_types::ir::NonASAPOp::TimeRange { .. } + ) + ) + }) .unwrap(); let inputs = BTreeMap::from([( - u64::from(raw.id.0), + raw.id as u64, InputContract::bounded(Arc::new(raw.output_schema.clone())), )]); - let roots = [u64::from(dag.root.0)]; + let roots = [dag.roots[0] as u64]; let compiled = compile(&dag, inputs.clone(), &roots).unwrap(); // The root reads its population through a Sort helper numbered by the root. let helper = u64::MAX - (roots[0] << 16); @@ -694,25 +703,24 @@ fn cut_candidate_rejects_invalid_frontiers() { let state = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { .. }) + ) + }) .unwrap(); - let readout = dag + let evaluation = dag .nodes .iter() .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } + PhysicalASAPOperatorPayload::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) ) }) .unwrap(); - let (state_id, rate_id, root) = ( - u64::from(state.id.0), - u64::from(readout.id.0), - u64::from(dag.root.0), - ); + let (state_id, rate_id, root) = (state.id as u64, evaluation.id as u64, dag.roots[0] as u64); let inputs = BTreeMap::from([( state_id, InputContract::bounded(Arc::new(state.output_schema.clone())), diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index 5b91c6e72..f156d6407 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -8,7 +8,12 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; +use planner_types::ir::physical_export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::ASAPOp; +use planner_types::ir::BinaryOperator; use planner_types::{ post_asap::*, pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, @@ -19,9 +24,9 @@ use std::{collections::BTreeMap, sync::Arc}; #[test] fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = |dtype| Schema { - closed: true, + let schema = |dtype| planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -50,39 +55,46 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (SummaryInputExpr::Constant(1.), 4.), ] { let nodes = vec![ - PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + PhysicalASAPDAGNode { + id: 0, + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryMerge { + children: vec![], + }), output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: state_schema.clone(), guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, + PhysicalASAPDAGNode { + id: 1, + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::FinalizeExactAccumulator { + child: 0, + }), output_state: ExecutionDataState::INGESTION_ROWS, output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, + PhysicalASAPDAGNode { + id: 2, + payload: PhysicalASAPOperatorPayload::NonASAP( + planner_types::ir::NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: 1, + rhs: 1, }, - }, + ), output_state: ExecutionDataState::INGESTION_ROWS, output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(3), - payload: PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPDAGNode { + id: 3, + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: family.clone(), input: SummaryUpdate { weight, @@ -91,7 +103,8 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { reduction: Reduction::Reduce(GroupKeys::by(vec![])), grouping: GroupingStrategy::PerSubpopulationInstance, filter: None, - }, + child: 2, + }), output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: state_schema.clone(), guarantee: None, @@ -104,20 +117,20 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (2, 3, EdgeRole::Input), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + .map(|(producer, consumer, role)| PhysicalASAPDAGEdge { + producer, + consumer, role, - intermediate_schema: nodes[producer as usize].output_schema.clone(), - data_state: nodes[producer as usize].output_state, + intermediate_schema: nodes[producer].output_schema.clone(), + data_state: nodes[producer].output_state, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes, edges, - root: PostAsapNodeId(3), + roots: vec![3], }; // Identity metadata must remain one non-null Utf8 column. for mutation in 0..3 { @@ -131,7 +144,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { assert!(precompute::compile(&invalid_identity, &[0], &[3]).is_err()); } let mut invalid_grouping = dag.clone(); - let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = + let PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &mut invalid_grouping.nodes[3].payload else { unreachable!() @@ -211,7 +224,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { .as_any() .downcast_ref::() .unwrap() - .readout(Statistic::Sum, None, None) + .evaluation(Statistic::Sum, None, None) .unwrap() .unwrap(), expected @@ -221,9 +234,9 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { } fn logical_schema(family: FieldDataType) -> Schema { - Schema { - closed: true, + planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -238,47 +251,48 @@ fn state_dag( target: Option, merge: bool, ) -> CompiledPhysicalDAG { - let mut nodes = vec![PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + let mut nodes = vec![PhysicalASAPDAGNode { + id: 0, + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryMerge { children: vec![] }), output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: logical_schema(family.clone()), guarantee: None, }]; if merge { - nodes.push(PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::SummaryMerge, + nodes.push(PhysicalASAPDAGNode { + id: 1, + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryMerge { children: vec![0] }), ..nodes[0].clone() }); } - let read_id = nodes.len() as u32; - nodes.push(PostAsapDAGNode { - id: PostAsapNodeId(read_id), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, + let read_id = nodes.len(); + nodes.push(PhysicalASAPDAGNode { + id: read_id, + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::FinalizeExactAccumulator { + child: read_id - 1, + }), output_state: ExecutionDataState::INGESTION_ROWS, output_schema: logical_schema(FieldDataType::Plain(DataType::Float64)), guarantee: None, }); if let Some(target) = target { - nodes.push(PostAsapDAGNode { - id: PostAsapNodeId(nodes.len() as u32), - payload: PostAsapOperatorPayload::SummaryAgg { + nodes.push(PhysicalASAPDAGNode { + id: nodes.len(), + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: target.clone(), input: SummaryUpdate::column(ColumnRef::SampleValue), reduction: Reduction::by(vec![]), grouping: GroupingStrategy::default(), filter: None, - }, + child: read_id, + }), output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: logical_schema(target), guarantee: None, }); } let edges = (1..nodes.len()) - .map(|i| PostAsapDAGEdge { + .map(|i| PhysicalASAPDAGEdge { producer: nodes[i - 1].id, consumer: nodes[i].id, role: EdgeRole::Input, @@ -290,9 +304,13 @@ fn state_dag( .collect(); let root = nodes.last().unwrap().id; precompute::compile( - &PostAsapDAG { nodes, edges, root }, + &PhysicalASAPDAG { + nodes, + edges, + roots: vec![root], + }, &[0], - &[u64::from(root.0)], + &[root as u64], ) .unwrap() } @@ -374,7 +392,7 @@ fn explicit_merge_changes_pane_cardinality() { .iter() .map(|row| match row[2] { Value::Float64(v) => v, - _ => panic!("numeric readout expected"), + _ => panic!("numeric evaluation expected"), }) .collect::>(); assert_eq!(values, expected); diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index 4eec88aa1..41d8802fc 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -6,19 +6,22 @@ use asap_physical_operators::{ values::{Batch, SchemaRef, Value}, }; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::physical_export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::ASAPOp; +use planner_types::ir::BinaryOperator; use planner_types::{ - post_asap::{ - BinaryOperator, ExecutionDataState, Field, FieldDataType, PostAsapDAGNode, PostAsapNodeId, - PostAsapOperatorPayload, Schema, - }, + post_asap::{ExecutionDataState, Field, FieldDataType}, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; fn schema() -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, @@ -61,10 +64,18 @@ fn program() -> CompiledPhysicalDAG { }) } fn program_for(operator: BinaryOperator) -> CompiledPhysicalDAG { + program_for_bool(operator, false) +} +fn program_for_bool(operator: BinaryOperator, return_bool: bool) -> CompiledPhysicalDAG { let schema = schema(); - let node = PostAsapDAGNode { - id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { operator }, + let node = PhysicalASAPDAGNode { + id: 2, + payload: PhysicalASAPOperatorPayload::NonASAP(planner_types::ir::NonASAPOp::BinaryOp { + operator, + return_bool, + lhs: 0, + rhs: 1, + }), output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, @@ -156,12 +167,15 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { use planner_types::pre_asap::CompareOpKind; for return_bool in [false, true] { let physical_dag = promql_values::compile_binary( - &BinaryOperator { - kind: BinaryOpKind::Compare(CompareOpKind::Lt), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, + &asap_physical_operators::expressions::binary::BinaryOperator::from_logical( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Lt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + ), return_bool, true, false, @@ -285,12 +299,15 @@ fn binary_obeys_memory_and_cancellation() { // A `bool` comparison over label-map vectors yields 1 or 0 and drops the name. #[test] fn label_map_bool_comparison_drops_the_name() { - let program = program_for(BinaryOperator { - kind: BinaryOpKind::CompareBool(planner_types::pre_asap::CompareOpKind::Gt), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }); + let program = program_for_bool( + BinaryOperator { + kind: BinaryOpKind::Compare(planner_types::pre_asap::CompareOpKind::Gt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + true, + ); let rows = evaluate_with( program, vec![row("a", "api", 6.)], @@ -309,9 +326,9 @@ fn label_map_bool_comparison_drops_the_name() { assert!(matches!(row[1], Value::Float64(v) if v == 1.)); } -// Stored temporal readouts drop metric names before filter comparisons and set matching. +// Stored temporal evaluations drop metric names before filter comparisons and set matching. #[test] -fn stored_series_readouts_support_filters_and_sets() { +fn stored_series_evaluations_support_filters_and_sets() { use asap_physical_operators::{ physical_planner::compile, summary_kernels::exact::ExactAccumulator, }; @@ -319,14 +336,15 @@ fn stored_series_readouts_support_filters_and_sets() { use planner_types::pre_asap::{ schema::PROMQL_SERIES_IDENTITY, CompareOpKind, PromQLVectorSetOpKind, }; + for (exact_kind, params) in [ (ExactKind::Sum, ExactParams::Sum), (ExactKind::Count, ExactParams::Count), ] { let family = FieldDataType::ExactAggregate(exact_kind.clone(), params); - let state_schema = Arc::new(Schema { - closed: true, + let state_schema = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, @@ -350,36 +368,44 @@ fn stored_series_readouts_support_filters_and_sets() { BinaryOpKind::Set(PromQLVectorSetOpKind::And), BinaryOpKind::Set(PromQLVectorSetOpKind::Or), ] { - let nodes = (0..5) - .map(|id| PostAsapDAGNode { - id: PostAsapNodeId(id), - payload: match id { - 0 | 1 => PostAsapOperatorPayload::SummaryMerge, - 2 | 3 => PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, + let nodes = + (0..5) + .map(|id| PhysicalASAPDAGNode { + id, + payload: match id { + 0 | 1 => PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryMerge { + children: vec![], + }), + 2 | 3 => PhysicalASAPOperatorPayload::ASAP( + ASAPOp::FinalizeExactAccumulator { child: id - 2 }, + ), + _ => PhysicalASAPOperatorPayload::NonASAP( + planner_types::ir::NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: kind.clone(), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: 2, + rhs: 3, + }, + ), }, - _ => PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: kind.clone(), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, + output_state: if id < 2 { + ExecutionDataState::INGESTION_SUMMARY + } else { + ExecutionDataState::QUERY_ROWS }, - }, - output_state: if id < 2 { - ExecutionDataState::INGESTION_SUMMARY - } else { - ExecutionDataState::QUERY_ROWS - }, - output_schema: if id < 2 { - (*state_schema).clone() - } else { - value_schema.clone() - }, - guarantee: None, - }) - .collect::>(); + output_schema: if id < 2 { + (*state_schema).clone() + } else { + value_schema.clone() + }, + guarantee: None, + }) + .collect::>(); let edges = [ (0, 2, EdgeRole::Input), (1, 3, EdgeRole::Input), @@ -387,20 +413,20 @@ fn stored_series_readouts_support_filters_and_sets() { (3, 4, EdgeRole::Right), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + .map(|(producer, consumer, role)| PhysicalASAPDAGEdge { + producer, + consumer, role, - intermediate_schema: nodes[producer as usize].output_schema.clone(), - data_state: nodes[producer as usize].output_state, + intermediate_schema: nodes[producer].output_schema.clone(), + data_state: nodes[producer].output_state, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes, edges, - root: PostAsapNodeId(4), + roots: vec![4], }; let physical_dag = compile( &dag, diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index 45b3526f6..49a8a31f8 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -1,27 +1,34 @@ //! A retained PromQL sub-DAG (`Fallback`) compiles from its typed expression. //! The deployment supplies only its selector's raw series; expected values are //! hand-computed with Prometheus semantics. +mod common; use asap_physical_operators::{ operators::Operator, physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDAG, InputContract}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_physical_asap_dag; use futures::{executor::block_on, StreamExt}; +use planner_types::ir::physical_export::PhysicalASAPDAG; use planner_types::{ - post_asap::{execution_data_state::lift_plain, *}, - pre_asap::QueryExpr, - types::AccuracyTarget, - workload::*, + post_asap::execution_data_state::lift_plain, types::AccuracyTarget, workload::*, }; use std::{collections::BTreeMap, rc::Rc}; /// Bare selectors look back one ingestion interval: 60s. -fn parse(query: &str) -> QueryExpr { +fn parse(query: &str) -> Rc { parse_with(query, AccuracyTarget::Exact) } -fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn parse_with(query: &str, accuracy: AccuracyTarget) -> Rc { + match parse_root(query, accuracy) { + planner_types::ir::QueryRoot::Operator(node) => node, + _ => panic!("expected operator query"), + } +} + +fn parse_root(query: &str, accuracy: AccuracyTarget) -> planner_types::ir::QueryRoot { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -46,24 +53,18 @@ fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { ..Default::default() }), }; - asap_frontend_promql::lower_promql_workload(&workload, 0) + asap_frontend_promql::lower_promql_query_workload(&workload, 0) .unwrap() .remove(0) } -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { promql_rows::with_series_identity(&parse(query)).unwrap() } /// The whole query retained as one pre-ASAP node. -fn fallback_dag(expression: QueryExpr) -> PostAsapDAG { - let schema = lift_plain(&expression.output_schema().unwrap()); - compile_post_asap_dag(&Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(expression)), - schema, - guarantee: None, - })) - .unwrap() +fn fallback_dag(expression: Rc) -> PhysicalASAPDAG { + compile_physical_asap_dag(&expression).unwrap() } /// `(labels, seconds, value)`. `labels` is `k=v,...`, or a bare `job` value. @@ -82,13 +83,14 @@ fn labels(spec: &str) -> BTreeMap { } /// The metric a selector reads. -fn metric(selector: &QueryExpr) -> String { - match selector { - QueryExpr::Scan { +fn metric(selector: &planner_types::ir::OperatorNode) -> String { + match selector.expect_non_asap() { + planner_types::ir::NonASAPOp::Scan { source: planner_types::pre_asap::Source::TimeSeries { metric }, .. } => metric.clone(), - QueryExpr::TimeRange { child, .. } | QueryExpr::TimeShift { child, .. } => metric(child), + planner_types::ir::NonASAPOp::TimeRange { child, .. } + | planner_types::ir::NonASAPOp::TimeShift { child, .. } => metric(child), other => panic!("not a selector: {other:?}"), } } @@ -99,8 +101,11 @@ fn compile_query(query: &str) -> Result { } /// Compile a DAG whose root is the Fallback computing `expression`. -fn compile_dag(expression: &QueryExpr, dag: &PostAsapDAG) -> Result { - let root = u64::from(dag.root.0); +fn compile_dag( + expression: &planner_types::ir::OperatorNode, + dag: &PhysicalASAPDAG, +) -> Result { + let root = dag.roots[0] as u64; let inputs = promql_fallback::raw_series(expression) .map_err(|e| e.to_string())? .into_iter() @@ -124,14 +129,26 @@ fn evaluate( metrics: &[(&str, &[Sample])], at: i64, ) -> Result, i64, f64)>, String> { - let expression = lower(query); - evaluate_dag(&expression, &fallback_dag(expression.clone()), metrics, at) + match parse_root(query, AccuracyTarget::Exact) { + planner_types::ir::QueryRoot::Operator(expression) => { + let expression = + promql_rows::with_series_identity(&expression).map_err(|e| e.to_string())?; + evaluate_dag(&expression, &fallback_dag(expression.clone()), metrics, at) + } + planner_types::ir::QueryRoot::Scalar(expr) => { + let expr = expr + .map_operator_refs(&mut |node| promql_rows::with_series_identity(node).unwrap()); + let (program, selectors) = + promql_fallback::compile_scalar_root(&expr).map_err(|e| e.to_string())?; + execute_program(program, selectors, metrics, at, None) + } + } } #[allow(clippy::type_complexity)] fn evaluate_dag( - expression: &QueryExpr, - dag: &PostAsapDAG, + expression: &planner_types::ir::OperatorNode, + dag: &PhysicalASAPDAG, metrics: &[(&str, &[Sample])], at: i64, ) -> Result, i64, f64)>, String> { @@ -140,15 +157,26 @@ fn evaluate_dag( #[allow(clippy::type_complexity)] fn evaluate_dag_with_range( - expression: &QueryExpr, - dag: &PostAsapDAG, + expression: &planner_types::ir::OperatorNode, + dag: &PhysicalASAPDAG, metrics: &[(&str, &[Sample])], at: i64, bounds: Option<(i64, i64)>, ) -> Result, i64, f64)>, String> { let program = compile_dag(expression, dag)?; - let mut sources = BTreeMap::new(); let selectors = promql_fallback::raw_series(expression).unwrap(); + execute_program(program, selectors, metrics, at, bounds) +} + +#[allow(clippy::type_complexity)] +fn execute_program( + program: CompiledPhysicalDAG, + selectors: Vec, + metrics: &[(&str, &[Sample])], + at: i64, + bounds: Option<(i64, i64)>, +) -> Result, i64, f64)>, String> { + let mut sources = BTreeMap::new(); for (i, (selector, schema)) in selectors.into_iter().enumerate() { let name = metric(&selector); let rows = metrics @@ -423,12 +451,15 @@ fn dense_subquery_grids_are_rejected() { fn raw_series_contract_is_explicit() { let expression = lower("rate(m[5m])"); let dag = fallback_dag(expression.clone()); - let root = u64::from(dag.root.0); + let root = dag.roots[0] as u64; let [(selector, schema)] = promql_fallback::raw_series(&expression) .unwrap() .try_into() .unwrap(); - assert!(matches!(selector, QueryExpr::TimeRange { .. })); + assert!(matches!( + selector.expect_non_asap(), + planner_types::ir::NonASAPOp::TimeRange { .. } + )); let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); assert!(missing.to_string().contains("raw series input")); let mut wrong = (*schema).clone(); @@ -445,54 +476,27 @@ fn raw_series_contract_is_explicit() { // A consumed bare selector is raw range rows for its consumer; it is not // turned into instant selection. let selector = lower("m"); - let schema = lift_plain(&selector.output_schema().unwrap()); - let node = |id, payload| PostAsapDAGNode { - id: PostAsapNodeId(id), - payload, - output_state: ExecutionDataState::QUERY_ROWS, - output_schema: schema.clone(), - guarantee: None, - }; - let consumed = PostAsapDAG { - nodes: vec![ - node( - 0, - PostAsapOperatorPayload::Fallback { - expression: selector.clone(), - }, - ), - node( - 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, - offset: 0, - partition_by: Default::default(), - }, - }, - ), - ], - edges: vec![PostAsapDAGEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), - role: EdgeRole::Input, - intermediate_schema: schema.clone(), - data_state: ExecutionDataState::QUERY_ROWS, - grouping: GroupingEdgeCompatibility::NotApplicable, - window: WindowEdgeCompatibility::NotApplicable, - }], - root: PostAsapNodeId(1), - }; + let _schema = lift_plain(&selector.schema.clone()); + let consumed = planner_types::ir::OperatorNode::new_shared( + planner_types::ir::Operator::NonASAP(planner_types::ir::NonASAPOp::Limit { + n: Some(1), + offset: 0, + partition_by: Default::default(), + child: selector.clone(), + }), + ) + .unwrap(); + let consumed = fallback_dag(consumed.clone()); let raw = promql_fallback::raw_series(&selector).unwrap().remove(0).1; assert!(compile( &consumed, BTreeMap::from([( - promql_fallback::raw_series_input(0, 0), + promql_fallback::raw_series_input(consumed.roots[0] as u64, 0), InputContract::bounded(raw) )]), - &[1], + &consumed.roots.iter().map(|r| *r as u64).collect::>() ) - .is_err()); + .is_ok()); // Implicit subquery resolution belongs to the deployment's evaluation interval. assert!(compile_query("max_over_time(m[5m:])").is_err()); } @@ -1231,9 +1235,8 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { "histogram_quantile(0.5, x_bucket)", "histogram_quantile(0.5, sum by (le, job) (x_bucket))", ] { - let root = Rc::new( - promql_rows::with_series_identity(&parse_with(query, target.clone())).unwrap(), - ); + let root = + promql_rows::with_series_identity(&parse_with(query, target.clone())).unwrap(); let space = search_workload_with_targets( vec![(query, root.clone(), Some(target.clone()))], &default_strategies(), @@ -1243,8 +1246,7 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { let candidates = &space.candidates_for_target(planned).unwrap().candidates; assert!( candidates.iter().all(|c| matches!(&c.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::KeepPreAsap(e) if **e == *root))), + Replacement::SubDAG(node) if !node.contains_asap() && node.operator == root.operator)), "{query}: {candidates:?}" ); let selected = space @@ -1252,7 +1254,7 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { .assemble_selected_dag(planned) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let dag = compile_physical_asap_dag(&selected).unwrap(); let rows = evaluate_dag(&root, &dag, &[("x_bucket", &samples)], 60).unwrap(); let values: Vec<_> = rows.iter().map(|(_, _, v)| *v).collect(); assert_eq!(values, vec![1.75], "{query} {target:?}"); @@ -1285,7 +1287,7 @@ fn nonfinite_literals_round_trip_in_plans() { ] { let expression = lower(query); let json = serde_json::to_vec(&expression).unwrap(); - let restored: QueryExpr = serde_json::from_slice(&json).unwrap(); + let restored: Rc = serde_json::from_slice(&json).unwrap(); let result = evaluate_dag(&restored, &fallback_dag(restored.clone()), &[], 60).unwrap(); assert_eq!(result.len(), 1); if expected.is_nan() { @@ -1579,7 +1581,7 @@ fn subquery_label_uniqueness_is_checked_per_evaluation_step() { #[test] fn logical_nonfinite_quantile_parameter_round_trips() { let expression = lower("histogram_quantile(NaN, x_bucket)"); - let restored: QueryExpr = + let restored: Rc = serde_json::from_slice(&serde_json::to_vec(&expression).unwrap()).unwrap(); let samples = buckets(&[("job=a", HISTOGRAM)]); let result = evaluate_dag( @@ -1592,3 +1594,56 @@ fn logical_nonfinite_quantile_parameter_round_trips() { assert_eq!(result.len(), 1); assert!(result[0].2.is_nan()); } + +/// The proposal's pointwise projections preserve names only for unary minus. +#[test] +fn pointwise_projection_names_and_dynamic_parameters() { + let samples = [("job=a", 300, -2.5)]; + assert_eq!( + labeled("-m", &[("m", &samples)], 300), + [("__name__=m,job=a".into(), 2.5)] + ); + assert_eq!( + labeled("abs(m)", &[("m", &samples)], 300), + [("job=a".into(), 2.5)] + ); + assert_eq!( + run("round(m, scalar(vector(2)))", &samples, 300).unwrap(), + [("a".into(), 300_000, -2.0)] + ); + assert_eq!( + run("clamp(m, time()-301, time())", &samples, 300).unwrap(), + [("a".into(), 300_000, -1.0)] + ); + assert!(run("clamp(m, 2, 1)", &samples, 300).unwrap().is_empty()); + assert_eq!( + run("year(m)", &[("a", 300, 0.0)], 300).unwrap(), + [("a".into(), 300_000, 1970.0)] + ); + assert_eq!( + run("hour()", &[], 3600).unwrap(), + [("".into(), 3_600_000, 1.0)] + ); +} + +/// Execute every PromQL root/conversion example in the scalar design document. +#[test] +fn scalar_design_document_examples_execute() { + let samples = [("job=a", 300, 1.0), ("job=b", 300, 2.0)]; + for (query, expected) in [ + ("2", 2.0), + ("time()", 300.0), + ("vector(time())", 300.0), + ("scalar(sum(up)) + 1", 4.0), + ] { + let root = parse_root(query, AccuracyTarget::Exact); + root.validate_structure().unwrap(); + let output = evaluate(query, &[("up", &samples)], 300).unwrap(); + assert_eq!(output.len(), 1, "{query}"); + assert_eq!(output[0].2, expected, "{query}"); + } + assert_eq!( + labeled("up * 2", &[("up", &samples)], 300), + [("job=a".into(), 2.0), ("job=b".into(), 4.0)] + ); +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index e7deb78b1..35b9f2c0b 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -1,4 +1,6 @@ //! Compile, persist and rebind dynamic-label computation without deployment lowering. +use asap_physical_operators::expressions::binary::{BinaryOpKind, BinaryOperator}; + use asap_physical_operators::{ operators::Operator, physical_planner::{promql_values::*, CompiledPhysicalDAG, Source}, @@ -7,6 +9,7 @@ use asap_physical_operators::{ }; use futures::{executor::block_on, StreamExt}; use planner_types::pre_asap::{AggIntent, ColumnRef, GroupKeys}; + use std::collections::BTreeMap; fn row(labels: &[(&str, &str)], value: f64) -> Vec { @@ -213,10 +216,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { runtime::{Input, OutputStream}, values::SchemaRef, }; - use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{ArithmeticOpKind, BinaryOpKind}, - }; + use planner_types::pre_asap::ArithmeticOpKind; struct Counted { source: Operator, starts: std::rc::Rc>, @@ -325,10 +325,7 @@ fn compiled_constant_needs_no_deployment_source() { // arithmetic or bool comparisons remove the metric name. #[test] fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { - use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}, - }; + use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind}; for left_scalar in [false, true] { for names in [["a", "a"], ["a", "b"]] { for (kind, return_bool) in [ @@ -398,10 +395,10 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { ); } -// Persisted exact readout DAGs, rather than the storage adapter, merge panes, +// Persisted exact evaluation graphs, rather than the storage adapter, merge panes, // finalize each population, and preserve the requested metric-name semantics. #[test] -fn exact_state_readouts_recover_and_finalize_panes() { +fn exact_state_evaluations_recover_and_finalize_panes() { use asap_physical_operators::factory::create_planner_accumulator; use planner_types::post_asap::*; use std::sync::Arc; @@ -436,7 +433,7 @@ fn exact_state_readouts_recover_and_finalize_panes() { }) .collect(); let output = run_inputs( - compile_exact_readout(family.clone(), 60_000, preserve).unwrap(), + compile_exact_evaluation(family.clone(), 60_000, preserve).unwrap(), vec![Batch::try_new(exact_state_schema(family.clone()).unwrap(), rows).unwrap()], ) .unwrap(); @@ -483,7 +480,7 @@ fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { }) .collect(); let output = run_inputs( - compile_exact_readout(family.clone(), 60_000, false).unwrap(), + compile_exact_evaluation(family.clone(), 60_000, false).unwrap(), vec![Batch::try_new(exact_state_schema(family).unwrap(), rows).unwrap()], ) .unwrap(); diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index cf56a8682..41f222312 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -6,26 +6,33 @@ use asap_physical_operators::dag::{ Error, Limits, OutputStream, RunContext, Scope, }; use futures::{executor::block_on, stream, StreamExt}; -use planner_types::pre_asap::Schema; +use planner_types::ir::physical_export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::NonASAPOp; +use planner_types::ir::Predicate; use planner_types::{ post_asap::*, - pre_asap::{DataType, Field, GroupKeys, Predicate, QueryExpr, Source}, + pre_asap::{DataType, Field, GroupKeys, Source}, }; use std::{ collections::BTreeMap, - rc::Rc, sync::{ atomic::{AtomicUsize, Ordering}, Arc, }, }; -fn fixture() -> (QueryExpr, SchemaRef, Vec) { - let schema = - planner_types::pre_asap::Schema::new(vec![Field::plain("value", DataType::Int64, true)]); - let output = Arc::new(Schema { - closed: true, +fn fixture() -> (planner_types::ir::NonASAPOp, SchemaRef, Vec) { + let schema = planner_types::pre_asap::Schema::new(vec![planner_types::pre_asap::Field::plain( + "value", + DataType::Int64, + true, + )]); + let output = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -34,13 +41,13 @@ fn fixture() -> (QueryExpr, SchemaRef, Vec) { }], time_index: None, }); - let scan = QueryExpr::Scan { + let scan = planner_types::ir::NonASAPOp::Scan { source: Source::Table { table_ref: "numbers".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( - QueryExpr::Column(0), - ))))], + predicates: vec![Predicate(planner_types::ir::ScalarExpr::IsNotNull( + Box::new(planner_types::ir::ScalarExpr::Column(0)), + ))], schema, }; let batches = vec![ @@ -57,52 +64,59 @@ fn fixture() -> (QueryExpr, SchemaRef, Vec) { ]; (scan, output, batches) } -fn plan(scan: QueryExpr, schema: &SchemaRef, state: ExecutionDataState) -> PostAsapDAG { - let node = |id, payload| PostAsapDAGNode { - id: PostAsapNodeId(id), +fn plan( + scan: planner_types::ir::NonASAPOp, + schema: &SchemaRef, + state: ExecutionDataState, +) -> PhysicalASAPDAG { + let node = |id, payload| PhysicalASAPDAGNode { + id, payload, output_state: state, output_schema: (**schema).clone(), guarantee: None, }; - let edge = |producer, consumer| PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + let edge = |producer, consumer| PhysicalASAPDAGEdge { + producer, + consumer, role: EdgeRole::Input, intermediate_schema: (**schema).clone(), data_state: state, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }; - PostAsapDAG { + PhysicalASAPDAG { nodes: vec![ - node(0, PostAsapOperatorPayload::Fallback { expression: scan }), + node( + 0, + PhysicalASAPOperatorPayload::NonASAP( + scan.map_children(|_| -> usize { panic!("no plan refs") }), + ), + ), node( 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Sort { - keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(0), - ascending: false, - nulls_first: false, - }], - partition_by: GroupKeys::by(vec![]), - }, - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Sort { + keys: vec![planner_types::ir::SortKey { + expr: planner_types::ir::ScalarExpr::Column(0), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::by(vec![]), + child: 0, + }), ), node( 2, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 2, - offset: 0, - partition_by: GroupKeys::by(vec![]), - }, - }, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Limit { + n: Some(2), + offset: 0, + partition_by: GroupKeys::by(vec![]), + child: 1, + }), ), ], edges: vec![edge(0, 1), edge(1, 2)], - root: PostAsapNodeId(2), + roots: vec![2], } } fn registry(source: Arc) -> DataSources { @@ -223,17 +237,27 @@ fn lazy_open_shared_producer_and_cancellation() { #[test] fn binding_errors_and_reader_errors_are_not_empty_results() { let (mut scan, schema, _) = fixture(); - assert!(DataSources::default().bind(&scan).is_err()); + assert!(DataSources::default() + .bind(&planner_types::ir::OperatorNode::with_schema( + planner_types::ir::Operator::NonASAP(scan.clone()), + scan.output_schema().unwrap() + )) + .is_err()); let opened = Arc::new(AtomicUsize::new(0)); let sources = registry(Arc::new(CountingSource { schema: schema.clone(), opened: opened.clone(), fail: true, })); - if let QueryExpr::Scan { predicates, .. } = &mut scan { - predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + if let planner_types::ir::NonASAPOp::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(planner_types::ir::ScalarExpr::Column(0))); } - assert!(sources.bind(&scan).is_err()); + assert!(sources + .bind(&planner_types::ir::OperatorNode::with_schema( + planner_types::ir::Operator::NonASAP(scan.clone()), + scan.output_schema().unwrap() + )) + .is_err()); assert_eq!(opened.load(Ordering::SeqCst), 0); let (scan, _, _) = fixture(); let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); @@ -297,19 +321,23 @@ fn schema_drift_and_memory_limits_fail_the_scan() { #[test] fn empty_sources_and_three_valued_predicates() { use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let (mut scan, schema, batches) = fixture(); - if let QueryExpr::Scan { + if let planner_types::ir::NonASAPOp::Scan { predicates, source, .. } = &mut scan { *source = Source::TimeSeries { metric: "samples".into(), }; - *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + *predicates = vec![Predicate(planner_types::ir::ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(planner_types::ir::ScalarExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), - }))]; + right: Box::new(planner_types::ir::ScalarExpr::Literal(ScalarValue::Int64( + 2, + ))), + })]; } for (batches, expected) in [(vec![], 0), (batches, 2)] { let mut sources = DataSources::default(); diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index 8cac1cbda..65acba34a 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -8,10 +8,15 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; +use planner_types::ir::physical_export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, +}; +use planner_types::ir::ASAPOp; +use planner_types::ir::NonASAPOp; use planner_types::{ post_asap::*, - pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, + pre_asap::{ColumnRef, DataType}, }; use std::{collections::BTreeMap, sync::Arc}; @@ -20,9 +25,9 @@ use std::{collections::BTreeMap, sync::Arc}; #[test] fn post_asap_summary_projection_survives_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, @@ -39,9 +44,9 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }); - let output = Schema { - closed: true, + let output = planner_types::pre_asap::Schema { unique_keys: vec![], + closed: false, fields: vec![ schema.fields[1].clone(), Field { @@ -51,44 +56,45 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }; - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes: vec![ - PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + PhysicalASAPDAGNode { + id: 0, + payload: PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryMerge { + children: vec![], + }), output_schema: (*schema).clone(), output_state: ExecutionDataState::INGESTION_SUMMARY, guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::Project { - cols: vec![1, 0] - .into_iter() - .map(|index| ProjectItem { - alias: None, - expr: QueryExpr::Column(index), - }) - .collect(), - qualifier: None, - }, - }, + PhysicalASAPDAGNode { + id: 1, + payload: PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Project { + cols: vec![1, 0] + .into_iter() + .map(|index| planner_types::ir::ProjectItem { + alias: None, + expr: planner_types::ir::ScalarExpr::Column(index), + }) + .collect(), + qualifier: None, + child: 0, + }), output_schema: output.clone(), output_state: ExecutionDataState::INGESTION_SUMMARY, guarantee: None, }, ], - edges: vec![PostAsapDAGEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + edges: vec![PhysicalASAPDAGEdge { + producer: 0, + consumer: 1, role: EdgeRole::Input, intermediate_schema: (*schema).clone(), data_state: ExecutionDataState::INGESTION_SUMMARY, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }], - root: PostAsapNodeId(1), + roots: vec![1], }; let program = compile( &dag, diff --git a/crates/asap-physical-operators/tests/unified_promql_fallback.rs b/crates/asap-physical-operators/tests/unified_promql_fallback.rs deleted file mode 100644 index fe1a39140..000000000 --- a/crates/asap-physical-operators/tests/unified_promql_fallback.rs +++ /dev/null @@ -1,1611 +0,0 @@ -//! A retained PromQL sub-DAG (`Fallback`) compiles from its typed expression. -//! The deployment supplies only its selector's raw series; expected values are -//! hand-computed with Prometheus semantics. -#[path = "unified_common/mod.rs"] -mod common; -use asap_physical_operators::{ - operators::Operator, - runtime::{Limits, RunContext, Scope}, - unified_physical_planner::{ - compile, promql_fallback, promql_rows, CompiledPhysicalDAG, InputContract, - }, - values::{Batch, Value}, -}; -use common::compile_physical_asap_dag; -use futures::{executor::block_on, StreamExt}; -use planner_types::ir::physical_export::PhysicalASAPDAG; -use planner_types::{ - post_asap::execution_data_state::lift_plain, types::AccuracyTarget, workload::*, -}; -use std::{collections::BTreeMap, rc::Rc}; - -/// Bare selectors look back one ingestion interval: 60s. -fn parse(query: &str) -> Rc { - parse_with(query, AccuracyTarget::Exact) -} - -fn parse_with(query: &str, accuracy: AccuracyTarget) -> Rc { - match parse_root(query, accuracy) { - planner_types::ir::QueryRoot::Operator(node) => node, - _ => panic!("expected operator query"), - } -} - -fn parse_root(query: &str, accuracy: AccuracyTarget) -> planner_types::ir::QueryRoot { - let workload = PlanningWorkload { - query_workload: QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: Some(vec![BatchEntry { - query: Query(query.into()), - requirements: QueryRequirements { - accuracy: AccuracyRequirement::Explicit(accuracy), - ..Default::default() - }, - predictability: Predictability::Unknown, - invocations: 1, - execute_at: None, - time_selection: TimeSelection::default(), - }]), - repeating_queries: None, - }, - data_workload: Some(DataWorkload { - data_ingestion_interval: Evidence { - value: Some(DurationMs(60_000)), - ..Default::default() - }, - ..Default::default() - }), - }; - asap_frontend_promql::unified::lower_promql_query_workload(&workload, 0) - .unwrap() - .remove(0) -} - -fn lower(query: &str) -> Rc { - promql_rows::with_series_identity(&parse(query)).unwrap() -} - -/// The whole query retained as one pre-ASAP node. -fn fallback_dag(expression: Rc) -> PhysicalASAPDAG { - compile_physical_asap_dag(&expression).unwrap() -} - -/// `(labels, seconds, value)`. `labels` is `k=v,...`, or a bare `job` value. -type Sample = (&'static str, i64, f64); - -fn labels(spec: &str) -> BTreeMap { - if !spec.contains('=') { - return BTreeMap::from([("job".into(), spec.into())]); - } - spec.split(',') - .map(|pair| { - let (k, v) = pair.split_once('=').unwrap(); - (k.to_string(), v.to_string()) - }) - .collect() -} - -/// The metric a selector reads. -fn metric(selector: &planner_types::ir::OperatorNode) -> String { - match selector.expect_non_asap() { - planner_types::ir::NonASAPOp::Scan { - source: planner_types::pre_asap::Source::TimeSeries { metric }, - .. - } => metric.clone(), - planner_types::ir::NonASAPOp::TimeRange { child, .. } - | planner_types::ir::NonASAPOp::TimeShift { child, .. } => metric(child), - other => panic!("not a selector: {other:?}"), - } -} - -fn compile_query(query: &str) -> Result { - let expression = lower(query); - compile_dag(&expression, &fallback_dag(expression.clone())) -} - -/// Compile a DAG whose root is the Fallback computing `expression`. -fn compile_dag( - expression: &planner_types::ir::OperatorNode, - dag: &PhysicalASAPDAG, -) -> Result { - let root = dag.roots[0] as u64; - let inputs = promql_fallback::raw_series(expression) - .map_err(|e| e.to_string())? - .into_iter() - .enumerate() - .map(|(i, (_, schema))| { - ( - promql_fallback::raw_series_input(root, i), - InputContract::bounded(schema), - ) - }) - .collect(); - let program = compile(dag, inputs, &[root]).map_err(|e| e.to_string())?; - Ok(serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap()) -} - -/// Evaluate at `at` seconds over samples of each named metric; returns -/// `(output labels, timestamp ms, value)` rows in order. -#[allow(clippy::type_complexity)] -fn evaluate( - query: &str, - metrics: &[(&str, &[Sample])], - at: i64, -) -> Result, i64, f64)>, String> { - match parse_root(query, AccuracyTarget::Exact) { - planner_types::ir::QueryRoot::Operator(expression) => { - let expression = - promql_rows::with_series_identity(&expression).map_err(|e| e.to_string())?; - evaluate_dag(&expression, &fallback_dag(expression.clone()), metrics, at) - } - planner_types::ir::QueryRoot::Scalar(expr) => { - let expr = expr - .map_operator_refs(&mut |node| promql_rows::with_series_identity(node).unwrap()); - let (program, selectors) = - promql_fallback::compile_scalar_root(&expr).map_err(|e| e.to_string())?; - execute_program(program, selectors, metrics, at, None) - } - } -} - -#[allow(clippy::type_complexity)] -fn evaluate_dag( - expression: &planner_types::ir::OperatorNode, - dag: &PhysicalASAPDAG, - metrics: &[(&str, &[Sample])], - at: i64, -) -> Result, i64, f64)>, String> { - evaluate_dag_with_range(expression, dag, metrics, at, None) -} - -#[allow(clippy::type_complexity)] -fn evaluate_dag_with_range( - expression: &planner_types::ir::OperatorNode, - dag: &PhysicalASAPDAG, - metrics: &[(&str, &[Sample])], - at: i64, - bounds: Option<(i64, i64)>, -) -> Result, i64, f64)>, String> { - let program = compile_dag(expression, dag)?; - let selectors = promql_fallback::raw_series(expression).unwrap(); - execute_program(program, selectors, metrics, at, bounds) -} - -#[allow(clippy::type_complexity)] -fn execute_program( - program: CompiledPhysicalDAG, - selectors: Vec, - metrics: &[(&str, &[Sample])], - at: i64, - bounds: Option<(i64, i64)>, -) -> Result, i64, f64)>, String> { - let mut sources = BTreeMap::new(); - for (i, (selector, schema)) in selectors.into_iter().enumerate() { - let name = metric(&selector); - let rows = metrics - .iter() - .filter(|(m, _)| *m == name) - .flat_map(|(_, samples)| samples.iter()) - .map(|(spec, seconds, value)| { - let mut labels = labels(spec); - // A sample may supply its own `__name__`, as a series of another metric. - labels.entry("__name__".into()).or_insert(name.clone()); - promql_rows::series_row(&schema, &labels, seconds * 1000, *value).unwrap() - }) - .collect(); - let batch = Batch::try_new(schema.clone(), rows).unwrap(); - sources.insert( - promql_fallback::raw_series_input(program.roots()[0], i), - Box::new(Operator::source(schema, vec![batch]).unwrap()) as _, - ); - } - let dag = program.instantiate(sources).map_err(|e| e.to_string())?; - let context = RunContext::new( - Scope::Query { - evaluation_time_ms: at * 1000, - revision: 0, - }, - Limits::default(), - ) - .unwrap(); - let context = match bounds { - Some((start, end)) => context - .with_query_range(start * 1000, end * 1000) - .map_err(|e| e.to_string())?, - None => context, - }; - block_on(async { - let mut stream = dag - .execute(program.roots(), context) - .map_err(|e| e.to_string())? - .remove(0); - let mut rows = Vec::new(); - while let Some(batch) = stream.next().await { - let batch = batch.map_err(|e| e.to_string())?; - let schema = batch.schema().clone(); - for row in batch.rows() { - let mut labels = BTreeMap::new(); - let mut time = -1; - let mut value = None; - for (field, cell) in schema.fields.iter().zip(row) { - match (field.name.as_str(), cell) { - (promql_rows::SERIES_IDENTITY_COLUMN, Value::Utf8(id)) => { - labels = promql_rows::decode_series_identity(id).unwrap() - } - (_, Value::Utf8(_) | Value::Null) => {} - (_, Value::Timestamp(t)) => time = *t, - (_, Value::Float64(v)) => value = Some(*v), - (_, Value::Int64(v)) => value = Some(*v as f64), - other => return Err(format!("unexpected cell {other:?}")), - } - } - if !schema - .fields - .iter() - .any(|f| f.name == promql_rows::SERIES_IDENTITY_COLUMN) - { - for (field, cell) in schema.fields.iter().zip(row) { - if let Value::Utf8(label) = cell { - if !label.is_empty() { - labels.insert(field.name.clone(), label.to_string()); - } - } - } - } - rows.push((labels, time, value.ok_or("missing value")?)); - } - } - Ok(rows) - }) -} - -/// Evaluate at `at` seconds over metric `m`; returns `(job or "", timestamp ms, value)`. -fn run(query: &str, samples: &[Sample], at: i64) -> Result, String> { - Ok(evaluate(query, &[("m", samples)], at)? - .into_iter() - .map(|(labels, time, value)| (labels.get("job").cloned().unwrap_or_default(), time, value)) - .collect()) -} - -/// Output rows as `(k=v,... sorted, value)`, including any `__name__`. -fn labeled(query: &str, metrics: &[(&str, &[Sample])], at: i64) -> Vec<(String, f64)> { - let mut rows = evaluate(query, metrics, at) - .unwrap_or_else(|e| panic!("{query}: {e}")) - .into_iter() - .map(|(labels, _, value)| { - let spec = labels - .iter() - .map(|(k, v)| format!("{k}={v}")) - .collect::>() - .join(","); - (spec, value) - }) - .collect::>(); - rows.sort_by(|a, b| a.0.cmp(&b.0)); - rows -} - -fn values(query: &str, samples: &[Sample], at: i64) -> Vec<(String, f64)> { - run(query, samples, at) - .unwrap_or_else(|e| panic!("{query}: {e}")) - .into_iter() - .map(|(job, _, value)| (job, value)) - .collect() -} - -fn one(query: &str, samples: &[Sample], at: i64) -> f64 { - match values(query, samples, at).as_slice() { - [(_, value)] => *value, - other => panic!("{query}: expected one sample, got {other:?}"), - } -} - -const COUNTER: &[Sample] = &[ - ("a", 60, 10.), - ("a", 120, 20.), - ("a", 180, 5.), - ("a", 240, 15.), -]; - -// rate/increase correct the reset at 180s and extrapolate half an interval at -// most; delta treats the same samples as a gauge. -#[test] -fn range_functions_follow_prometheus_extrapolation_and_resets() { - // Reset-corrected increase is 25 over 180s of samples; 60s on each side extrapolates. - let increase = 25. * (180. + 60. + 60.) / 180.; - assert!((one("increase(m[5m])", COUNTER, 300) - increase).abs() < 1e-9); - assert!((one("rate(m[5m])", COUNTER, 300) - increase / 300.).abs() < 1e-12); - let delta = 5. * (180. + 60. + 60.) / 180.; - assert!((one("delta(m[5m])", COUNTER, 300) - delta).abs() < 1e-9); - // Fewer than two samples yield no rate. - assert!(values("rate(m[2m])", COUNTER, 300).is_empty()); - for (query, expected) in [ - ("sum_over_time(m[5m])", 50.), - ("avg_over_time(m[5m])", 12.5), - ("min_over_time(m[5m])", 5.), - ("max_over_time(m[5m])", 20.), - ("count_over_time(m[5m])", 4.), - ] { - assert_eq!(one(query, COUNTER, 300), expected, "{query}"); - } -} - -// Ranges are left-open: a sample at `t - range` is excluded, one at `t` is included. -#[test] -fn ranges_exclude_their_start_and_offsets_shift_them() { - let samples = &[("a", 60, 1.), ("a", 90, 1.), ("a", 120, 1.), ("a", 150, 1.)]; - assert_eq!(one("count_over_time(m[1m])", samples, 120), 2.); - // offset 1m reads (60s, 120s] at 180s; output keeps the evaluation time. - let rows = run("count_over_time(m[1m] offset 1m)", samples, 180).unwrap(); - assert_eq!(rows, vec![("a".into(), 180_000, 2.)]); -} - -// A bare selector takes the latest sample within the lookback; a stale marker -// hides the series rather than exposing an older value. -#[test] -fn instant_selection_uses_lookback_and_stale_markers() { - let stale = f64::from_bits(0x7ff0_0000_0000_0002); - let samples = &[("a", 0, 1.), ("a", 30, 2.), ("b", 30, 3.), ("b", 50, stale)]; - assert_eq!(values("m", samples, 60), vec![("a".into(), 2.)]); - // The lookback (30s, 90s] excludes the sample at 30s. - assert!(values("m", samples, 90).is_empty()); - // Range functions skip stale markers. - assert_eq!( - values("sum_over_time(m[1m])", samples, 60), - vec![("a".into(), 2.), ("b".into(), 3.)] - ); -} - -// NaN samples follow Prometheus: min/max skip them, sums propagate them. -#[test] -fn nan_samples() { - let samples = &[("a", 10, f64::NAN), ("a", 20, 3.), ("a", 30, 1.)]; - assert_eq!(one("max_over_time(m[1m])", samples, 60), 3.); - assert_eq!(one("min_over_time(m[1m])", samples, 60), 1.); - assert!(one("sum_over_time(m[1m])", samples, 60).is_nan()); -} - -// Aggregation over no series is an empty vector, not one zero or null row; -// sort_desc orders the selected series. -#[test] -fn cross_series_aggregates_and_empty_inputs() { - let samples = &[("a", 50, 1.), ("b", 40, 2.), ("b", 55, 4.)]; - assert_eq!(values("sum(m)", samples, 60), vec![(String::new(), 5.)]); - assert_eq!(values("count(m)", samples, 60), vec![(String::new(), 2.)]); - assert_eq!( - values("max by (job) (m)", samples, 60), - vec![("a".into(), 1.), ("b".into(), 4.)] - ); - assert_eq!( - values("sort_desc(m)", samples, 60), - vec![("b".into(), 4.), ("a".into(), 1.)] - ); - // topk by (job) keeps the top series of each job, not one overall. - let jobs = &[("a", 50, 1.), ("b", 50, 2.)]; - let mut top = values("topk by (job) (1, m)", jobs, 60); - top.sort_by(|x, y| x.0.cmp(&y.0)); - assert_eq!(top, vec![("a".into(), 1.), ("b".into(), 2.)]); - assert_eq!(values("topk(1, m)", jobs, 60), vec![("b".into(), 2.)]); - for query in ["sum(m)", "count(m)", "max(m)", "sum by (job) (rate(m[5m]))"] { - assert!(values(query, &[], 60).is_empty(), "{query}"); - } -} - -// scalar() is the single series' value and NaN otherwise; vector() needs no input. -#[test] -fn scalar_and_vector_bridges() { - assert_eq!(one("scalar(m)", &[("a", 50, 7.)], 60), 7.); - assert!(one("scalar(m)", &[("a", 50, 7.), ("b", 50, 8.)], 60).is_nan()); - assert!(one("scalar(m)", &[], 60).is_nan()); - assert_eq!( - run("vector(3)", &[], 60).unwrap(), - vec![(String::new(), 60_000, 3.)] - ); - assert_eq!( - values("2 - m", &[("a", 50, 7.)], 60), - vec![("a".into(), -5.)] - ); - assert_eq!( - values("m * 2", &[("a", 50, 7.)], 60), - vec![("a".into(), 14.)] - ); -} - -// Subquery steps are absolute multiples of the resolution in the left-open -// range; each step evaluates the operand, and the outer function reduces them. -#[test] -fn subqueries_evaluate_their_operand_on_the_aligned_grid() { - // Steps 60..300: selections 1, 7, 3, (none at 240s), 4. - let samples = &[ - ("a", 50, 1.), - ("a", 110, 7.), - ("a", 170, 3.), - ("a", 290, 4.), - ]; - assert_eq!(one("max_over_time(m[5m:1m])", samples, 300), 7.); - assert_eq!(one("count_over_time(m[5m:1m])", samples, 300), 4.); - // At 190s the steps are 60, 120, 180 (not 70, 130, 190): counts 1 + 2 + 2. - let samples = &[("a", 30, 1.), ("a", 90, 1.), ("a", 150, 1.), ("a", 185, 1.)]; - assert_eq!( - one("sum_over_time(count_over_time(m[2m])[3m:1m])", samples, 190), - 5. - ); - // offset 1m moves the grid to (-50s, 130s]: steps 0, 60, 120 count 0 + 1 + 2. - assert_eq!( - one( - "sum_over_time(count_over_time(m[2m])[3m:1m] offset 1m)", - samples, - 190 - ), - 3. - ); -} - -// Subquery work is bounded by the query: at most 100000 steps. -#[test] -fn dense_subquery_grids_are_rejected() { - assert!(compile_query("max_over_time(m[100s:1ms])").is_ok()); - assert!(compile_query("max_over_time(m[30d:1ms])").is_err()); -} - -// The deployment must supply the selector's raw rows under the documented slot -// with the exact selector schema; unsupported shapes stay rejected. -#[test] -fn raw_series_contract_is_explicit() { - let expression = lower("rate(m[5m])"); - let dag = fallback_dag(expression.clone()); - let root = dag.roots[0] as u64; - let [(selector, schema)] = promql_fallback::raw_series(&expression) - .unwrap() - .try_into() - .unwrap(); - assert!(matches!( - selector.expect_non_asap(), - planner_types::ir::NonASAPOp::TimeRange { .. } - )); - let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); - assert!(missing.to_string().contains("raw series input")); - let mut wrong = (*schema).clone(); - wrong.fields.pop(); - let wrong = compile( - &dag, - BTreeMap::from([( - promql_fallback::raw_series_input(root, 0), - InputContract::bounded(std::sync::Arc::new(wrong)), - )]), - &[root], - ); - assert!(wrong.is_err()); - // A consumed bare selector is raw range rows for its consumer; it is not - // turned into instant selection. - let selector = lower("m"); - let _schema = lift_plain(&selector.schema.clone()); - let consumed = planner_types::ir::OperatorNode::new_shared( - planner_types::ir::Operator::NonASAP(planner_types::ir::NonASAPOp::Limit { - n: Some(1), - offset: 0, - partition_by: Default::default(), - child: selector.clone(), - }), - ) - .unwrap(); - let consumed = fallback_dag(consumed.clone()); - let raw = promql_fallback::raw_series(&selector).unwrap().remove(0).1; - assert!(compile( - &consumed, - BTreeMap::from([( - promql_fallback::raw_series_input(consumed.roots[0] as u64, 0), - InputContract::bounded(raw) - )]), - &consumed.roots.iter().map(|r| *r as u64).collect::>() - ) - .is_ok()); - // Implicit subquery resolution belongs to the deployment's evaluation interval. - assert!(compile_query("max_over_time(m[5m:])").is_err()); -} - -// irate/idelta use the last two samples (irate corrects a reset to the last -// value); changes/resets count value changes and decreases; quantile_over_time -// interpolates; all skip stale markers. -#[test] -fn instant_and_counting_range_functions() { - // COUNTER in (0s, 300s]: 10, 20, 5, 15. - assert!((one("irate(m[5m])", COUNTER, 300) - 10. / 60.).abs() < 1e-12); - assert_eq!(one("idelta(m[5m])", COUNTER, 300), 10.); - // At 200s the last pair 20 -> 5 is a reset: irate uses 5 as the increase. - assert!((one("irate(m[5m])", COUNTER, 200) - 5. / 60.).abs() < 1e-12); - assert_eq!(one("idelta(m[5m])", COUNTER, 200), -15.); - assert!(values("irate(m[1m])", COUNTER, 300).is_empty()); - assert_eq!(one("changes(m[5m])", COUNTER, 300), 3.); - assert_eq!(one("resets(m[5m])", COUNTER, 300), 1.); - assert_eq!(one("changes(m[2m])", COUNTER, 300), 0.); - // NaN to NaN is not a change; any other transition involving NaN is. - let flat = &[ - ("a", 10, 1.), - ("a", 20, 1.), - ("a", 30, 2.), - ("a", 40, f64::NAN), - ("a", 50, f64::NAN), - ("a", 55, 1.), - ]; - assert_eq!(one("changes(m[1m])", flat, 60), 3.); - let stale = f64::from_bits(0x7ff0_0000_0000_0002); - let ended = &[("a", 240, 15.), ("a", 250, stale)]; - assert_eq!(one("last_over_time(m[5m])", ended, 300), 15.); - assert!(values("m", ended, 300).is_empty()); - // Sorted 5, 10, 15, 20: rank 1.5 and 0.75; outside [0, 1] is +-Inf. - assert_eq!(one("quantile_over_time(0.5, m[5m])", COUNTER, 300), 12.5); - assert_eq!(one("quantile_over_time(0.25, m[5m])", COUNTER, 300), 8.75); - assert_eq!( - one("quantile_over_time(2, m[5m])", COUNTER, 300), - f64::INFINITY - ); - assert_eq!( - one("quantile_over_time(-1, m[5m])", COUNTER, 300), - f64::NEG_INFINITY - ); -} - -// `@ ` evaluates the selector or subquery at `t`, minus any offset, and the -// result keeps the query's evaluation time. -#[test] -fn at_modifier_fixes_the_evaluation_instant() { - let samples = &[("a", 60, 1.), ("a", 120, 2.), ("a", 180, 3.)]; - assert_eq!( - run("m @ 120", samples, 1000).unwrap(), - vec![("a".into(), 1_000_000, 2.)] - ); - assert!(values("m", samples, 1000).is_empty()); - assert_eq!(one("count_over_time(m[2m] @ 180)", samples, 1000), 2.); - assert_eq!(one("m @ 180 offset 1m", samples, 1000), 2.); - // The subquery grid is (60s, 180s]: steps 120 and 180 select 2 and 3. - assert_eq!(one("max_over_time(m[2m:1m] @ 180)", samples, 1000), 3.); - assert_eq!( - one("sum_over_time(m[2m:1m] @ 180 offset 1m)", samples, 1000), - 3. - ); - // An inner @ pins every step to the same instant. - assert_eq!(one("sum_over_time((m @ 60)[2m:1m])", samples, 180), 2.); - // start() and end() depend on the range query, which is the deployment's. - assert!(evaluate("m @ start()", &[], 60) - .unwrap_err() - .contains("query range bounds")); -} - -const A: &[Sample] = &[("job=x", 50, 10.), ("job=y", 50, 20.), ("job=w", 50, 0.)]; -const B: &[Sample] = &[("job=x", 50, 2.), ("job=z", 50, 5.), ("job=w", 50, 0.)]; - -// Vector-vector arithmetic matches series one-to-one on label sets without -// the metric name, and the result drops the metric name. -#[test] -fn vector_arithmetic_matches_label_sets() { - let metrics = &[("a", A), ("b", B)]; - let quotient = labeled("a / b", metrics, 60); - assert_eq!(quotient.len(), 2); - assert_eq!(quotient[0].0, "job=w"); - assert!(quotient[0].1.is_nan(), "0 / 0 is NaN"); - assert_eq!(quotient[1], ("job=x".into(), 5.)); - // Each selector reads its own raw rows, even a repeated metric. - assert_eq!( - labeled("(a - b) * a", metrics, 60), - vec![("job=w".into(), 0.), ("job=x".into(), 80.)] - ); - assert_eq!( - labeled("sum by (job) (a) - sum by (job) (b)", metrics, 60), - vec![("job=w".into(), 0.), ("job=x".into(), 8.)] - ); - // Rates of two counters over their own windows. - let up: &[Sample] = &[("job=x", 0, 0.), ("job=x", 60, 60.)]; - let down: &[Sample] = &[("job=x", 0, 0.), ("job=x", 60, 30.)]; - assert_eq!( - labeled("rate(a[2m]) / rate(b[2m])", &[("a", up), ("b", down)], 60), - vec![("job=x".into(), 2.)] - ); -} - -// on() keeps only the listed labels and ignoring() drops them; a duplicate -// match group is an error unless the left duplicates never match. -#[test] -fn on_and_ignoring_select_the_matching_labels() { - let a: &[Sample] = &[("job=x,inst=1", 50, 10.)]; - let b: &[Sample] = &[("job=x,inst=2", 50, 4.)]; - let metrics = &[("a", a), ("b", b)]; - assert!(labeled("a - b", metrics, 60).is_empty()); - assert_eq!( - labeled("a - on(job) b", metrics, 60), - vec![("job=x".into(), 6.)] - ); - assert_eq!( - labeled("a - ignoring(inst) b", metrics, 60), - vec![("job=x".into(), 6.)] - ); - let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, 2.)]; - let other: &[Sample] = &[("job=y", 50, 1.)]; - assert!(evaluate("a + on(job) b", &[("a", a), ("b", pair)], 60).is_err()); - assert!(evaluate("a + on(job) b", &[("a", pair), ("b", b)], 60).is_err()); - assert!(labeled("a + on(job) b", &[("a", pair), ("b", other)], 60).is_empty()); -} - -// without() groups by every label except the listed ones and the metric name. -#[test] -fn without_grouping_drops_labels_and_the_name() { - let a: &[Sample] = &[ - ("job=x,inst=1", 50, 1.), - ("job=x,inst=2", 50, 2.), - ("job=y,inst=1", 50, 4.), - ]; - let metrics = &[("a", a)]; - assert_eq!( - labeled("sum without (inst) (a)", metrics, 60), - vec![("job=x".into(), 3.), ("job=y".into(), 4.)] - ); - assert_eq!( - labeled("count without (inst) (a)", metrics, 60), - vec![("job=x".into(), 2.), ("job=y".into(), 1.)] - ); - assert_eq!( - labeled("max without (job, inst) (a)", metrics, 60), - vec![(String::new(), 4.)] - ); - assert!(labeled("sum without (inst) (a)", &[], 60).is_empty()); - // Series equal without the name share a group rather than colliding. - let named: &[Sample] = &[ - ("job=x,inst=1", 50, 1.), - ("__name__=b,job=x,inst=1", 50, 2.), - ]; - assert_eq!( - labeled("sum without (inst) (a)", &[("a", named)], 60), - vec![("job=x".into(), 3.)] - ); -} - -// Arithmetic with a literal drops the metric name; series that then share a -// label set are an error, as in Prometheus. -#[test] -fn literal_arithmetic_drops_the_name_and_rejects_equal_label_sets() { - let a: &[Sample] = &[ - ("job=x,inst=1", 50, 1.), - ("__name__=b,job=x,inst=2", 50, 2.), - ]; - assert_eq!( - labeled("a * 2", &[("a", a)], 60), - vec![("inst=1,job=x".into(), 2.), ("inst=2,job=x".into(), 4.)] - ); - let equal: &[Sample] = &[("job=x", 50, 1.), ("__name__=b,job=x", 50, 2.)]; - let error = evaluate("a * 2", &[("a", equal)], 60).unwrap_err(); - assert!(error.contains("same labelset"), "{error}"); -} - -// An empty label value is an absent label, and an empty side yields an empty -// result before any duplicate check, as in Prometheus. -#[test] -fn empty_labels_and_empty_sides_match_prometheus() { - let a: &[Sample] = &[("job=x,env=", 50, 3.)]; - let b: &[Sample] = &[("job=x", 50, 1.)]; - assert_eq!( - labeled("a + b", &[("a", a), ("b", b)], 60), - vec![("job=x".into(), 4.)] - ); - let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, 2.)]; - assert!(labeled("a + on(job) b", &[("b", pair)], 60).is_empty()); - assert!(labeled("b + on(job) a", &[("b", pair)], 60).is_empty()); - assert!(labeled("a - time()", &[], 60).is_empty()); -} - -// Sums and averages use Prometheus' Kahan-Neumaier compensation, and an -// average whose running sum overflows switches to an incremental mean. -#[test] -fn sums_and_averages_are_compensated_like_prometheus() { - let cancel = &[("a", 10, 1e100), ("a", 20, 1.), ("a", 30, -1e100)]; - assert_eq!(one("sum_over_time(m[1m])", cancel, 60), 1.); - assert_eq!(one("avg_over_time(m[1m])", cancel, 60), 1. / 3.); - let huge = &[("a", 10, 1.7e308), ("a", 20, 1.7e308)]; - assert_eq!(one("avg_over_time(m[1m])", huge, 60), 1.7e308); - assert_eq!(one("sum_over_time(m[1m])", huge, 60), f64::INFINITY); - let infinite = &[("a", 10, f64::INFINITY), ("a", 20, 1.)]; - assert_eq!(one("sum_over_time(m[1m])", infinite, 60), f64::INFINITY); - assert_eq!(one("avg_over_time(m[1m])", infinite, 60), f64::INFINITY); - let opposite = &[("a", 10, f64::INFINITY), ("a", 20, f64::NEG_INFINITY)]; - assert!(one("sum_over_time(m[1m])", opposite, 60).is_nan()); - assert!(one("avg_over_time(m[1m])", opposite, 60).is_nan()); - let cancel = &[("a", 50, 1e100), ("b", 50, 1.), ("c", 50, -1e100)]; - assert_eq!(one("sum(m)", cancel, 60), 1.); - assert_eq!(one("avg(m)", cancel, 60), 1. / 3.); - let huge = &[("a", 50, 1.7e308), ("b", 50, 1.7e308)]; - assert_eq!(one("avg(m)", huge, 60), 1.7e308); -} - -/// `(k=v,... sorted, value)` rows for readable expectations. -fn rows(pairs: &[(&str, f64)]) -> Vec<(String, f64)> { - let mut rows = pairs - .iter() - .map(|(spec, value)| (spec.to_string(), *value)) - .collect::>(); - rows.sort_by(|a, b| a.0.cmp(&b.0)); - rows -} - -/// `labeled`, with NaN values rendered comparable. -fn labeled_nan(query: &str, metrics: &[(&str, &[Sample])], at: i64) -> Vec<(String, String)> { - labeled(query, metrics, at) - .into_iter() - .map(|(labels, value)| (labels, format!("{value:?}"))) - .collect() -} - -const C: &[Sample] = &[ - ("job=x", 50, 10.), - ("job=y", 50, 20.), - ("job=w", 50, 0.), - ("job=n", 50, f64::NAN), -]; - -// A comparison with a scalar keeps the matching series with their value and -// metric name, whichever side the scalar is on; `bool` yields 1 or 0 for every -// series and drops the name. NaN compares unequal to everything. -#[test] -fn scalar_comparisons_filter_or_return_bool() { - let metrics = &[("a", C)]; - let kept = rows(&[("__name__=a,job=x", 10.), ("__name__=a,job=y", 20.)]); - assert_eq!(labeled("a > 5", metrics, 60), kept); - assert_eq!(labeled("5 < a", metrics, 60), kept); - assert_eq!( - labeled("a <= 10", metrics, 60), - rows(&[("__name__=a,job=w", 0.), ("__name__=a,job=x", 10.)]) - ); - assert_eq!( - labeled("a > bool 5", metrics, 60), - rows(&[("job=n", 0.), ("job=w", 0.), ("job=x", 1.), ("job=y", 1.)]) - ); - assert_eq!( - labeled("10 == bool a", metrics, 60), - rows(&[("job=n", 0.), ("job=w", 0.), ("job=x", 1.), ("job=y", 0.)]) - ); - // scalar() of no series is NaN. - assert_eq!(labeled("a != scalar(b)", metrics, 60).len(), 4); - assert!(labeled("a == scalar(b)", metrics, 60).is_empty()); - assert!(labeled("a > 5", &[], 60).is_empty()); - // Only `bool` drops the name, so only it can make label sets collide. - let equal: &[Sample] = &[("job=x", 50, 1.), ("__name__=b,job=x", 50, 2.)]; - assert_eq!(labeled("a > 0", &[("a", equal)], 60).len(), 2); - let error = evaluate("a > bool 0", &[("a", equal)], 60).unwrap_err(); - assert!(error.contains("same labelset"), "{error}"); -} - -// Vector comparisons match one-to-one like arithmetic. A filter keeps the -// left series, name included, unless `on` reduces its labels; `bool` drops the -// name. A left duplicate is an error only if more than one of it is kept. -#[test] -fn vector_comparisons_match_one_to_one() { - let metrics = &[("a", A), ("b", B)]; - assert_eq!( - labeled("a > b", metrics, 60), - rows(&[("__name__=a,job=x", 10.)]) - ); - assert_eq!( - labeled("a >= b", metrics, 60), - rows(&[("__name__=a,job=w", 0.), ("__name__=a,job=x", 10.)]) - ); - assert_eq!( - labeled("a > bool b", metrics, 60), - rows(&[("job=w", 0.), ("job=x", 1.)]) - ); - assert!(labeled("a < b", metrics, 60).is_empty()); - let a: &[Sample] = &[("job=x,inst=1", 50, 10.)]; - let b: &[Sample] = &[("job=x,inst=2", 50, 4.)]; - let metrics = &[("a", a), ("b", b)]; - assert_eq!( - labeled("a > on(job) b", metrics, 60), - rows(&[("job=x", 10.)]) - ); - assert_eq!( - labeled("a > ignoring(inst) b", metrics, 60), - rows(&[("__name__=a,job=x", 10.)]) - ); - let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, 5.)]; - let metrics = &[("a", pair), ("b", b)]; - assert_eq!( - labeled("a > on(job) b", metrics, 60), - rows(&[("job=x", 5.)]) - ); - let error = evaluate("a > bool on(job) b", metrics, 60).unwrap_err(); - assert!(error.contains("many-to-one"), "{error}"); - let nan: &[Sample] = &[("job=x", 50, f64::NAN)]; - let metrics = &[("a", nan), ("b", nan)]; - assert_eq!(labeled("a == bool b", metrics, 60), rows(&[("job=x", 0.)])); - assert_eq!( - labeled_nan("a != b", metrics, 60), - vec![("__name__=a,job=x".into(), "NaN".into())] - ); -} - -const S: &[Sample] = &[ - ("job=x", 50, 1.), - ("job=y", 50, 2.), - ("job=z,inst=1", 50, 3.), -]; -const T: &[Sample] = &[ - ("job=x", 50, 10.), - ("job=w", 50, 20.), - ("job=z,inst=2", 50, 30.), -]; - -// Set operators match label sets many-to-many, ignoring the name by default, -// and return the original series unchanged. -#[test] -fn set_operators_match_label_sets() { - let metrics = &[("a", S), ("b", T)]; - assert_eq!( - labeled("a and b", metrics, 60), - rows(&[("__name__=a,job=x", 1.)]) - ); - assert_eq!( - labeled("a and on(job) b", metrics, 60), - rows(&[("__name__=a,job=x", 1.), ("__name__=a,inst=1,job=z", 3.)]) - ); - assert_eq!( - labeled("a and ignoring(inst) b", metrics, 60), - labeled("a and on(job) b", metrics, 60) - ); - assert_eq!( - labeled("a or b", metrics, 60), - rows(&[ - ("__name__=a,job=x", 1.), - ("__name__=a,job=y", 2.), - ("__name__=a,inst=1,job=z", 3.), - ("__name__=b,job=w", 20.), - ("__name__=b,inst=2,job=z", 30.), - ]) - ); - assert_eq!( - labeled("a or on(job) b", metrics, 60), - rows(&[ - ("__name__=a,job=x", 1.), - ("__name__=a,job=y", 2.), - ("__name__=a,inst=1,job=z", 3.), - ("__name__=b,job=w", 20.), - ]) - ); - assert_eq!( - labeled("a unless b", metrics, 60), - rows(&[("__name__=a,job=y", 2.), ("__name__=a,inst=1,job=z", 3.)]) - ); - assert_eq!( - labeled("a unless on(job) b", metrics, 60), - rows(&[("__name__=a,job=y", 2.)]) - ); - assert_eq!(labeled("a and on() b", metrics, 60).len(), 3); - // Empty sides, and duplicates on either side, which set operators allow. - let a_only = &[("a", S)]; - assert!(labeled("a and b", a_only, 60).is_empty()); - assert_eq!(labeled("a unless b", a_only, 60).len(), 3); - assert_eq!(labeled("b or a", a_only, 60).len(), 3); - let pair: &[Sample] = &[("job=x,inst=1", 50, 1.), ("job=x,inst=2", 50, f64::NAN)]; - assert_eq!( - labeled_nan("a and on(job) b", &[("a", pair), ("b", pair)], 60), - vec![ - ("__name__=a,inst=1,job=x".into(), "1.0".into()), - ("__name__=a,inst=2,job=x".into(), "NaN".into()), - ] - ); -} - -const MANY: &[Sample] = &[ - ("job=x,inst=1", 50, 2.), - ("job=x,inst=2", 50, 3.), - ("job=y,inst=1", 50, 4.), -]; -const ONE: &[Sample] = &[("job=x,team=t1", 50, 10.), ("job=y", 50, 100.)]; - -// group_left/group_right match many series to one; the result keeps the many -// side's labels plus the listed labels of the one side, which a missing label -// removes. A filter keeps the left value. -#[test] -fn group_modifiers_match_many_to_one() { - let metrics = &[("a", MANY), ("info", ONE)]; - assert_eq!( - labeled("a * on(job) group_left(team) info", metrics, 60), - rows(&[ - ("inst=1,job=x,team=t1", 20.), - ("inst=2,job=x,team=t1", 30.), - ("inst=1,job=y", 400.), - ]) - ); - assert_eq!( - labeled("info - on(job) group_right a", metrics, 60), - rows(&[ - ("inst=1,job=x", 8.), - ("inst=2,job=x", 7.), - ("inst=1,job=y", 96.) - ]) - ); - assert_eq!( - labeled("info > on(job) group_right a", metrics, 60), - rows(&[ - ("__name__=a,inst=1,job=x", 10.), - ("__name__=a,inst=2,job=x", 10.), - ("__name__=a,inst=1,job=y", 100.), - ]) - ); - assert_eq!( - labeled("a > bool ignoring(inst, team) group_left info", metrics, 60), - rows(&[ - ("inst=1,job=x", 0.), - ("inst=2,job=x", 0.), - ("inst=1,job=y", 0.) - ]) - ); - // Two "one" series for a match group, or two results with equal labels. - let two: &[Sample] = &[("job=x,team=t1", 50, 1.), ("job=x,team=t2", 50, 2.)]; - let error = evaluate( - "a * on(job) group_left info", - &[("a", MANY), ("info", two)], - 60, - ) - .unwrap_err(); - assert!(error.contains("duplicate series"), "{error}"); - let error = evaluate( - "info * on(job) group_right a", - &[("a", two), ("info", MANY)], - 60, - ) - .unwrap_err(); - assert!(error.contains("left hand-side"), "{error}"); - let named: &[Sample] = &[("job=x", 50, 1.), ("__name__=c,job=x", 50, 2.)]; - let error = evaluate( - "a * on(job) group_left info", - &[("a", named), ("info", ONE)], - 60, - ) - .unwrap_err(); - assert!(error.contains("unique matches"), "{error}"); - assert!(labeled("a * on(job) group_left info", &[("a", MANY)], 60).is_empty()); -} - -// A non-literal scalar applies like a literal; scalar-scalar arithmetic yields -// a scalar; and a literal applies to aggregated rows whose value has another name. -#[test] -fn scalar_operands_and_aggregates() { - let three: &[Sample] = &[("job=b", 50, 3.)]; - let metrics = &[("a", A), ("b", three)]; - assert_eq!( - labeled("a * scalar(b)", metrics, 60), - rows(&[("job=w", 0.), ("job=x", 30.), ("job=y", 60.)]) - ); - assert_eq!( - labeled("a > scalar(b)", metrics, 60), - rows(&[("__name__=a,job=x", 10.), ("__name__=a,job=y", 20.)]) - ); - assert_eq!(labeled("scalar(b) * 2", metrics, 60), rows(&[("", 6.)])); - assert_eq!( - labeled("scalar(b) > bool 2", metrics, 60), - rows(&[("", 1.)]) - ); - // scalar() of several series is NaN. - assert!(labeled("scalar(a) - 1", metrics, 60)[0].1.is_nan()); - assert_eq!( - labeled("sum by (job) (a) * 2", metrics, 60), - rows(&[("job=w", 0.), ("job=x", 20.), ("job=y", 40.)]) - ); - assert_eq!( - labeled("sum by (job) (a) > bool 5", metrics, 60), - rows(&[("job=w", 0.), ("job=x", 1.), ("job=y", 1.)]) - ); -} - -// Range functions other than last_over_time drop the metric name, so series -// that then share a label set are an error, as in Prometheus. -#[test] -fn range_functions_drop_the_name_and_reject_equal_label_sets() { - let equal: &[Sample] = &[ - ("job=x", 10, 1.), - ("job=x", 50, 2.), - ("__name__=b,job=x", 10, 1.), - ("__name__=b,job=x", 50, 4.), - ]; - let error = evaluate("rate(a[1m])", &[("a", equal)], 60).unwrap_err(); - assert!(error.contains("same labelset"), "{error}"); - assert_eq!( - labeled("last_over_time(a[1m])", &[("a", equal)], 60), - rows(&[("__name__=a,job=x", 2.), ("__name__=b,job=x", 4.)]) - ); - assert_eq!( - labeled("max_over_time(a[1m])", &[("a", &equal[..2])], 60), - rows(&[("job=x", 2.)]) - ); -} - -// Scalar-valued expressions are scalars too; `or vector(0)` fills an empty -// aggregate; a range function inside a subquery drops the name. -#[test] -fn scalar_expressions_or_vector_and_subquery_names() { - let three: &[Sample] = &[("job=b", 50, 3.)]; - let metrics = &[("a", A), ("b", three)]; - assert_eq!( - labeled("a + (scalar(b) * 2)", metrics, 60), - rows(&[("job=w", 6.), ("job=x", 16.), ("job=y", 26.)]) - ); - assert_eq!( - labeled("a + -scalar(b)", metrics, 60), - rows(&[("job=w", -3.), ("job=x", 7.), ("job=y", 17.)]) - ); - assert_eq!( - labeled("sum(a) or vector(0)", metrics, 60), - rows(&[("", 30.)]) - ); - assert_eq!(labeled("sum(a) or vector(0)", &[], 60), rows(&[("", 0.)])); - let counter: &[Sample] = &[("job=x", 0, 0.), ("job=x", 30, 3.), ("job=x", 60, 6.)]; - let result = labeled("last_over_time(rate(a[1m])[2m:1m])", &[("a", counter)], 60); - assert_eq!(result.len(), 1); - assert_eq!(result[0].0, "job=x"); -} - -/// Instant `x_bucket` samples at 50s: `(labels without le, [(le, count)])`. -fn buckets(series: &[(&'static str, &[(&'static str, f64)])]) -> Vec { - series - .iter() - .flat_map(|(labels, buckets)| { - buckets.iter().map(move |(le, count)| { - let spec = if labels.is_empty() { - format!("le={le}") - } else { - format!("{labels},le={le}") - }; - (&*Box::leak(spec.into_boxed_str()), 50, *count) - }) - }) - .collect() -} - -fn quantile(query: &str, samples: &[Sample]) -> Vec<(String, f64)> { - labeled(query, &[("x_bucket", samples)], 60) -} - -const HISTOGRAM: &[(&str, f64)] = &[("1", 2.), ("2", 6.), ("4", 8.), ("+Inf", 10.)]; - -// histogram_quantile interpolates linearly within the bucket holding rank q·count, -// returns the highest finite bound for the +Inf bucket, and maps q outside -// [0, 1] to ∓Inf and a NaN q to NaN. Output labels drop le and __name__. -#[test] -fn histogram_quantile_interpolates_classic_buckets() { - let samples = buckets(&[("job=a", HISTOGRAM)]); - for (q, expected) in [ - ("0", 0.), - ("0.1", 0.5), - ("0.5", 1.75), - ("0.9", 4.), - ("1", 4.), - ("-0.5", f64::NEG_INFINITY), - ("1.5", f64::INFINITY), - ] { - let query = format!("histogram_quantile({q}, x_bucket)"); - assert_eq!( - quantile(&query, &samples), - vec![("job=a".into(), expected)], - "{query}" - ); - } - let nan = quantile("histogram_quantile(NaN, x_bucket)", &samples); - assert!(matches!(nan.as_slice(), [(labels, v)] if labels == "job=a" && v.is_nan())); -} - -// Each label set other than le is its own histogram. Degenerate histograms -// yield NaN: no +Inf bucket, fewer than two buckets, or zero observations. -#[test] -fn histogram_quantile_groups_series_and_rejects_degenerate_histograms() { - let samples = buckets(&[ - ("job=a", HISTOGRAM), - ("job=b", &[("1", 1.), ("2", 2.)]), - ("job=c", &[("+Inf", 5.)]), - ("job=d", &[("1", 0.), ("+Inf", 0.)]), - ("job=e,inst=1", HISTOGRAM), - ]); - let rows = quantile("histogram_quantile(0.5, x_bucket)", &samples); - let labels: Vec<_> = rows.iter().map(|(l, _)| l.as_str()).collect(); - assert_eq!( - labels, - vec!["inst=1,job=e", "job=a", "job=b", "job=c", "job=d"] - ); - assert_eq!(rows[0].1, 1.75); - assert_eq!(rows[1].1, 1.75); - assert!(rows[2..].iter().all(|(_, v)| v.is_nan()), "{rows:?}"); -} - -// Buckets sort by bound, unparsable or missing le values are skipped, equal -// bounds merge, and decreasing cumulative counts are raised to be monotonic. -#[test] -fn histogram_quantile_normalizes_buckets_like_prometheus() { - let unordered = buckets(&[("job=a", &[("+Inf", 10.), ("4", 8.), ("1", 2.), ("2", 6.)])]); - assert_eq!( - quantile("histogram_quantile(0.5, x_bucket)", &unordered), - vec![("job=a".into(), 1.75)] - ); - let mut invalid = buckets(&[("job=a", &[("abc", 100.), ("1", 2.), ("+Inf", 4.)])]); - invalid.push(("job=a", 50, 100.)); - assert_eq!( - quantile("histogram_quantile(0.5, x_bucket)", &invalid), - vec![("job=a".into(), 1.)] - ); - // Go's ParseFloat rejects an out-of-range bound rather than rounding it to +Inf. - let overflow = buckets(&[("job=a", &[("1", 1.), ("1e400", 2.)])]); - let rows = quantile("histogram_quantile(0.5, x_bucket)", &overflow); - assert!( - matches!(rows.as_slice(), [(_, v)] if v.is_nan()), - "{rows:?}" - ); - let duplicate = buckets(&[("job=a", &[("1", 1.), ("1.0", 1.), ("+Inf", 4.)])]); - assert_eq!( - quantile("histogram_quantile(0.5, x_bucket)", &duplicate), - vec![("job=a".into(), 1.)] - ); - // Counts [6, 2→6, 8, 8]: rank 7 lies in (2, 4], 1 of its 2 observations in. - let decreasing = buckets(&[("job=a", &[("1", 6.), ("2", 2.), ("4", 8.), ("+Inf", 8.)])]); - assert_eq!( - quantile("histogram_quantile(0.875, x_bucket)", &decreasing), - vec![("job=a".into(), 3.)] - ); -} - -// A lowest bucket with a non-positive bound is returned as is, not -// interpolated from zero. -#[test] -fn histogram_quantile_non_positive_lowest_bucket() { - let samples = buckets(&[("job=a", &[("-1", 2.), ("1", 4.), ("+Inf", 4.)])]); - for (q, expected) in [("0.25", -1.), ("0.75", 0.)] { - let query = format!("histogram_quantile({q}, x_bucket)"); - assert_eq!( - quantile(&query, &samples), - vec![("job=a".into(), expected)], - "{query}" - ); - } -} - -// The common shapes: an aggregated rate keeps its by labels other than le, and -// a per-series rate keeps every label but le and __name__. -#[test] -fn histogram_quantile_over_rates_and_sums() { - // Each counter grows by c per minute, so its rate is c/60. - let counter = |labels: &'static str, le: &str, c: f64| { - let spec: &'static str = Box::leak(format!("{labels},le={le}").into_boxed_str()); - (60..=240) - .step_by(60) - .map(move |t| (spec, t as i64, c * (t / 60) as f64)) - .collect::>() - }; - let mut samples = Vec::new(); - for inst in ["job=a,inst=1", "job=a,inst=2"] { - for (le, count) in HISTOGRAM { - samples.extend(counter(inst, le, *count)); - } - } - let metrics = &[("x_bucket", samples.as_slice())]; - let close = |rows: Vec<(String, f64)>, expected: &[(&str, f64)]| { - assert_eq!(rows.len(), expected.len(), "{rows:?}"); - for ((labels, v), (want, w)) in rows.iter().zip(expected) { - assert_eq!(labels, want); - assert!((v - w).abs() < 1e-9, "{labels}: {v} vs {w}"); - } - }; - close( - labeled( - "histogram_quantile(0.5, sum by (le, job) (rate(x_bucket[5m])))", - metrics, - 300, - ), - &[("job=a", 1.75)], - ); - close( - labeled( - "histogram_quantile(0.5, sum by (le) (x_bucket))", - metrics, - 250, - ), - &[("", 1.75)], - ); - close( - labeled("histogram_quantile(0.5, rate(x_bucket[5m]))", metrics, 300), - &[("inst=1,job=a", 1.75), ("inst=2,job=a", 1.75)], - ); -} - -// Histograms that differ only in __name__ collide once it is dropped, which -// Prometheus reports as an error rather than merging them. -#[test] -fn histogram_quantile_rejects_equal_output_label_sets() { - let mut samples = buckets(&[("job=a", HISTOGRAM)]); - samples.extend(buckets(&[("__name__=y_bucket,job=a", HISTOGRAM)])); - let error = evaluate( - "histogram_quantile(0.5, x_bucket)", - &[("x_bucket", &samples)], - 60, - ) - .unwrap_err(); - assert!(error.contains("same labelset"), "{error}"); -} - -// time() uses the query evaluation instant in seconds in scalar and vector operands. -#[test] -fn evaluation_time_operands_use_runtime_scope() { - assert_eq!(labeled("time()", &[], 60), rows(&[("", 60.)])); - assert_eq!(labeled("vector(time())", &[], 60), rows(&[("", 60.)])); - assert_eq!( - labeled("a + time()", &[("a", A)], 60), - rows(&[("job=w", 60.), ("job=x", 70.), ("job=y", 80.)]) - ); - assert_eq!( - labeled("time() - scalar(b)", &[("b", &[("job=x", 60, 3.)])], 61), - rows(&[("", 58.)]) - ); -} - -// Non-finite scalar operands survive both logical and physical plan JSON round trips. -#[test] -fn nonfinite_literals_round_trip_in_plans() { - for (query, expected) in [ - ("vector(NaN)", f64::NAN), - ("vector(+Inf)", f64::INFINITY), - ("vector(-Inf)", f64::NEG_INFINITY), - ] { - let expression = lower(query); - let json = serde_json::to_vec(&expression).unwrap(); - let restored: Rc = serde_json::from_slice(&json).unwrap(); - let result = evaluate_dag(&restored, &fallback_dag(restored.clone()), &[], 60).unwrap(); - assert_eq!(result.len(), 1); - if expected.is_nan() { - assert!(result[0].2.is_nan()); - } else { - assert_eq!(result[0].2, expected); - } - } -} - -// Classic histogram results remain aggregatable and support multi-quantile label branches. -#[test] -fn histogram_quantiles_and_nested_aggregation() { - let samples = buckets(&[("job=a", HISTOGRAM), ("job=b", HISTOGRAM)]); - assert_eq!( - quantile("sum(histogram_quantile(0.5, x_bucket))", &samples), - rows(&[("", 3.5)]) - ); - assert_eq!( - quantile("histogram_quantiles(x_bucket, \"q\", 0.5, 0.9)", &samples), - rows(&[ - ("job=a,q=0.5", 1.75), - ("job=a,q=0.9", 4.0), - ("job=b,q=0.5", 1.75), - ("job=b,q=0.9", 4.0) - ]) - ); -} - -// Relabeling anchors regexes, expands captures, preserves nonmatches and removes empty labels. -#[test] -fn label_replace_preserves_promql_labels() { - let samples: &[Sample] = &[ - ("job=api:one,team=old", 50, 1.0), - ("job=other,team=old", 50, 2.0), - ]; - assert_eq!( - labeled( - "label_replace(a, \"team\", \"$1\", \"job\", \"(.*):.*\")", - &[("a", samples)], - 60 - ), - rows(&[ - ("__name__=a,job=api:one,team=api", 1.0), - ("__name__=a,job=other,team=old", 2.0) - ]) - ); - assert_eq!( - labeled( - "label_replace(a, \"team\", \"\", \"job\", \".*\")", - &[("a", samples)], - 60 - ), - rows(&[ - ("__name__=a,job=api:one", 1.0), - ("__name__=a,job=other", 2.0) - ]) - ); -} - -// Binary results over aggregates retain labels contributed by the other operand. -#[test] -fn binary_aggregates_accept_additional_labels() { - let a: &[Sample] = &[("job=x", 50, 2.0)]; - let info: &[Sample] = &[("job=x,team=blue", 50, 3.0)]; - assert_eq!( - labeled( - "sum by(job)(a) * on(job) group_left(team) info", - &[("a", a), ("info", info)], - 60 - ), - rows(&[("job=x,team=blue", 6.0)]) - ); - assert_eq!( - labeled("sum by(job)(a) or info", &[("a", a), ("info", info)], 60), - rows(&[("job=x", 2.0), ("__name__=info,job=x,team=blue", 3.0)]) - ); -} - -// Relabeling handles missing sources and named captures, and rejects label-set collisions. -#[test] -fn label_replace_missing_labels_named_captures_and_duplicates() { - let a: &[Sample] = &[("job=api:one", 50, 2.)]; - assert_eq!( - labeled( - r#"label_replace(a, "team", "${part}", "job", "(?P.*):.*")"#, - &[("a", a)], - 60 - ), - rows(&[("__name__=a,job=api:one,team=api", 2.)]) - ); - assert_eq!( - labeled( - r#"label_replace(a, "team", "unknown", "missing", "^$")"#, - &[("a", a)], - 60 - ), - rows(&[("__name__=a,job=api:one,team=unknown", 2.)]) - ); - assert!(evaluate(r#"label_replace(a, "", "x", "job", ".*")"#, &[("a", a)], 60).is_err()); - let duplicate: &[Sample] = &[("job=a", 50, 1.), ("job=b", 50, 2.)]; - assert!(evaluate( - r#"label_replace(a, "job", "same", "job", ".*")"#, - &[("a", duplicate)], - 60 - ) - .unwrap_err() - .contains("same labelset")); -} - -// Right-side grouped rows and group_right labels survive an aggregated left schema. -#[test] -fn grouped_binary_right_rows_preserve_all_labels() { - let a: &[Sample] = &[("job=x", 50, 2.)]; - let info: &[Sample] = &[("job=x,team=blue", 50, 3.)]; - let metrics = &[("a", a), ("info", info)]; - assert_eq!( - labeled("sum by(job)(a) * on(job) group_right info", metrics, 60), - rows(&[("job=x,team=blue", 6.)]) - ); - assert_eq!( - labeled("sum by(job)(a) or sum by(job,team)(info)", metrics, 60), - rows(&[("job=x", 2.), ("job=x,team=blue", 3.)]) - ); - let samples = buckets(&[("job=a", HISTOGRAM)]); - assert!(evaluate( - r#"histogram_quantiles(x_bucket, "q", 0.5, 0.5)"#, - &[("x_bucket", &samples)], - 60 - ) - .unwrap_err() - .contains("same labelset")); -} - -// Range-bound anchors compile without freezing the evaluation instant into the plan. -#[test] -fn range_bound_anchors_compile() { - for query in [ - "a @ start()", - "sum_over_time(a[1m] @ end())", - "max_over_time(a[2m:1m] @ start())", - ] { - assert!(compile_query(query).is_ok(), "{query}"); - } -} - -// Stored programs resolve outer range anchors per run, including offsets and subquery grids. -#[test] -fn range_bound_anchors_use_outer_query_bounds() { - let samples: &[Sample] = &[ - ("job=a", 30, 1.), - ("job=a", 60, 2.), - ("job=a", 90, 3.), - ("job=a", 120, 4.), - ]; - for (query, expected) in [ - ("a @ start()", 2.), - ("a @ end()", 4.), - ("a @ start() offset 30s", 1.), - ("sum_over_time(a[1m] @ end())", 7.), - ("max_over_time(a[2m:1m] @ start())", 2.), - ("max_over_time(a[2m:1m] @ end())", 4.), - ("max_over_time(a[2m:1m] @ end() offset 1m)", 2.), - ("max_over_time(a @ end()[2m:1m])", 4.), - ] { - let expression = lower(query); - let dag = fallback_dag(expression.clone()); - let output = - evaluate_dag_with_range(&expression, &dag, &[("a", samples)], 90, Some((60, 120))) - .unwrap(); - assert_eq!(output.len(), 1, "{query}"); - assert_eq!(output[0].2, expected, "{query}"); - assert_eq!(output[0].1, 90_000, "{query}"); - let error = evaluate_dag(&expression, &dag, &[("a", samples)], 90).unwrap_err(); - assert!(error.contains("query range bounds"), "{query}: {error}"); - } -} - -// Regression functions use float samples per series; prediction is anchored -// at the evaluation time even when offset or @ selects an older window. -#[test] -fn regression_range_functions_use_evaluation_time_and_drop_names() { - let samples: &[Sample] = &[("job=x", 10, 3.), ("job=x", 30, 7.), ("job=x", 50, 11.)]; - for (query, at, expected) in [ - ("deriv(a[1m])", 60, 0.2), - ("predict_linear(a[1m], 10)", 60, 15.), - ("predict_linear(a[1m] offset 30s, 10)", 90, 21.), - ("predict_linear(a[1m] @ 60, 10)", 90, 21.), - ] { - let result = labeled(query, &[("a", samples)], at); - assert_eq!(result.len(), 1, "{query}"); - assert_eq!(result[0].0, "job=x"); - assert!( - (result[0].1 - expected).abs() < 1e-12, - "{query}: {result:?}" - ); - } - let constant: &[Sample] = &[("job=x", 10, 1e300), ("job=x", 50, 1e300)]; - assert_eq!( - labeled("deriv(a[1m])", &[("a", constant)], 60), - rows(&[("job=x", 0.)]) - ); - assert_eq!( - labeled("predict_linear(a[1m], 10)", &[("a", constant)], 60), - rows(&[("job=x", 1e300)]) - ); - assert!(labeled("deriv(a[1m])", &[("a", &samples[..1])], 60).is_empty()); - let infinite: &[Sample] = &[("job=x", 10, f64::INFINITY), ("job=x", 50, f64::INFINITY)]; - assert!(labeled("deriv(a[1m])", &[("a", infinite)], 60)[0] - .1 - .is_nan()); -} - -// An `@`-pinned range function is step-invariant, as in Prometheus: it is -// evaluated once, at the query start or at the subquery's first step, so -// predict_linear's anchor does not move with each evaluation step. -#[test] -fn pinned_range_functions_are_step_invariant() { - // `a` rises by one per second, sampled every 10 s. - let rising: Vec = (0..=30) - .map(|i| ("job=x", i * 10, (i * 10) as f64)) - .collect(); - // (query, range-query bounds, Prometheus value at T = 300 s). - for (query, bounds, expected) in [ - // Subquery grid (180, 300] steps 240 and 300; one evaluation at 240. - ( - "max_over_time(predict_linear(a[1m] @ 100, 0)[2m:1m])", - None, - 240., - ), - // A range query starting at 240 evaluates its grid from (120, 240]: at 180. - ( - "max_over_time(predict_linear(a[1m] @ 100, 0)[2m:1m])", - Some((240, 300)), - 180., - ), - ( - "max_over_time(predict_linear(a[1m] @ start(), 0)[2m:1m])", - Some((240, 300)), - 180., - ), - // A top-level pinned call is evaluated at the query start. - ("predict_linear(a[1m] @ 100, 0)", Some((240, 300)), 240.), - ("predict_linear(a[1m] @ 100, 0)", None, 300.), - // Functions that do not read the evaluation time are unchanged. - ("max_over_time(deriv(a[1m] @ 100)[2m:1m])", None, 1.), - ("max_over_time(rate(a[1m] @ 100)[2m:1m])", None, 1.), - ("max_over_time(a @ 100[2m:1m])", None, 100.), - // Without `@`, the anchor is each step: the latest step, 300, wins. - ("max_over_time(predict_linear(a[1m], 0)[2m:1m])", None, 300.), - ] { - let expression = lower(query); - let dag = fallback_dag(expression.clone()); - let output = evaluate_dag_with_range(&expression, &dag, &[("a", &rising)], 300, bounds) - .unwrap_or_else(|e| panic!("{query}: {e}")); - assert_eq!(output.len(), 1, "{query}"); - assert!( - (output[0].2 - expected).abs() < 1e-9, - "{query} {bounds:?}: {} != {expected}", - output[0].2 - ); - } -} - -// Dropping an inner range function's metric name rejects equal labels within -// each subquery step, while allowing that labelset at different steps. -#[test] -fn subquery_label_uniqueness_is_checked_per_evaluation_step() { - let equal: &[Sample] = &[ - ("job=x", 10, 1.), - ("job=x", 50, 2.), - ("__name__=b,job=x", 10, 1.), - ("__name__=b,job=x", 50, 4.), - ]; - let query = "last_over_time(rate(a[1m])[2m:1m])"; - let error = evaluate(query, &[("a", equal)], 60).unwrap_err(); - assert!(error.contains("same labelset"), "{error}"); - let disjoint: &[Sample] = &[ - ("job=x", -50, 1.), - ("job=x", -10, 2.), - ("__name__=b,job=x", 10, 3.), - ("__name__=b,job=x", 50, 5.), - ]; - let result = labeled(query, &[("a", disjoint)], 60); - assert_eq!(result.len(), 1); - assert_eq!(result[0].0, "job=x"); - assert!((result[0].1 - 0.05).abs() < 1e-12); -} - -// Non-finite histogram quantile parameters survive the logical DAG JSON boundary too. -#[test] -fn logical_nonfinite_quantile_parameter_round_trips() { - let expression = lower("histogram_quantile(NaN, x_bucket)"); - let restored: Rc = - serde_json::from_slice(&serde_json::to_vec(&expression).unwrap()).unwrap(); - let samples = buckets(&[("job=a", HISTOGRAM)]); - let result = evaluate_dag( - &restored, - &fallback_dag(restored.clone()), - &[("x_bucket", &samples)], - 60, - ) - .unwrap(); - assert_eq!(result.len(), 1); - assert!(result[0].2.is_nan()); -} - -/// The proposal's pointwise projections preserve names only for unary minus. -#[test] -fn pointwise_projection_names_and_dynamic_parameters() { - let samples = [("job=a", 300, -2.5)]; - assert_eq!( - labeled("-m", &[("m", &samples)], 300), - [("__name__=m,job=a".into(), 2.5)] - ); - assert_eq!( - labeled("abs(m)", &[("m", &samples)], 300), - [("job=a".into(), 2.5)] - ); - assert_eq!( - run("round(m, scalar(vector(2)))", &samples, 300).unwrap(), - [("a".into(), 300_000, -2.0)] - ); - assert_eq!( - run("clamp(m, time()-301, time())", &samples, 300).unwrap(), - [("a".into(), 300_000, -1.0)] - ); - assert!(run("clamp(m, 2, 1)", &samples, 300).unwrap().is_empty()); - assert_eq!( - run("year(m)", &[("a", 300, 0.0)], 300).unwrap(), - [("a".into(), 300_000, 1970.0)] - ); - assert_eq!( - run("hour()", &[], 3600).unwrap(), - [("".into(), 3_600_000, 1.0)] - ); -} - -/// Execute every PromQL root/conversion example in the scalar design document. -#[test] -fn scalar_design_document_examples_execute() { - let samples = [("job=a", 300, 1.0), ("job=b", 300, 2.0)]; - for (query, expected) in [ - ("2", 2.0), - ("time()", 300.0), - ("vector(time())", 300.0), - ("scalar(sum(up)) + 1", 4.0), - ] { - let root = parse_root(query, AccuracyTarget::Exact); - root.validate_structure().unwrap(); - let output = evaluate(query, &[("up", &samples)], 300).unwrap(); - assert_eq!(output.len(), 1, "{query}"); - assert_eq!(output[0].2, expected, "{query}"); - } - assert_eq!( - labeled("up * 2", &[("up", &samples)], 300), - [("job=a".into(), 2.0), ("job=b".into(), 4.0)] - ); -} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index fe79948d6..d26b180e2 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -1,10 +1,11 @@ //! Planner output binds directly to the shared runtime at a declared rate-value frontier. +mod common; use asap_aware_mapping::{ accuracy::{ AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, }, cost_model::DefaultCostModel, - Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG, }; use asap_physical_operators::dag::{ operators::Operator, @@ -12,16 +13,15 @@ use asap_physical_operators::dag::{ values::{Batch, Value}, Limits, RunContext, Scope, }; +use common::compile_physical_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::{ - post_asap::*, - pre_asap::{DataType, QueryExpr}, - types::AccuracyTarget, -}; +use planner_types::ir::physical_export::{PhysicalASAPDAG, PhysicalASAPOperatorPayload}; +use planner_types::ir::ASAPOp; +use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; struct Evidence; impl AccuracyEvidenceProvider for Evidence { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &planner_types::ir::OperatorNode) -> Option { Some(1000) } fn propagation_stats( @@ -63,14 +63,12 @@ fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { } fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: SketchAlgorithm) { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.1), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.1), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -80,7 +78,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if candidate.rationale.contains(&format!("{algorithm:?}")) => { Some(node) @@ -88,8 +86,8 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S _ => None, }) .unwrap(); - let dag = compile_post_asap_dag(&plan).unwrap(); - let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:FieldDataType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let dag = compile_physical_asap_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg {family:FieldDataType::Sketch(kind,_),..})if kind.algorithm()==&algorithm)).unwrap(); let rate_id = dag .edges .iter() @@ -155,18 +153,18 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S as Source<'static>; let compiled = compile( &placed, - BTreeMap::from([(rate_id.0 as u64, InputContract::bounded(rates.clone()))]), - &[dag.root.0 as u64], + BTreeMap::from([(rate_id as u64, InputContract::bounded(rates.clone()))]), + &[dag.roots[0] as u64], ) .unwrap(); let physical_dag = compiled - .instantiate(BTreeMap::from([(rate_id.0 as u64, source)])) + .instantiate(BTreeMap::from([(rate_id as u64, source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let output = block_on(async { let mut output = Vec::new(); let mut stream = physical_dag - .execute(&[dag.root.0 as u64], context) + .execute(&[dag.roots[0] as u64], context) .unwrap() .remove(0); while let Some(batch) = stream.next().await { @@ -203,7 +201,7 @@ use planner_types::workload::{ pub fn lower_promql( query: &str, accuracy: AccuracyTarget, -) -> Result { +) -> Result, asap_frontend_promql::PromqlError> { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -272,7 +270,7 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { check_direct_rate_topk(false); } -// Unreferenced labels still distinguish series throughout Rate and heap readout. +// Unreferenced labels still distinguish series throughout Rate and heap evaluation. #[test] fn direct_rate_topk_preserves_dynamic_unreferenced_labels() { check_direct_rate_topk(true); @@ -284,12 +282,16 @@ fn check_direct_rate_topk(dynamic: bool) { }; let mut logical = lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); - fn resolve_catalog(node: &mut QueryExpr) { - match node { - QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { - resolve_catalog(Rc::make_mut(child)) - } - QueryExpr::Scan { schema, .. } => { + fn resolve_catalog(node: &mut planner_types::ir::OperatorNode) { + match &mut node.operator { + planner_types::ir::Operator::NonASAP( + planner_types::ir::NonASAPOp::Aggregate { child, .. } + | planner_types::ir::NonASAPOp::TimeRange { child, .. }, + ) => resolve_catalog(Rc::make_mut(child)), + planner_types::ir::Operator::NonASAP(planner_types::ir::NonASAPOp::Scan { + schema, + .. + }) => { schema.closed = true; schema .fields @@ -301,14 +303,15 @@ fn check_direct_rate_topk(dynamic: bool) { } _ => panic!("unexpected input shape: {node:?}"), } + node.schema = node.operator.output_schema().unwrap(); } if dynamic { logical = with_series_identity(&logical).unwrap(); } else { - resolve_catalog(&mut logical); + resolve_catalog(Rc::make_mut(&mut logical)); } - let root = Rc::new(logical); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = logical; + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -322,7 +325,7 @@ fn check_direct_rate_topk(dynamic: bool) { let candidate = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if candidate.rationale.contains(&format!("{algorithm:?}")) => { Some(node) @@ -337,26 +340,25 @@ fn check_direct_rate_topk(dynamic: bool) { ) .unwrap(); assert!(matches!( - source.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + source.operator, + planner_types::ir::Operator::ASAP( + planner_types::ir::ASAPOp::FinalizeExactAccumulator { .. } + ) )); assert_eq!(ranked.input_contracts().count(), 1); let encoded = String::from_utf8(serde_json::to_vec(&ranked).unwrap()).unwrap(); assert!(encoded.contains("KeyedSummaryBuild")); - assert!(encoded.contains("KeyedReadout")); + assert!(encoded.contains("KeyedEvaluation")); assert!( !encoded.contains("\"Rate\""), - "Rate must be supplied by its exact stored-state readout" + "Rate must be supplied by its exact stored-state evaluation" ); } - let dag = compile_post_asap_dag(candidate).unwrap(); + let dag = compile_physical_asap_dag(candidate).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &algorithm))); let build = dag.nodes.iter().find(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &algorithm)).unwrap(); let input_id = dag .edges .iter() @@ -377,20 +379,17 @@ fn check_direct_rate_topk(dynamic: bool) { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } - } + PhysicalASAPOperatorPayload::NonASAP( + planner_types::ir::NonASAPOp::TimeRange { .. } + ) ) }) .unwrap_or_else(|| panic!("no raw counter source: {dag:?}")); let raw_schema = Arc::new(raw.output_schema.clone()); let raw_compiled = compile( &dag, - BTreeMap::from([( - u64::from(raw.id.0), - InputContract::bounded(raw_schema.clone()), - )]), - &[u64::from(dag.root.0)], + BTreeMap::from([(raw.id as u64, InputContract::bounded(raw_schema.clone()))]), + &[dag.roots[0] as u64], ) .unwrap(); let bytes = serde_json::to_vec(&raw_compiled).unwrap(); @@ -469,13 +468,13 @@ fn check_direct_rate_topk(dynamic: bool) { Operator::source(raw_schema.clone(), vec![raw_batch.clone()]).unwrap(), ) as Source<'static>; let physical_dag = raw_compiled - .instantiate(BTreeMap::from([(u64::from(raw.id.0), source)])) + .instantiate(BTreeMap::from([(raw.id as u64, source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let mut raw_scores = block_on(async { let mut scores = Vec::new(); let mut stream = physical_dag - .execute(&[u64::from(dag.root.0)], context) + .execute(&[dag.roots[0] as u64], context) .unwrap() .remove(0); while let Some(batch) = stream.next().await { @@ -521,11 +520,8 @@ fn check_direct_rate_topk(dynamic: bool) { } let compiled = compile( &dag, - BTreeMap::from([( - u64::from(input_id.0), - InputContract::bounded(schema.clone()), - )]), - &[u64::from(dag.root.0)], + BTreeMap::from([(input_id as u64, InputContract::bounded(schema.clone()))]), + &[dag.roots[0] as u64], ) .unwrap(); for (time, values, expected) in [ @@ -585,13 +581,13 @@ fn check_direct_rate_topk(dynamic: bool) { Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) as Source<'static>; let physical_dag = compiled - .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) + .instantiate(BTreeMap::from([(input_id as u64, source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let mut scores = block_on(async { let mut scores = vec![]; let mut stream = physical_dag - .execute(&[u64::from(dag.root.0)], context) + .execute(&[dag.roots[0] as u64], context) .unwrap() .remove(0); while let Some(batch) = stream.next().await { @@ -634,7 +630,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { }; let logical = lower_promql("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)).unwrap(); let root = Rc::new(with_series_identity(&logical).unwrap()); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -649,34 +645,34 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { let selected = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CountSketchWithHeap") => { + Replacement::SubDAG(node) if candidate.rationale.contains("CountSketchWithHeap") => { Some(node) } _ => None, }) .expect("signed spatial TopK must expose CountSketch with heap"); - let dag = compile_post_asap_dag(selected).unwrap(); + let dag = compile_physical_asap_dag(selected).unwrap(); let raw = dag .nodes .iter() .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } - } + PhysicalASAPOperatorPayload::NonASAP( + planner_types::ir::NonASAPOp::TimeRange { .. } + ) ) }) .unwrap(); let schema = Arc::new(raw.output_schema.clone()); let program = compile( &dag, - BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], + BTreeMap::from([(raw.id as u64, InputContract::bounded(schema.clone()))]), + &[dag.roots[0] as u64], ) .unwrap(); let snapshot_program = - asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + asap_physical_operators::physical_planner::promql_rows::compile_current_series_evaluation( selected, ) .unwrap(); @@ -684,7 +680,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { serde_json::from_slice(&serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); assert!(!encoded.to_string().contains("CurrentSeries")); assert!(encoded.to_string().contains("KeyedSummaryBuild")); - assert!(encoded.to_string().contains("KeyedReadout")); + assert!(encoded.to_string().contains("KeyedEvaluation")); for (values, expected, score) in [ ([100., 20.], "a", 100.), ([1., 20.], "b", 20.), @@ -709,7 +705,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { let batch = Batch::try_new(schema.clone(), rows).unwrap(); let physical_dag = program .instantiate(BTreeMap::from([( - u64::from(raw.id.0), + raw.id as u64, Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, )])) .unwrap(); @@ -761,7 +757,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { /// Deployment-side lifecycle choice: every summary state of `candidate` is /// continuously maintained, and the chosen lifecycles set execution timing. -fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDAG { +fn continuously_maintained_dag(candidate: &Rc) -> PhysicalASAPDAG { use asap_aware_mapping::{ cost_model::{Cost, CostModel}, enumerate_summary_maintenance_lifecycles, CostRate, Horizon, @@ -782,7 +778,7 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDAG { } fn summary_maintenance_lifecycle_cost_inputs( &self, - _: &SummaryNode, + _: &planner_types::ir::OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.)), @@ -794,7 +790,7 @@ fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDAG { } fn summary_maintenance_capabilities( &self, - _: &SummaryNode, + _: &planner_types::ir::OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -864,7 +860,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { ) .unwrap(), ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -874,7 +870,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .filter_map(|candidate| match candidate.replacement { - Replacement::Summary(root) if candidate.rationale.contains("WithHeap") => Some(root), + Replacement::SubDAG(root) if candidate.rationale.contains("WithHeap") => Some(root), _ => None, }) .collect::>(); @@ -887,10 +883,10 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. - } + }) ) }) .unwrap(); @@ -900,10 +896,10 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(..), .. - } + }) ) }) .unwrap(); @@ -911,11 +907,11 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { let physical = compile_candidate( &dag, BTreeMap::from([( - u64::from(state.id.0), + state.id as u64, InputContract::bounded(Arc::new(state.output_schema.clone())), )]), - &[u64::from(dag.root.0)], - &[u64::from(heap.id.0)], + &[dag.roots[0] as u64], + &[heap.id as u64], ) .unwrap(); let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&dag).unwrap(); @@ -949,12 +945,12 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { }) }; let (family, input, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family, input, grouping, .. - } => (family, input, grouping), + }) => (family, input, grouping), _ => unreachable!(), }; for (end, samples, leader) in [ diff --git a/crates/devtools/examples/canonical_examples.rs b/crates/devtools/examples/canonical_examples.rs index b0df222a3..fd4b703a1 100644 --- a/crates/devtools/examples/canonical_examples.rs +++ b/crates/devtools/examples/canonical_examples.rs @@ -1,6 +1,6 @@ // cargo run -p asap-lower --example canonical_examples // -// One-off: pretty-print the QueryExpr for one canonical query per variant, +// One-off: pretty-print the `OperatorNode` DAG for one canonical query per variant, // plus custom Join/SetOp/Dedup/CTE probes, to eyeball the actual shape. use asap_devtools::lower_promql_with_data_ingestion_interval; @@ -48,7 +48,7 @@ fn bgp_catalog() -> SqlCatalog { async fn main() { let promql_examples: &[(&str, &str)] = &[ ("Scan", "up"), - ("BinaryOp + PromqlScalarBridge", "up > 1"), + ("Filter + scalar predicate", "up > 1"), ("EvalTimestamp", "time()"), ("Aggregate", "sum(up)"), ( diff --git a/crates/devtools/src/bin/analyze_corpora.rs b/crates/devtools/src/bin/analyze_corpora.rs index 7aa87c8e7..0eebe1381 100644 --- a/crates/devtools/src/bin/analyze_corpora.rs +++ b/crates/devtools/src/bin/analyze_corpora.rs @@ -218,7 +218,7 @@ fn run_corpus(name: &str, source: &str, interval_ms: u64) -> CorpusResult { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("OperatorNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), @@ -493,7 +493,7 @@ async fn run_sql_corpora(out_dir: PathBuf) { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("OperatorNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index 2728dab12..5f2be890a 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -12,7 +12,7 @@ // `--epsilon ` is optional and applies to every query in the run: it // lowers with `AccuracyTarget::Epsilon()` instead of the default // `AccuracyTarget::Exact`. Without it, every `AggIntent` lowers exact and -// `asap_aware_mapping::SketchAlgorithmStrategy` never has a genuine sketch +// `asap_aware_mapping::ASAPStrategies` never has a genuine sketch // alternative to report — so no node ever picks up a `SketchApproximation` // note. Pass it to actually exercise that path, e.g.: // cargo run -p asap-lower --bin dag_export -- \ @@ -41,7 +41,7 @@ // `asap_types::dag_export::export_post_asap`. // // Together these surface every one of the four concrete replacement kinds: -// the sketch family `SketchAlgorithmStrategy`/`HydraGroupingStrategy` bound, +// the sketch family `ASAPStrategies`/`HydraGroupingStrategy` bound, // the CSE share/recompute choice `SharedSubDAGStrategy` found, the // workload-aware roll-up `RollupStrategy` derived, and the `avg -> // sum/count` rewrite `AvgToSumOverCountStrategy` proposes. Without @@ -88,8 +88,8 @@ use asap_aware_mapping::physical_plan_cost_model::{ }; use asap_aware_mapping::query_physical_lowering::PhysicalNodeRequest; use asap_aware_mapping::replacement::{ - default_strategies_with_evidence, search_workload, search_workload_with, Replacement, - ReplacementSubDAG, + default_strategies_with_evidence, is_logical_rewrite, search_workload, search_workload_with, + Replacement, ReplacementSubDAG, }; use asap_aware_mapping::{AccuracyEvidenceProvider, PropagationStats}; use asap_types::cost::{BaselineRef, CostAnnotation, CostInput, CostSource, CostUnit}; @@ -97,11 +97,9 @@ use asap_types::dag_export::{ self, DAGDecision, DAGNote, ExportDAG, NamedDAG, PostAsapSubstitution, TargetRejection, TargetReplacement, TargetReplacementAfter, WorkloadDAG, }; -use asap_types::post_asap::SummaryExpr; -use asap_types::post_asap::SummaryNode; +use asap_types::ir::cse::{structural_hash, HashCache}; +use asap_types::ir::OperatorNode; use asap_types::post_asap::{CompositionOperator, FieldDataType, SketchStatistic}; -use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::resources::CacheProfile; use asap_types::types::AccuracyTarget; @@ -137,7 +135,7 @@ fn parse_planner_cost_document(raw: &str) -> Result #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct TargetPhysicalEvidence { - target: QueryExpr, + target: Rc, scope: ComparisonScopeEvidence, candidates: Vec, } @@ -192,7 +190,7 @@ impl ComparisonScopeEvidence { #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct QueryNodePhysicalEvidence { - logical_node: QueryExpr, + logical_node: OperatorNode, operator: asap_aware_mapping::analytical_cost::PhysicalOperator, occurrence: usize, synthetic: bool, @@ -228,11 +226,11 @@ impl CandidatePhysicalEvidence { fn matches(&self, candidate: &ReplacementSubDAG) -> bool { let actual = match (self, &candidate.replacement) { - (Self::Summary { .. }, Replacement::Summary(summary)) => { - serde_json::to_value(dag_export::export_summary(summary)) + (Self::Summary { .. }, Replacement::SubDAG(node)) if !is_logical_rewrite(node) => { + serde_json::to_value(dag_export::export(node)) } - (Self::Rewrite { .. }, Replacement::Rewrite(query)) => { - serde_json::to_value(dag_export::export(query)) + (Self::Rewrite { .. }, Replacement::SubDAG(node)) if is_logical_rewrite(node) => { + serde_json::to_value(dag_export::export(node)) } _ => return false, }; @@ -369,7 +367,7 @@ impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, ) -> Result { if snapshot.scope != self.target.scope.resolve()? { @@ -400,7 +398,7 @@ impl ExportPlannerCostModel<'_> { .document .targets .iter() - .filter(|entry| entry.target == **target.root); + .filter(|entry| entry.target == *target.root); let target_evidence = targets.next()?; if targets.next().is_some() { return None; @@ -429,7 +427,7 @@ impl ExportPlannerCostModel<'_> { fn annotations( &self, candidate: &ReplacementSubDAG, - target: &Rc, + target: &Rc, ) -> (CostAnnotation, CostAnnotation, CostAnnotation) { let target = asap_aware_mapping::replacement::TargetSubDAG::new(target); let Some((provider, calibration)) = self.bound(candidate, &target) else { @@ -960,12 +958,11 @@ fn parse_args_from(argv: impl Iterator) -> ParsedArgs { } } -/// Attach workload-wide replacement explanations to their exact DAG nodes. -/// `node_hash` is only a narrowing filter; `source_expr == Some(target)` is -/// the collision-safe identity check (`source_expr` is `None` only for a -/// post-ASAP-originated node inside a `--post-asap` `post_dag`, which this -/// function is never called on — every node it sees, from an ordinary -/// [`dag_export::export`], carries `Some`). +/// Attach workload-wide replacement explanations to their exact dag nodes. +/// `node_hash` is only a narrowing filter; `source_node == Some(target)` is +/// the collision-safe identity check (every node an ordinary +/// [`dag_export::export`] produces carries `Some`; the `None` arm is +/// defensive only). fn annotate_with_explanations( dag: &mut ExportDAG, explanations: &[asap_aware_mapping::ReplacementExplanation], @@ -974,7 +971,7 @@ fn annotate_with_explanations( for (i, explanation) in explanations.iter().enumerate() { for node in dag.nodes.iter_mut() { if node.hash == Some(explanation.node_hash) - && node.source_expr.as_ref() == Some(explanation.target.as_ref()) + && node.source_node.as_ref() == Some(&explanation.target) { node.notes.push(DAGNote { kind: format!("{:?}", explanation.kind), @@ -992,7 +989,7 @@ fn annotate_with_explanations( /// never disagree about which candidate won for a given target. #[allow(dead_code)] struct Winner<'a> { - target: &'a Rc, + target: &'a Rc, candidate: &'a ReplacementSubDAG, costs: (CostAnnotation, CostAnnotation, CostAnnotation), } @@ -1056,7 +1053,7 @@ fn lookup_winner( by_hash: &HashMap>, winners: &[Winner<'_>], cache: &mut HashCache, - expr: &QueryExpr, + expr: &OperatorNode, ) -> Option { let hash = structural_hash(expr, cache); by_hash @@ -1102,12 +1099,10 @@ fn target_replacement( let strategy = winner.candidate.strategy.to_string(); let before = dag_export::export(winner.target); let after = match &winner.candidate.replacement { - Replacement::Summary(node) => { - TargetReplacementAfter::Summary(dag_export::export_summary(node)) - } - Replacement::Rewrite(rewritten) => { - TargetReplacementAfter::Rewrite(dag_export::export(rewritten)) + Replacement::SubDAG(node) if is_logical_rewrite(node) => { + TargetReplacementAfter::Rewrite(dag_export::export(node)) } + Replacement::SubDAG(node) => TargetReplacementAfter::Summary(dag_export::export(node)), Replacement::ExactComposition(_) => { unreachable!("composition candidates are materialized by GlobalSelection") } @@ -1131,6 +1126,20 @@ fn target_replacement( } } +/// Is `replacement` `retain_exact`'s conservative no-op fallback — the +/// target itself, unbound, carrying only an exact "kept pre-ASAP" guarantee? +/// `ASAPStrategies` emits it for an intent with no summary +/// realization at all (`STDDEV_POP`, `AVG`, ... dispatch to +/// `Realization::PassThrough`). It is "nothing to bind here", not a +/// replacement decision. A logical rewrite (no guarantee yet) and any sub-DAG +/// with an ASAP operator are real candidates. +fn is_trivial_retain_exact(replacement: &Replacement) -> bool { + matches!( + replacement, + Replacement::SubDAG(node) if node.guarantee.is_some() && !node.contains_asap() + ) +} + /// The two additive `--post-asap` outputs — see this file's top-of-file /// usage doc for what each is for. struct PostAsapResults { @@ -1159,7 +1168,7 @@ fn raw_only_post_asap_results() -> PostAsapResults { } /// Assign collision-free, explicit identities to structurally equal nodes -/// across a set of exported query DAGs. The full canonical sub-DAG string +/// across a set of exported query graphs. The full canonical sub-DAG string /// is the equality key; the compact integer is what JSON consumers receive. /// Consequently the viewer never needs to guess identity from labels, /// hashes, or a client-side node signature. @@ -1210,7 +1219,7 @@ fn assign_workload_node_ids(dags: &mut [&mut ExportDAG]) { /// which candidate won for a given target. #[allow(dead_code)] fn run_post_asap_with_progress( - lowered_queries: &[(String, String, QueryExpr)], + lowered_queries: &[(String, String, Rc)], progress: bool, cost_model: &dyn CostModel, export_model: Option<&ExportPlannerCostModel<'_>>, @@ -1220,9 +1229,9 @@ fn run_post_asap_with_progress( if progress { eprintln!("[3/4] ASAP-aware mapping is running…"); } - let roots: Vec<(String, Rc)> = lowered_queries + let roots: Vec<(String, Rc)> = lowered_queries .iter() - .map(|(name, _, qe)| (name.clone(), Rc::new(qe.clone()))) + .map(|(name, _, qe)| (name.clone(), Rc::clone(qe))) .collect(); let strategies; let space = if let Some(evidence) = evidence { @@ -1233,28 +1242,23 @@ fn run_post_asap_with_progress( }; let selection = space.global_selection(cost_model); - // A group's top candidate can be `keep_pre_asap`'s own conservative - // fallback — `Replacement::Summary(SummaryNode { expr: - // KeepPreAsap(Rc::new(target.clone())), .. })` — the *whole target* - // wrapped as unbound, e.g. for a multi-measure/`HAVING`-bearing - // aggregate, or (the case that actually surfaces this: `STDDEV_POP`/ - // `AVG`/`VARIANCE` dispatch to `Realization::PassThrough` with no - // alternative at all, per `realizations_for_intent`'s own doc) an - // intent with no summary realization whatsoever. This isn't a - // replacement decision — it's `SketchAlgorithmStrategy` saying "nothing - // to bind here" — the identical "no-op candidate" concept - // `explanation.rs`'s own `sketch_finding_reason` already excludes from - // being reported as a finding ("a candidate list containing only the - // trivial no-op realization... isn't an opportunity, it's just the - // target's existing shape reflected back"). Filtered out here for a - // second, load-bearing reason beyond just matching that precedent: - // `export_post_asap`'s `find_winner` re-checks every node reached - // inside a spliced-in `KeepPreAsap` payload (by design, so a target - // nested underneath one still gets found) — if that payload structurally - // *is* the enclosing target, `find_winner` immediately matches the same - // winner again, forever. Treating this candidate as "no winner" (same - // as an empty candidate list) avoids ever handing `export_post_asap` a - // winner that can't help but recurse into itself. + // A group's top candidate can be `retain_exact`'s own conservative + // fallback — the *whole target* itself, unbound, carrying only an exact + // "kept pre-ASAP" guarantee (see `is_trivial_retain_exact`) — e.g. for + // a multi-measure/`HAVING`-bearing aggregate, or (the case that actually + // surfaces this: `STDDEV_POP`/`AVG`/`VARIANCE` dispatch to + // `Realization::PassThrough` with no alternative at all, per + // `realizations_for_intent`'s own doc) an intent with no summary + // realization whatsoever. This isn't a replacement decision — it's + // `ASAPStrategies` saying "nothing to bind here" — the + // identical "no-op candidate" concept `explanation.rs`'s own + // `sketch_finding_reason` already excludes from being reported as a + // finding ("a candidate list containing only the trivial no-op + // realization... isn't an opportunity, it's just the target's existing + // shape reflected back"). Treating this candidate as "no winner" (same + // as an empty candidate list) also keeps `post_dag` honest: splicing + // the target in for itself would tag every node of an unchanged sub-DAG + // with a "replacement" decision. let winners: Vec> = selection .target_selections() .filter_map(|group| { @@ -1267,10 +1271,7 @@ fn run_post_asap_with_progress( if matches!(candidate.replacement, Replacement::ExactComposition(_)) { return None; } - if matches!( - &candidate.replacement, - Replacement::Summary(node) if matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - ) { + if is_trivial_retain_exact(&candidate.replacement) { return None; } Some(Winner { @@ -1318,7 +1319,7 @@ fn run_post_asap_with_progress( } let post_started = Instant::now(); let mut post_dag_cache = HashCache::new(); - let mut find_winner = |expr: &QueryExpr| -> Option { + let mut find_winner = |expr: &Rc| -> Option { let i = lookup_winner(&by_hash, &winners, &mut post_dag_cache, expr)?; let winner = &winners[i]; let (baseline_cost, selected_cost, benefit) = winner.costs.clone(); @@ -1337,11 +1338,11 @@ fn run_post_asap_with_progress( benefit: Some(benefit), }; Some(match &winners[i].candidate.replacement { - Replacement::Rewrite(rc) => PostAsapSubstitution::Rewrite { + Replacement::SubDAG(rc) if is_logical_rewrite(rc) => PostAsapSubstitution::Rewrite { replacement: Rc::clone(rc), decision, }, - Replacement::Summary(rc) => PostAsapSubstitution::Summary { + Replacement::SubDAG(rc) => PostAsapSubstitution::Summary { replacement: Rc::clone(rc), decision, }, @@ -1389,20 +1390,20 @@ fn run_post_asap_with_progress( for (name, _, qe) in lowered_queries { let dag = dag_export::export(qe); for node in &dag.nodes { - let Some(source_expr) = node.source_expr.as_ref() else { + let Some(source_node) = node.source_node.as_ref() else { continue; // never true for a plain `export` — defensive only. }; - if let Some(i) = lookup_winner(&by_hash, &winners, &mut lookup_cache, source_expr) { + if let Some(i) = lookup_winner(&by_hash, &winners, &mut lookup_cache, source_node) { replacements.push(( name.clone(), target_replacement(i as u32, node.id, &winners[i]), )); matched[i] = true; } - let hash = structural_hash(source_expr, &mut lookup_cache); + let hash = structural_hash(source_node, &mut lookup_cache); for &i in rejected_by_hash.get(&hash).into_iter().flatten() { let group = rejected_groups[i]; - if *source_expr != *group.target { + if *source_node != group.target { continue; } rejections.extend(group.rejected.iter().map(|rejected| { @@ -1465,7 +1466,7 @@ fn run_post_asap_with_progress( } #[cfg(test)] -fn run_post_asap(lowered_queries: &[(String, String, QueryExpr)]) -> PostAsapResults { +fn run_post_asap(lowered_queries: &[(String, String, Rc)]) -> PostAsapResults { run_post_asap_with_progress(lowered_queries, false, &DefaultCostModel, None, None) } @@ -1704,14 +1705,26 @@ mod tests { }; use asap_aware_mapping::query_physical_lowering::lower_query_physical_dag; use asap_devtools::PromqlError; + use asap_types::ir::NonASAPOp; use asap_types::pre_asap::{DataType, Field, Reduction, Schema, Source}; - fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { + fn lower_promql( + query: &str, + accuracy: AccuracyTarget, + ) -> Result, PromqlError> { lower_promql_with_data_ingestion_interval(query, accuracy, 1_000) } - fn non_topk_query() -> QueryExpr { - QueryExpr::Aggregate { + fn non_topk_query() -> Rc { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "events".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), + })) + .expect("scan leaf derives its schema"); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![asap_types::pre_asap::AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.1), @@ -1719,23 +1732,18 @@ mod tests { output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "events".into(), - }, - predicates: vec![], - schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), - }), - } + child: scan, + })) + .expect("count aggregate derives its schema") } fn fixture_raw_dag( - query: &QueryExpr, + query: &Rc, candidate: &ReplacementSubDAG, document: &PlannerCostDocument, ) -> PhysicalDAG { let model = ExportPlannerCostModel { document }; - let root = Rc::new(query.clone()); + let root = Rc::clone(query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); let (provider, _) = model.bound(candidate, &target).unwrap(); let snapshot = provider.capture_evidence_snapshot(&target).unwrap(); @@ -1751,7 +1759,7 @@ mod tests { let (query, candidate, mut document) = cost_fixture(); let raw = fixture_raw_dag(&query, &candidate, &document); let candidate_dag = cheap_candidate_dag(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); assert!(ExportPlannerCostModel { document: &document @@ -1914,7 +1922,7 @@ mod tests { let (query, candidate, mut document) = cost_fixture(); let raw = fixture_raw_dag(&query, &candidate, &document); let candidate_dag = cheap_candidate_dag(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); assert!(ExportPlannerCostModel { document: &document @@ -2194,7 +2202,7 @@ mod tests { EdgeStatistics { rows, bytes } } - fn query_evidence(query: &QueryExpr) -> Vec { + fn query_evidence(query: &Rc) -> Vec { let entries = RefCell::new(Vec::new()); let scope = test_scope().resolve().unwrap(); let provider = |request: PhysicalNodeRequest<'_>| { @@ -2242,7 +2250,7 @@ mod tests { }); Ok(evidence) }; - lower_query_physical_dag(&Rc::new(query.clone()), &scope, &provider).unwrap(); + lower_query_physical_dag(query, &scope, &provider).unwrap(); entries.into_inner() } @@ -2282,49 +2290,28 @@ mod tests { fn candidate_plan(candidate: &ReplacementSubDAG) -> serde_json::Value { match &candidate.replacement { - Replacement::Summary(summary) => { - serde_json::to_value(dag_export::export_summary(summary)).unwrap() - } - Replacement::Rewrite(rewrite) => { - serde_json::to_value(dag_export::export(rewrite)).unwrap() - } + Replacement::SubDAG(node) => serde_json::to_value(dag_export::export(node)).unwrap(), Replacement::ExactComposition(_) => { unreachable!("cost fixtures select directly materialized candidates") } } } - fn cost_fixture() -> (QueryExpr, ReplacementSubDAG, PlannerCostDocument) { + fn cost_fixture() -> (Rc, ReplacementSubDAG, PlannerCostDocument) { let query = non_topk_query(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let space = search_workload(vec![(String::from("q"), Rc::clone(&root))]); let group = space .target_subdag_candidates() - .find(|group| *group.target == query) + .find(|group| group.target == query) .expect("aggregate memo group"); let candidate = group .candidates .iter() - .find(|candidate| { - !matches!( - &candidate.replacement, - Replacement::Summary(node) - if matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - ) - }) + .find(|candidate| !is_trivial_retain_exact(&candidate.replacement)) .expect("summary candidate") .clone(); - let plan = match &candidate.replacement { - Replacement::Summary(summary) => { - serde_json::to_value(dag_export::export_summary(summary)).unwrap() - } - Replacement::Rewrite(rewrite) => { - serde_json::to_value(dag_export::export(rewrite)).unwrap() - } - Replacement::ExactComposition(_) => { - unreachable!("cost fixtures select directly materialized candidates") - } - }; + let plan = candidate_plan(&candidate); let document = PlannerCostDocument { storage_io: None, handoffs: None, @@ -2339,12 +2326,14 @@ mod tests { target: query.clone(), scope: test_scope(), candidates: vec![match &candidate.replacement { - Replacement::Summary(_) => CandidatePhysicalEvidence::Summary { - plan, - query_nodes: query_evidence(&query), - physical_dag: cheap_candidate_dag(), - }, - Replacement::Rewrite(_) => CandidatePhysicalEvidence::Rewrite { + Replacement::SubDAG(node) if !is_logical_rewrite(node) => { + CandidatePhysicalEvidence::Summary { + plan, + query_nodes: query_evidence(&query), + physical_dag: cheap_candidate_dag(), + } + } + Replacement::SubDAG(_) => CandidatePhysicalEvidence::Rewrite { plan, query_nodes: query_evidence(&query), }, @@ -2365,7 +2354,7 @@ mod tests { assert_eq!(parsed.targets[0].target, query); assert!(parsed.targets[0].candidates[0].matches(&candidate)); let model = ExportPlannerCostModel { document: &parsed }; - let target_rc = Rc::new(query.clone()); + let target_rc = Rc::clone(&query); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let (provider, calibration) = model.bound(&candidate, &target).expect("exact binding"); let estimate = PhysicalPlanCostModel::new(&provider, calibration.clone()) @@ -2373,7 +2362,7 @@ mod tests { .estimate_candidate(&candidate, &target) .unwrap(); assert!(estimate.candidate_cost < estimate.raw_cost); - let (baseline, selected, benefit) = model.annotations(&candidate, &Rc::new(query)); + let (baseline, selected, benefit) = model.annotations(&candidate, &query); assert!(baseline.value.is_some()); assert!(selected.value.is_some()); assert!(benefit.value.is_some()); @@ -2405,7 +2394,7 @@ mod tests { .unwrap() .remove("cache_profile"); let parsed = parse_planner_cost_document(&json.to_string()).unwrap(); - let target = Rc::new(query); + let target = query; let legacy = ExportPlannerCostModel { document: &parsed }.annotations(&candidate, &target); let explicit = ExportPlannerCostModel { document: &document, @@ -2426,7 +2415,7 @@ mod tests { fn cache_json_affects_ranking_and_exports_declared_evidence() { // Identical repeats hit the result cache; distinct evaluations still execute. let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let no_cache = ExportPlannerCostModel { document: &document, @@ -2513,7 +2502,7 @@ mod tests { #[test] fn duplicate_target_candidate_and_query_evidence_each_fail_closed() { let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let mut duplicate_target = document.clone(); @@ -2555,7 +2544,7 @@ mod tests { #[test] fn incomplete_or_unused_json_evidence_fails_closed() { let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); let mut missing = document.clone(); @@ -2596,7 +2585,7 @@ mod tests { physical_dag.nodes.push(physical_dag.nodes[0].clone()); let document = parse_planner_cost_document(&serde_json::to_string(&document).unwrap()) .expect("invalid physical semantics are checked by the estimator"); - let target_rc = Rc::new(query); + let target_rc = query; let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); assert!(ExportPlannerCostModel { document: &document @@ -2608,18 +2597,17 @@ mod tests { #[test] fn global_selection_uses_the_cheapest_complete_physical_candidate() { let query = non_topk_query(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let space = search_workload(vec![(String::from("q"), Rc::clone(&root))]); let group = space .target_subdag_candidates() - .find(|group| *group.target == query) + .find(|group| group.target == query) .expect("aggregate memo group"); let candidates: Vec<_> = group .candidates .iter() .filter(|candidate| { - matches!(candidate.replacement, Replacement::Summary(ref node) - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_))) + matches!(&candidate.replacement, Replacement::SubDAG(node) if node.contains_asap()) }) .take(2) .collect(); @@ -2677,7 +2665,7 @@ mod tests { let selection = space.global_selection(&model); let chosen = selection .target_selections() - .find(|selected| selected.target.as_ref() == &query) + .find(|selected| *selected.target == query) .and_then(|selected| selected.chosen) .expect("one complete physical candidate should win"); assert!(document.targets[0].candidates[1].matches(chosen)); @@ -2779,13 +2767,13 @@ mod tests { let selected_query = lower_promql("up", AccuracyTarget::Exact).unwrap(); let other_query = lower_promql("process_cpu_seconds_total", AccuracyTarget::Exact).unwrap(); let selected = ReplacementSubDAG { - replacement: Replacement::Rewrite(Rc::new(selected_query.clone())), + replacement: Replacement::SubDAG(Rc::clone(&selected_query)), strategy: "same-strategy", provenance: asap_aware_mapping::replacement::ReplacementProvenance::LogicalRewrite, rationale: String::new(), }; let other = ReplacementSubDAG { - replacement: Replacement::Rewrite(Rc::new(other_query)), + replacement: Replacement::SubDAG(other_query), strategy: "same-strategy", provenance: asap_aware_mapping::replacement::ReplacementProvenance::LogicalRewrite, rationale: String::new(), @@ -3038,8 +3026,8 @@ mod tests { .1; let q3_root = &q3.nodes[q3.root as usize]; let q4_root = &q4.nodes[q4.root as usize]; - assert!(q3_root.label.contains("Limit { n: 5,")); - assert!(q4_root.label.contains("Limit { n: 10,")); + assert!(q3_root.label.contains("Limit(5)")); + assert!(q4_root.label.contains("Limit(10)")); assert_ne!(q3_root.workload_node_id, q4_root.workload_node_id); let q3_ranked = &q3.nodes[q3_root.children[0] as usize]; let q4_ranked = &q4.nodes[q4_root.children[0] as usize]; @@ -3147,21 +3135,19 @@ mod tests { /// against real corpus queries (a `STDDEV_POP` aggregate, which — like /// `AVG` — dispatches to `Realization::PassThrough` with no /// alternative strategy of its own, so its *only* candidate is - /// `keep_pre_asap`'s conservative fallback: `Replacement::Summary` - /// wrapping the *entire target* as `SummaryExpr::KeepPreAsap`). - /// `run_post_asap` must not treat that as a real winner: splicing it - /// into `export_post_asap` would recurse forever, since `find_winner` - /// re-checks every node inside a spliced `KeepPreAsap` payload by - /// design, and this payload structurally *is* the enclosing target — a - /// fresh `find_winner` call finds the identical winner again, - /// unconditionally, every time. Filtering this shape out of `winners` - /// (same "no-op candidate" concept `explanation.rs`'s own - /// `sketch_finding_reason` already excludes from being a finding) is - /// what keeps this terminating: this test's only assertion that matters - /// is that `run_post_asap` returns at all instead of overflowing the - /// stack. + /// `retain_exact`'s conservative fallback: the *entire target* itself, + /// unbound, carrying only an exact "kept pre-ASAP" guarantee). + /// `run_post_asap` must not treat that as a real winner: under the old + /// IR, splicing it into `export_post_asap` recursed forever (the spliced + /// payload structurally *was* the enclosing target, so every fresh + /// `find_winner` call found the identical winner again). Filtering this + /// shape out of `winners` (same "no-op candidate" concept + /// `explanation.rs`'s own `sketch_finding_reason` already excludes from + /// being a finding) is what keeps this terminating and keeps the output + /// free of a fake replacement: this test asserts both that + /// `run_post_asap` returns at all and that it reports nothing. #[tokio::test] - async fn post_asap_does_not_recurse_forever_on_a_trivial_keep_pre_asap_winner() { + async fn post_asap_does_not_recurse_forever_on_a_trivial_retain_exact_winner() { let cat = default_catalog(); let stddev_query = lower_sql( "SELECT STDDEV_POP(latency) FROM metrics", @@ -3178,12 +3164,12 @@ mod tests { let results = run_post_asap(&lowered_queries); - // A trivial keep_pre_asap winner must be filtered before it ever + // A trivial retain_exact winner must be filtered before it ever // becomes a flat `TargetReplacement` — there's no real replacement // to report for a target with no alternative at all. assert!( results.replacements.is_empty(), - "a target whose only candidate is the trivial keep_pre_asap fallback \ + "a target whose only candidate is the trivial retain_exact fallback \ shouldn't produce a flat replacement entry: {:?}", results .replacements diff --git a/crates/devtools/src/bin/show_post_asap_ir.rs b/crates/devtools/src/bin/show_post_asap_ir.rs index 4b2cbf917..722f3289a 100644 --- a/crates/devtools/src/bin/show_post_asap_ir.rs +++ b/crates/devtools/src/bin/show_post_asap_ir.rs @@ -3,10 +3,11 @@ // // Lowers a batch of ad-hoc SQL/PromQL queries to pre-ASAP IR, then runs the // `asap-aware-mapping` pre-ASAP → post-ASAP binding pass and prints the -// resulting **post-ASAP IR** (the sketch-bound IR: `SummaryExpr`/`SummaryNode` -// — the concrete `SummaryKind`/`SummaryParams` committed per aggregate, or -// `KeepPreAsap` for whatever the pass left untouched). See `show_pre_asap_ir` -// for the sketch-agnostic IR one layer upstream. +// resulting **post-ASAP IR** (the sketch-bound IR: an `OperatorNode` DAG in +// which `ASAPOp` operators — the concrete summary family/params committed per +// aggregate — replace the bound aggregates, while whatever the pass left +// untouched stays a plain `NonASAPOp` sub-DAG carrying an exact guarantee). +// See `show_pre_asap_ir` for the sketch-agnostic IR one layer upstream. // // File format: one query per line, prefixed with "sql>" or "promql>". // Blank lines and lines starting with '#' are ignored. @@ -20,12 +21,12 @@ // `metrics(ts, service, region, latency, bytes)` catalog — the same table // used in cross_language.rs and topk_ir.rs. -use asap_aware_mapping::replacement::keep_pre_asap; +use asap_aware_mapping::replacement::retain_exact; use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use std::io::Read; @@ -33,18 +34,17 @@ use std::rc::Rc; const ACCURACY: AccuracyTarget = AccuracyTarget::Epsilon(0.01); -/// `SketchAlgorithmStrategy::replacements` returns every candidate. This +/// `ASAPStrategies::replacements` returns every candidate. This /// debug tool prints all of them so callers can inspect the planner's choices. /// If the strategy has none, preserve the single pre-ASAP fallback output. -fn bind_all(expr: &QueryExpr) -> Result>, String> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - let candidates = SketchAlgorithmStrategy::default_cost_model() +fn bind_all(root: &Rc) -> Result>, String> { + let target = TargetSubDAG::new(root); + let candidates = ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .filter_map(|candidate| match candidate { ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. } => Some(node), _ => None, @@ -52,7 +52,7 @@ fn bind_all(expr: &QueryExpr) -> Result>(); if candidates.is_empty() { - Ok(vec![keep_pre_asap(&root).map_err(|e| e.to_string())?]) + Ok(vec![retain_exact(root).map_err(|e| e.to_string())?]) } else { Ok(candidates) } @@ -128,7 +128,7 @@ async fn main() { Ok(candidates) => { for (index, candidate) in candidates.iter().enumerate() { println!("--- candidate {} ---", index + 1); - println!("{:#?}", candidate.expr); + println!("{:#?}", candidate.operator); } } Err(e) => println!("ERR: {e}"), @@ -149,9 +149,8 @@ mod tests { 1_000, ) .expect("query lowers to pre-ASAP IR"); - let root = Rc::new(expr.clone()); - let expected = SketchAlgorithmStrategy::default_cost_model() - .replacements(&TargetSubDAG::new(&root)) + let expected = ASAPStrategies::default_cost_model() + .replacements(&TargetSubDAG::new(&expr)) .len(); assert!(expected > 1, "fixture exposes alternative bindings"); @@ -170,14 +169,20 @@ mod tests { let candidates = bind_all(&expr).expect("binding succeeds"); assert_eq!(candidates.len(), 1); assert!(matches!( - candidates[0].expr, - asap_types::post_asap::SummaryExpr::BinaryOp { .. } + candidates[0].non_asap(), + Some(asap_types::ir::NonASAPOp::BinaryOp { .. }) )); assert!( candidates[0].guarantee.is_none(), "missing evidence must not claim a certified ratio bound" ); - asap_types::post_asap::compile_post_asap_dag(&candidates[0]) + let timed = asap_types::ir::timing::apply_lifecycle_timings( + &candidates[0], + &asap_types::ir::timing::LifecycleAssignment::default_maintained(), + &mut asap_types::ir::timing::TimingMemo::new(), + ) + .expect("the demo candidate has a legal default timing"); + asap_types::ir::physical_export::compile_physical_asap_dag(&timed) .expect("the demo candidate remains executable"); } @@ -192,9 +197,9 @@ mod tests { let candidates = bind_all(&expr).expect("binding succeeds"); assert_eq!(candidates.len(), 1); - assert!(matches!( - candidates[0].expr, - asap_types::post_asap::SummaryExpr::KeepPreAsap(_) - )); + assert!( + !candidates[0].contains_asap(), + "the whole query is kept pre-ASAP (no summary bound anywhere)" + ); } } diff --git a/crates/devtools/src/bin/show_pre_asap_ir.rs b/crates/devtools/src/bin/show_pre_asap_ir.rs index 491b48ff2..bde7cfb3c 100644 --- a/crates/devtools/src/bin/show_pre_asap_ir.rs +++ b/crates/devtools/src/bin/show_pre_asap_ir.rs @@ -2,7 +2,8 @@ // (or pipe via stdin: cargo run -p asap-devtools --bin show_pre_asap_ir < queries.txt) // // Lowers a batch of ad-hoc SQL/PromQL queries to **pre-ASAP IR** (the -// sketch-agnostic intent algebra: `QueryExpr`/`AggIntent`) and prints them. +// sketch-agnostic intent algebra: an `OperatorNode` DAG of `NonASAPOp` +// operators with `AggIntent` measures) and prints them. // See `show_post_asap_ir` for the post-ASAP sketch-bound IR one layer // downstream — this tool never picks a sketch, it only shows what a query // means. diff --git a/crates/devtools/src/bin/sketch_coverage.rs b/crates/devtools/src/bin/sketch_coverage.rs index 78bb4cbbd..290bc88dc 100644 --- a/crates/devtools/src/bin/sketch_coverage.rs +++ b/crates/devtools/src/bin/sketch_coverage.rs @@ -15,7 +15,7 @@ // // `--epsilon ` (default 0.01) sets the `AccuracyTarget` every query in // every corpus lowers with. Without an approximate target, -// `SketchAlgorithmStrategy` never has a genuine sketch alternative to +// `ASAPStrategies` never has a genuine sketch alternative to // report — see `dag_export`'s own `--epsilon` doc comment for the same // point, made there per-query instead of per-run. // @@ -28,11 +28,12 @@ use asap_aware_mapping::{explain_replacements, ExplanationKind}; use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; +use asap_types::ir::OperatorNode; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; +use std::rc::Rc; /// Line-based `#`/`--` comment stripping, then split on `;` — the shape every /// SQL corpus test in this repo already uses (copied from `variant_coverage` @@ -157,7 +158,7 @@ fn root_label(id: &str) -> String { /// reachable from. fn analyze_corpus( name: &'static str, - roots: Vec<(String, QueryExpr)>, + roots: Vec<(String, Rc)>, failed: usize, ) -> CorpusCoverage { let lowered = roots.len(); diff --git a/crates/devtools/src/bin/variant_coverage.rs b/crates/devtools/src/bin/variant_coverage.rs index fe494a0f6..83d797005 100644 --- a/crates/devtools/src/bin/variant_coverage.rs +++ b/crates/devtools/src/bin/variant_coverage.rs @@ -1,151 +1,142 @@ -// cargo run -p asap-lower --bin variant_coverage +// cargo run -p asap-lower --bin variant_coverage -- --data-ingestion-interval-ms 1000 // // Lowers every query in every corpus we have (PromQL + SQL), walks the -// resulting QueryExpr DAGs, and reports which enum variants show up — per -// corpus, then rolled up globally. Used to find the minimal QueryExpr node set. +// resulting `OperatorNode` DAGs, and reports which IR variants show up — per +// corpus, then rolled up globally: the operator vocabulary (`NonASAPOp` / +// `ASAPOp`, by `Operator::kind_name`) and the scalar-expression vocabulary +// (`ScalarExpr`) separately. Used to find the minimal IR node set. use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; +use asap_types::ir::{OperatorNode, ScalarExpr}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; +use std::rc::Rc; -const ALL_VARIANTS: &[&str] = &[ +/// Every `Operator::kind_name()`: all `NonASAPOp` variants, then all `ASAPOp` +/// variants. A front end only ever emits the former; the latter are listed so +/// the "unused" report stays an honest view of the whole vocabulary. +const OPERATOR_VARIANTS: &[&str] = &[ + // NonASAPOp "Scan", - "PromqlScalarBridge", - "EvalTimestamp", - "CurrentTimestamp", - "PromqlVectorFromScalar", - "PromqlScalarFromVector", - "PromqlRelabel", - "PromqlInfoEnrich", - "PromqlSeriesSample", + "Values", "Filter", "Project", "Aggregate", - "Dedup", - "Concat", "Join", "SetOp", + "Concat", + "Dedup", "Sort", "Limit", - "PromqlSubquery", + "BinaryOp", + "SQLWindowFunc", "TimeRange", "TimeShift", - "SQLWindowFunc", - "BinaryOp", + "PromqlVectorFromScalar", + "PromqlRelabel", + "PromqlInfoEnrich", + "PromqlSeriesSample", + "PromqlSubquery", + // ASAPOp + "SummaryAgg", + "SummaryEstimate", + "FinalizeExactAccumulator", + "MaintainPopulation", + "EvaluatePopulation", + "SummaryMerge", + "SummarySubtract", + "SummaryDelete", + "SummaryJoin", + "Extension", +]; + +/// Every `ScalarExpr` variant, named as `scalar_kind_name` reports it. +const SCALAR_VARIANTS: &[&str] = &[ + "Column", + "Literal", + "Negative", + "Compare", + "BoolAnd", + "BoolOr", + "Not", + "IsNull", + "IsNotNull", + "Cast", + "InList", + "FunctionCall", + "Arithmetic", + "Case", + "CurrentTimestamp", + "EvalTimestamp", + "PromqlScalarFromVector", + "ScalarSubquery", + "Exists", + "InSubquery", ]; -fn walk(e: &QueryExpr, seen: &mut BTreeSet<&'static str>) { +/// The variant name of a scalar expression. Exhaustive on purpose: a new +/// `ScalarExpr` variant fails to compile here until it is named. +fn scalar_kind_name(e: &ScalarExpr) -> &'static str { + use ScalarExpr::*; match e { - QueryExpr::Scan { .. } => { - seen.insert("Scan"); - } - QueryExpr::PromqlScalarBridge(_) => { - seen.insert("PromqlScalarBridge"); - } - QueryExpr::EvalTimestamp => { - seen.insert("EvalTimestamp"); - } - QueryExpr::CurrentTimestamp => { - seen.insert("CurrentTimestamp"); - } - QueryExpr::PromqlVectorFromScalar(inner) => { - seen.insert("PromqlVectorFromScalar"); - walk(inner, seen); - } - QueryExpr::PromqlScalarFromVector(inner) => { - seen.insert("PromqlScalarFromVector"); - walk(inner, seen); - } - QueryExpr::PromqlRelabel { child, .. } => { - seen.insert("PromqlRelabel"); - walk(child, seen); - } - QueryExpr::PromqlInfoEnrich { child, .. } => { - seen.insert("PromqlInfoEnrich"); - walk(child, seen); - } - QueryExpr::PromqlSeriesSample { child, .. } => { - seen.insert("PromqlSeriesSample"); - walk(child, seen); - } - QueryExpr::Filter { child, .. } => { - seen.insert("Filter"); - walk(child, seen); - } - QueryExpr::Project { child, .. } => { - seen.insert("Project"); - walk(child, seen); - } - QueryExpr::Aggregate { child, .. } => { - seen.insert("Aggregate"); - walk(child, seen); - } - QueryExpr::Dedup { child, .. } => { - seen.insert("Dedup"); - walk(child, seen); - } - QueryExpr::Concat { children, .. } => { - seen.insert("Concat"); - children.iter().for_each(|c| walk(c, seen)); - } - QueryExpr::Join { left, right, .. } => { - seen.insert("Join"); - walk(left, seen); - walk(right, seen); - } - QueryExpr::SetOp { left, right, .. } => { - seen.insert("SetOp"); - walk(left, seen); - walk(right, seen); - } - QueryExpr::Sort { child, .. } => { - seen.insert("Sort"); - walk(child, seen); - } - QueryExpr::Limit { child, .. } => { - seen.insert("Limit"); - walk(child, seen); - } - QueryExpr::PromqlSubquery { child, .. } => { - seen.insert("PromqlSubquery"); - walk(child, seen); - } - QueryExpr::TimeRange { child, .. } => { - seen.insert("TimeRange"); - walk(child, seen); - } - QueryExpr::TimeShift { child, .. } => { - seen.insert("TimeShift"); - walk(child, seen); - } - QueryExpr::SQLWindowFunc { child, .. } => { - seen.insert("SQLWindowFunc"); - walk(child, seen); - } - QueryExpr::BinaryOp { lhs, rhs, .. } => { - seen.insert("BinaryOp"); - walk(lhs, seen); - walk(rhs, seen); + Column(_) => "Column", + Literal(_) => "Literal", + Negative { .. } => "Negative", + Compare { .. } => "Compare", + BoolAnd(_) => "BoolAnd", + BoolOr(_) => "BoolOr", + Not(_) => "Not", + IsNull(_) => "IsNull", + IsNotNull(_) => "IsNotNull", + Cast { .. } => "Cast", + InList { .. } => "InList", + FunctionCall { .. } => "FunctionCall", + Arithmetic { .. } => "Arithmetic", + Case { .. } => "Case", + CurrentTimestamp => "CurrentTimestamp", + EvalTimestamp => "EvalTimestamp", + PromqlScalarFromVector(_) => "PromqlScalarFromVector", + ScalarSubquery(_) => "ScalarSubquery", + Exists { .. } => "Exists", + InSubquery { .. } => "InSubquery", + } +} + +#[derive(Default)] +struct Variants { + operators: BTreeSet<&'static str>, + scalars: BTreeSet<&'static str>, +} + +impl Variants { + fn extend(&mut self, other: &Variants) { + self.operators.extend(other.operators.iter().copied()); + self.scalars.extend(other.scalars.iter().copied()); + } +} + +fn walk_scalar(e: &ScalarExpr, seen: &mut BTreeSet<&'static str>) { + seen.insert(scalar_kind_name(e)); + for child in e.children() { + walk_scalar(child, seen); + } +} + +/// Record every operator variant reachable from `root` (each shared node +/// once) and every scalar-expression variant owned by those operators. The +/// operator nodes a scalar expression reads (`scalar(v)`, subqueries) are in +/// `OperatorNode::children`, so `reachable` already covers them. +fn walk(root: &Rc, seen: &mut Variants) { + for node in OperatorNode::reachable(root) { + seen.operators.insert(node.operator.kind_name()); + if let Some(op) = node.non_asap() { + for expr in op.scalar_exprs() { + walk_scalar(expr, &mut seen.scalars); + } } - // Scalar expression variants (issue #205) aren't relational nodes; - // this walk only reports on the relational skeleton, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} } } @@ -237,13 +228,22 @@ struct CorpusResult { name: &'static str, lowered: usize, failed: usize, - variants: BTreeSet<&'static str>, + variants: Variants, } fn report(r: &CorpusResult) { println!("--- {} ---", r.name); println!("lowered: {}, failed: {}", r.lowered, r.failed); - println!("variants ({}): {:?}", r.variants.len(), r.variants); + println!( + "operator variants ({}): {:?}", + r.variants.operators.len(), + r.variants.operators + ); + println!( + "scalar variants ({}): {:?}", + r.variants.scalars.len(), + r.variants.scalars + ); println!(); } @@ -292,7 +292,7 @@ async fn main() { ), ]; for (name, corpus) in promql_corpora { - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in promql_lines(corpus) { @@ -320,7 +320,7 @@ async fn main() { ]; for (name, corpus, catalog_fn) in sql_corpora { let catalog = catalog_fn(); - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in sql_stmts(corpus) { @@ -353,7 +353,7 @@ async fn main() { let corpus = include_str!("../../../frontend-sql/tests/bgp_analytics/data/bgp_analytics.sql"); let catalog = bgp_catalog(); - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in sql_stmts(corpus) { @@ -384,25 +384,30 @@ async fn main() { report(r); } - let mut global: BTreeSet<&'static str> = BTreeSet::new(); + let mut global = Variants::default(); let mut total_lowered = 0; let mut total_failed = 0; for r in &results { - global.extend(r.variants.iter().copied()); + global.extend(&r.variants); total_lowered += r.lowered; total_failed += r.failed; } println!("=== global ==="); println!("total lowered: {total_lowered}, total failed: {total_failed}\n"); - println!("used variants ({}):", global.len()); - for v in &global { - println!(" {v}"); - } - println!("\nunused variants ({}):", ALL_VARIANTS.len() - global.len()); - for v in ALL_VARIANTS { - if !global.contains(v) { + for (label, used, all) in [ + ("operator", &global.operators, OPERATOR_VARIANTS), + ("scalar", &global.scalars, SCALAR_VARIANTS), + ] { + println!("used {label} variants ({}):", used.len()); + for v in used { + println!(" {v}"); + } + let unused: Vec<_> = all.iter().filter(|v| !used.contains(*v)).collect(); + println!("\nunused {label} variants ({}):", unused.len()); + for v in unused { println!(" {v}"); } + println!(); } } diff --git a/crates/devtools/src/lib.rs b/crates/devtools/src/lib.rs index 5e6a0208a..0316622dc 100644 --- a/crates/devtools/src/lib.rs +++ b/crates/devtools/src/lib.rs @@ -2,7 +2,7 @@ //! //! Re-exports both language paths so a caller can depend on a single crate for //! PromQL *and* SQL. Both front ends end at the canonical intent algebra via -//! the same shared [`resolve_root`](asap_types::pre_asap::resolve_root). +//! the same unified operator IR ([`asap_types::ir::OperatorNode`]). //! //! ## Dependency isolation //! @@ -23,7 +23,7 @@ pub fn lower_promql_with_data_ingestion_interval( query: &str, accuracy: asap_types::types::AccuracyTarget, interval_ms: u64, -) -> Result { +) -> Result, PromqlError> { use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryRequirements, QueryWorkload, TimeSelection, diff --git a/crates/devtools/tests/cross_language.rs b/crates/devtools/tests/cross_language.rs index 2e4d6ff4f..597362e99 100644 --- a/crates/devtools/tests/cross_language.rs +++ b/crates/devtools/tests/cross_language.rs @@ -4,7 +4,7 @@ //! canonical intent algebra**, so a post-ASAP binding rule matching on //! `AggIntent` sees one spelling regardless of source language. These tests //! are the executable spec -//! for the shared [`canonicalize`](asap_types::pre_asap::canonicalize) pass: they pin the +//! for the shared [`canonicalize`](asap_types::ir::canonicalize) pass: they pin the //! canonical heavy-hitter shape and assert both front ends reach it. //! //! A literal `lower_sql(S) == lower_promql(P)` cannot hold — the two count @@ -14,9 +14,11 @@ //! explicit inner `Aggregate([Count])`. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; +use std::rc::Rc; fn col(name: &str, dtype: DataType) -> Field { Field::plain(name, dtype, false) @@ -39,26 +41,26 @@ fn catalog() -> SqlCatalog { ) } -async fn sql(q: &str) -> QueryExpr { +async fn sql(q: &str) -> Rc { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("SQL {q:?} failed to lower: {e:?}")) } -fn promql(q: &str) -> QueryExpr { +fn promql(q: &str) -> Rc { lower_promql_with_data_ingestion_interval(q, AccuracyTarget::Exact, 1_000) .unwrap_or_else(|e| panic!("PromQL {q:?} failed to lower: {e:?}")) } /// The canonical heavy-hitter shape: an outer `Aggregate([TopK{k}])` (grouped by /// `by`) over an inner `Aggregate([Count])`. Returns `(k, outer_by)`. -fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { - let QueryExpr::Aggregate { +fn heavy_hitter(qe: &OperatorNode) -> Option<(usize, GroupKeys)> { + let Some(NonASAPOp::Aggregate { reduction, measures, child, .. - } = qe + }) = qe.non_asap() else { return None; }; @@ -67,9 +69,9 @@ fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { }; // The child must be the explicit inner Count (not a raw Scan) — this is the // structural unification #25 asked for. - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { measures: inner, .. - } = child.as_ref() + }) = child.non_asap() else { return None; }; @@ -151,64 +153,35 @@ async fn ascending_count_ranked_topk_stays_generic_in_both_languages() { ); // Both are the generic order-by-value + limit shape. assert!( - matches!(&s, QueryExpr::Limit { .. }), + matches!(s.non_asap(), Some(NonASAPOp::Limit { .. })), "SQL stays a Limit: {s:?}" ); assert!( - matches!(&p, QueryExpr::Limit { .. }), + matches!(p.non_asap(), Some(NonASAPOp::Limit { .. })), "PromQL stays a Limit: {p:?}" ); } -/// Descend through a leading `Project` (the derived-table SELECT list). -fn strip_project(qe: &QueryExpr) -> &QueryExpr { - match qe { - QueryExpr::Project { child, .. } => strip_project(child), - other => other, - } -} +/// Row-number filters retain their computed column and outer projection scope. #[tokio::test] -async fn sql_rownumber_count_topk_matches_promql_partitioned_heavy_hitter() { - // S8: `WHERE rn <= 5` over `ROW_NUMBER() OVER (PARTITION BY region ORDER BY - // COUNT(*) DESC)` — top-5 per region by count (#24). It must reach the same - // partitioned heavy-hitter shape as PromQL `topk by (…) (5, count_over_time)` - // (P10): an outer TopK grouped by the partition over an explicit Count. - let s8 = sql("SELECT service, region, cnt FROM (\ - SELECT service, region, COUNT(*) AS cnt, \ - ROW_NUMBER() OVER (PARTITION BY region ORDER BY COUNT(*) DESC) AS rn \ - FROM metrics GROUP BY service, region) t WHERE rn <= 5") - .await; - let (k, by) = heavy_hitter(strip_project(&s8)).expect("S8 is a partitioned heavy-hitter"); - assert_eq!(k, 5); - assert!(!by.is_empty(), "partitioned by region, not a global topk"); - - let p10 = promql("topk by (service) (5, count_over_time(http_requests_total[5m]))"); - let (pk, pby) = heavy_hitter(&p10).expect("P10 is a partitioned heavy-hitter"); - assert_eq!(pk, 5); - assert!(!pby.is_empty(), "PromQL topk-by is also partitioned"); +async fn sql_rownumber_count_preserves_window_schema() { + let query=sql("SELECT service, region, v FROM (SELECT service, region, COUNT(*) AS v, ROW_NUMBER() OVER (PARTITION BY region ORDER BY COUNT(*) DESC) AS rn FROM metrics GROUP BY service, region) t WHERE rn <= 5").await; + query.validate_structure().unwrap(); + assert_eq!(query.schema.fields.len(), 3); + assert!(OperatorNode::reachable(&query) + .iter() + .any(|node| matches!(node.non_asap(), Some(NonASAPOp::SQLWindowFunc { .. })))); } #[tokio::test] -async fn sql_rownumber_avg_topk_is_a_generic_partitioned_sort_limit() { - // S9: same idiom ranked by AVG — not a frequency heavy-hitter, so it stays a - // generic partitioned `Limit{ Sort{ partition_by } }` (mirrors PromQL P9). - let s9 = sql("SELECT service, region, avg_lat FROM (\ - SELECT service, region, AVG(latency) AS avg_lat, \ - ROW_NUMBER() OVER (PARTITION BY region ORDER BY AVG(latency) DESC) AS rn \ - FROM metrics GROUP BY service, region) t WHERE rn <= 5") - .await; - assert!( - heavy_hitter(strip_project(&s9)).is_none(), - "AVG-ranked is not a heavy-hitter" - ); - let QueryExpr::Limit { child, .. } = strip_project(&s9) else { - panic!("expected a Limit, got {:?}", strip_project(&s9)); - }; - let QueryExpr::Sort { partition_by, .. } = child.as_ref() else { - panic!("expected a Sort under the Limit"); - }; - assert!(!partition_by.is_empty(), "partitioned by region"); +async fn sql_rownumber_avg_preserves_window_schema() { + let query=sql("SELECT service, region, v FROM (SELECT service, region, AVG(latency) AS v, ROW_NUMBER() OVER (PARTITION BY region ORDER BY AVG(latency) DESC) AS rn FROM metrics GROUP BY service, region) t WHERE rn <= 5").await; + query.validate_structure().unwrap(); + assert_eq!(query.schema.fields.len(), 3); + assert!(OperatorNode::reachable(&query) + .iter() + .any(|node| matches!(node.non_asap(), Some(NonASAPOp::SQLWindowFunc { .. })))); } #[tokio::test] diff --git a/crates/frontend-metricsql/src/lib.rs b/crates/frontend-metricsql/src/lib.rs index 4417a0be3..3544a9801 100644 --- a/crates/frontend-metricsql/src/lib.rs +++ b/crates/frontend-metricsql/src/lib.rs @@ -1,11 +1,14 @@ -//! MetricsQL AST to canonical `QueryExpr` frontend. +//! MetricsQL AST → the name-based `UnresolvedOp` tree → the unified operator DAG. use std::{rc::Rc, time::Duration}; +use asap_frontend_common::{ + resolve_root, UnresolvedOp as U, UnresolvedPredicate, UnresolvedScalar, +}; +use asap_types::ir::{BinaryOperator, ExprSemantics, OperatorNode, TimeRangeKind}; use asap_types::pre_asap::{ - resolve_root, AggIntent, ArithmeticOpKind, BinaryOpKind, ColumnRef, CompareOpKind, GroupKeys, - Predicate, PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, - UnresolvedQueryExpr as U, + AggIntent, ArithmeticOpKind, BinaryOpKind, ColumnRef, CompareOpKind, GroupKeys, + PromQLVectorSetOpKind, Reduction, ScalarValue, Source, }; use asap_types::types::AccuracyTarget; use metricsql_parser::ast::{AggregateModifier, DurationExpr, Expr, MetricExpr, RollupExpr}; @@ -33,10 +36,31 @@ pub fn canonical_metricsql(query: &str) -> Result { Ok(parse_metricsql(query)?.to_string()) } -pub fn lower_metricsql(query: &str, accuracy: AccuracyTarget) -> Result { +pub fn lower_metricsql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, MetricsqlError> { + match lower_metricsql_query(query, accuracy)? { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + _ => Err(unsupported("scalar root: use lower_metricsql_query")), + } +} + +/// Lower scalar constants without fabricating a relational operator. +pub fn lower_metricsql_query( + query: &str, + accuracy: AccuracyTarget, +) -> Result { let ast = parse_metricsql(query)?; + if let Expr::NumberLiteral(number) = &ast { + return Ok(asap_types::ir::QueryRoot::Scalar( + asap_types::ir::ScalarExpr::literal_f64(number.value), + )); + } let unresolved = Lowerer { accuracy }.lower(&ast)?; - resolve_root(&unresolved).map_err(|e| MetricsqlError::Resolve(e.to_string())) + resolve_root(&unresolved) + .map(asap_types::ir::QueryRoot::Operator) + .map_err(|e| MetricsqlError::Resolve(e.to_string())) } struct Lowerer { @@ -50,12 +74,16 @@ impl Lowerer { Expr::Rollup(e) => self.rollup(e), Expr::Function(e) => self.function(e), Expr::Aggregation(e) => self.aggregate(e), - Expr::NumberLiteral(e) => Ok(U::promql_scalar(e.value)), - Expr::UnaryOperator(e) => Ok(U::BinaryOp { + Expr::NumberLiteral(_) => { + Err(unsupported("scalar root requires lower_metricsql_query")) + } + // Vector negation is `x * -1` (as in the PromQL front end). + Expr::UnaryOperator(e) => Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(&e.expr)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(-1.0)), op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(self.lower(&e.expr)?), - rhs: Rc::new(U::promql_scalar(-1.0)), - vector_match: None, + scalar_left: false, + return_bool: false, }), Expr::BinaryOperator(e) => self.binary(e), Expr::Parens(e) if e.expressions.len() == 1 => self.lower(&e.expressions[0]), @@ -83,7 +111,7 @@ impl Lowerer { }, predicates: filters .into_iter() - .map(|f| Predicate(Rc::new(matcher(f)))) + .map(|f| UnresolvedPredicate(matcher(f))) .collect(), schema: None, }) @@ -101,6 +129,7 @@ impl Lowerer { None => Ok(child), Some(window) => Ok(U::TimeRange { range: duration(window)?, + kind: TimeRangeKind::Range, child: Rc::new(child), }), } @@ -258,12 +287,41 @@ impl Lowerer { return Err(unsupported(format!("MetricsQL operator `{}`", expr.op))) } }; - Ok(U::BinaryOp { + for (scalar, vector, scalar_left) in [ + (&expr.left, &expr.right, true), + (&expr.right, &expr.left, false), + ] { + if let Expr::NumberLiteral(n) = scalar.as_ref() { + return Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(vector)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(n.value)), + op, + scalar_left, + return_bool: false, + }); + } + } + Ok(binary_op( op, - lhs: Rc::new(self.lower(&expr.left)?), - rhs: Rc::new(self.lower(&expr.right)?), + self.lower(&expr.left)?, + self.lower(&expr.right)?, + )) + } +} + +/// A `BinaryOp` with default matching; MetricsQL modifiers (including `bool`) +/// are rejected before reaching here. +fn binary_op(kind: BinaryOpKind, lhs: U, rhs: U) -> U { + U::BinaryOp { + operator: BinaryOperator { + kind, vector_match: None, - }) + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), } } @@ -282,17 +340,22 @@ fn aggregate(reduction: Reduction, intent: AggIntent, chil } } -fn matcher(filter: &LabelFilter) -> U { +fn matcher(filter: &LabelFilter) -> UnresolvedScalar { let op = match filter.op { LabelFilterOp::Equal => CompareOpKind::Eq, LabelFilterOp::NotEqual => CompareOpKind::Ne, LabelFilterOp::RegexEqual => CompareOpKind::Regex, LabelFilterOp::RegexNotEqual => CompareOpKind::NotRegex, }; - U::Compare { - left: Rc::new(U::Column(ColumnRef::Named(filter.label.clone()))), + UnresolvedScalar::Compare { + left: Box::new(UnresolvedScalar::Column(ColumnRef::Named( + filter.label.clone(), + ))), op, - right: Rc::new(U::Literal(ScalarValue::Utf8(filter.value.clone()))), + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Utf8( + filter.value.clone(), + ))), + semantics: ExprSemantics::Promql, } } @@ -324,6 +387,3 @@ fn require_arity(name: &str, actual: usize, expected: usize) -> Result<(), Metri fn unsupported(message: impl Into) -> MetricsqlError { MetricsqlError::UnsupportedFeature(message.into()) } - -/// Unified lowering, promoted to the root API at planner cutover. -pub mod unified; diff --git a/crates/frontend-metricsql/tests/lowering.rs b/crates/frontend-metricsql/tests/lowering.rs index 3add8ee2a..ef828d04d 100644 --- a/crates/frontend-metricsql/tests/lowering.rs +++ b/crates/frontend-metricsql/tests/lowering.rs @@ -1,62 +1,65 @@ +use std::rc::Rc; use std::time::Duration; use asap_frontend_metricsql::{ canonical_metricsql, lower_metricsql, parse_metricsql, MetricsqlError, }; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; +use asap_types::pre_asap::{AggIntent, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { lower_metricsql(query, AccuracyTarget::Epsilon(0.01)).unwrap() } #[test] fn selector_range_aggregate_and_call_share_the_canonical_shape() { let query = r#"sum by (job) (rate(http_requests_total{status=~"5.."}[5m]))"#; - let dag = lower(query); - let QueryExpr::Aggregate { + let tree = lower(query); + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = dag + } = tree.expect_non_asap() else { panic!("expected outer aggregate"); }; - assert_eq!(reduction, Reduction::by(vec![2])); + assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected rate aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected range"); }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(child.as_ref(), QueryExpr::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) + matches!(child.expect_non_asap(), NonASAPOp::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) ); } #[test] fn default_rollup_with_explicit_range_is_last_over_time() { - let dag = lower("default_rollup(cpu_usage[5m])"); - let QueryExpr::Aggregate { + let tree = lower("default_rollup(cpu_usage[5m])"); + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = dag + } = tree.expect_non_asap() else { panic!("expected aggregate"); }; - assert_eq!(reduction, Reduction::PerEntity); + assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::LastOverTime])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if *range == Duration::from_secs(300)) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, .. } + if *range == Duration::from_secs(300) && *kind == TimeRangeKind::Range) ); } @@ -147,7 +150,13 @@ fn metricsql_multi_argument_aggregates_fail_closed() { #[test] fn supported_parameterized_functions_require_their_exact_arity() { let quantile = lower("quantile(0.9, requests_total)"); - assert!(matches!(quantile, QueryExpr::Aggregate { .. })); + assert!(matches!( + quantile.expect_non_asap(), + NonASAPOp::Aggregate { .. } + )); let rollup = lower("quantile_over_time(0.9, requests_total[5m])"); - assert!(matches!(rollup, QueryExpr::Aggregate { .. })); + assert!(matches!( + rollup.expect_non_asap(), + NonASAPOp::Aggregate { .. } + )); } diff --git a/crates/frontend-promql/Cargo.toml b/crates/frontend-promql/Cargo.toml index b8576ae9b..569c87c9a 100644 --- a/crates/frontend-promql/Cargo.toml +++ b/crates/frontend-promql/Cargo.toml @@ -3,8 +3,9 @@ name = "asap-frontend-promql" version = "0.1.0" edition = "2021" -# PromQL front end: L1 (parse) → L2 relational, then the shared L2→L3 converter -# — both in asap-types. Pulls the PromQL parser only — never DataFusion. +# PromQL front end: parse → the shared name-based `UnresolvedOp` tree +# (asap-frontend-common) → the unified IR. Pulls the PromQL parser only — +# never DataFusion. [dependencies] asap-types = { path = "../types" } asap-frontend-common = { path = "../frontend-common" } diff --git a/crates/frontend-promql/src/error.rs b/crates/frontend-promql/src/error.rs index 6f996b11d..ebbcf49c2 100644 --- a/crates/frontend-promql/src/error.rs +++ b/crates/frontend-promql/src/error.rs @@ -1,12 +1,12 @@ use std::fmt; -use asap_types::pre_asap::ResolveDAGError; +use asap_frontend_common::ResolveDAGError; use asap_types::workload::WorkloadError; -/// Errors from lowering a PromQL query (parse → the canonical, unresolved -/// DAG, built directly → -/// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved DAG, issue #179). +/// Errors from lowering a PromQL query (parse → the name-based unresolved +/// tree, built directly → +/// [`resolve_root`](asap_frontend_common::resolve_root) binds it to the +/// unified operator DAG, issue #179). /// /// Carries no DataFusion type — the PromQL front end never depends on the SQL /// stack. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` diff --git a/crates/frontend-promql/src/histogram.rs b/crates/frontend-promql/src/histogram.rs index f3f971a55..ecb8cd2c4 100644 --- a/crates/frontend-promql/src/histogram.rs +++ b/crates/frontend-promql/src/histogram.rs @@ -1,18 +1,10 @@ //! Sample-type metadata for the `histogram_quantile` discrimination (issue #79). //! -//! `histogram_quantile(φ, m)` has two lowerings: exact interpolation over -//! classic cumulative `le` buckets (`AggIntent::HistogramQuantile`, **not** -//! sketch-able) versus the generic sketch-able `Quantile` (native histograms / -//! raw samples, which post-ASAP binding can approximate to an accuracy -//! target). The true -//! signal is the argument's **sample type**, which query structure only -//! *proxies* — see the structural `is_classic_bucket_arg` heuristic, whose -//! false-positive (`…_bucket`-named non-histogram) and false-negative -//! (suffix-less classic histogram) cases this metadata fixes. -//! -//! A client that knows its sample types supplies a [`HistogramCatalog`]; it is -//! consulted first, and the structural heuristic remains the fallback when a -//! metric is undeclared. +//! Classic cumulative buckets use exact interpolation. The explicitly declared +//! `RawSamples` extension permits generic quantile sketches; it is not standard +//! PromQL histogram semantics. Native samples are rejected until the IR has a +//! native histogram sample type. Undeclared metrics require classic bucket +//! evidence (`by (le)`, a `_bucket` metric, or an `le` matcher). use std::cell::RefCell; use std::collections::HashMap; @@ -25,7 +17,7 @@ pub enum HistogramKind { /// distribution can't be reconstructed from them, so it is **not** /// sketch-able: `histogram_quantile` is exact bucket interpolation. ClassicBucket, - /// Native (exponential) histogram — sketch-able to an accuracy target. + /// Native histogram samples; currently rejected because the IR lacks their type. Native, /// Raw float samples the client retains — sketch-able. This is the case the /// generic `Quantile` lowering exists for (a client holding raw samples can @@ -37,7 +29,7 @@ impl HistogramKind { /// Whether `histogram_quantile` over this kind lowers to the sketch-able /// generic `Quantile` (`true`) rather than exact bucket interpolation. pub fn is_sketchable(self) -> bool { - !matches!(self, HistogramKind::ClassicBucket) + matches!(self, HistogramKind::RawSamples) } } @@ -106,9 +98,9 @@ mod tests { use super::*; #[test] - fn only_classic_buckets_are_not_sketchable() { + fn only_explicit_raw_samples_are_sketchable() { assert!(!HistogramKind::ClassicBucket.is_sketchable()); - assert!(HistogramKind::Native.is_sketchable()); + assert!(!HistogramKind::Native.is_sketchable()); assert!(HistogramKind::RawSamples.is_sketchable()); } diff --git a/crates/frontend-promql/src/lib.rs b/crates/frontend-promql/src/lib.rs index 348fa0263..e7d99fe4c 100644 --- a/crates/frontend-promql/src/lib.rs +++ b/crates/frontend-promql/src/lib.rs @@ -1,25 +1,26 @@ -//! PromQL front end: parse (via `promql-parser`) → the canonical, unresolved -//! shape, built directly (issue #179) → [`resolve_root`]. +//! PromQL front end: parse (via `promql-parser`) → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree, built directly +//! in canonical shape (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved -//! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational DAG; `resolve_root` runs the -//! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. -//! Depends on the PromQL parser only — never on the SQL / DataFusion stack. +//! `resolve_root` runs the +//! [`SchemaResolver`](asap_frontend_common::SchemaResolver) for positional +//! name resolution and returns the unified +//! [`OperatorNode`](asap_types::ir::OperatorNode) DAG. Depends on the PromQL +//! parser only — never on the SQL / DataFusion stack. pub mod error; pub mod histogram; pub mod promql; -use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use std::rc::Rc; + +use asap_types::ir::OperatorNode; use asap_types::workload::{DurationMs, PlanningWorkload, QueryLanguage, WorkloadError}; pub use error::PromqlError; pub use histogram::{HistogramCatalog, HistogramKind}; -/// Lower every normalized PromQL workload entry to a plan-ready `QueryExpr`. +/// Lower every normalized PromQL workload entry to a plan-ready operator DAG. /// /// PromQL workloads must declare a non-zero `data_ingestion_interval`; it is /// injected around each bare instant selector. Explicit range selectors keep @@ -29,7 +30,7 @@ pub use histogram::{HistogramCatalog, HistogramKind}; pub fn lower_promql_workload( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { lower_promql_workload_inner(workload, now_ms) } @@ -39,15 +40,47 @@ pub fn lower_promql_workload_with_histograms( workload: &PlanningWorkload, histograms: HistogramCatalog, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { let _guard = histogram::CatalogGuard::install(histograms); lower_promql_workload_inner(workload, now_ms) } +/// Lower scalar and vector query roots without introducing constant operators. +pub fn lower_promql_query_workload( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms) +} + +pub fn lower_promql_query_workload_with_histograms( + workload: &PlanningWorkload, + histograms: HistogramCatalog, + now_ms: u64, +) -> Result, PromqlError> { + let _guard = histogram::CatalogGuard::install(histograms); + lower_promql_query_workload_inner(workload, now_ms) +} + fn lower_promql_workload_inner( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms)? + .into_iter() + .map(|root| match root { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + asap_types::ir::QueryRoot::Scalar(_) => Err(PromqlError::UnsupportedFeature( + "scalar root: use lower_promql_query_workload".into(), + )), + }) + .collect() +} + +fn lower_promql_query_workload_inner( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { if !matches!(workload.query_workload.language, QueryLanguage::PromQL) { return Err(PromqlError::WrongLanguage(format!( "{:?}", @@ -66,12 +99,12 @@ fn lower_promql_workload_inner( .query_workload .entries() .map(|entry| { - let unresolved = promql::PromqlLowerer::lower_with_ingestion_interval( + let root = promql::PromqlLowerer::lower_query_with_ingestion_interval( &entry.query.0, &entry.requirements.accuracy.target(), std::time::Duration::from_millis(interval_ms), )?; - Ok(resolve_root(&unresolved)?) + Ok(root) }) .collect() } @@ -123,7 +156,7 @@ mod tests { } use std::time::Duration; - use asap_types::pre_asap::QueryExpr; + use asap_types::ir::{NonASAPOp, TimeRangeKind}; use asap_types::workload::{ BatchEntry, DataWorkload, Evidence, PlanningWorkload, Query, QueryRequirements, QueryWorkload, TimeSelection, @@ -155,27 +188,34 @@ mod tests { } } + // A bare instant selector reads the latest sample within the declared + // ingestion interval: an `Instant` lookback of that length. #[test] fn instant_selector_uses_declared_ingestion_interval() { let query = lower_promql_workload(&workload("sum by (job) (data)"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(1) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(1) + && *kind == TimeRangeKind::Instant + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } + // An explicit `m[5m]` keeps its own window as a `Range` selection. #[test] fn explicit_range_selector_keeps_its_query_range() { let query = lower_promql_workload(&workload("sum_over_time(data[5m])"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(300) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(300) + && *kind == TimeRangeKind::Range + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } @@ -191,6 +231,3 @@ mod tests { )); } } - -/// Unified lowering, promoted to the root API at planner cutover. -pub mod unified; diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index 83251053e..42b39eb60 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -1,14 +1,13 @@ -//! PromQL string → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! PromQL string → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree. //! //! - **Parsing** is delegated to `promql-parser` 0.8. //! - **Lowering** builds *directly in canonical shape* here (issue #179): the //! walk interprets PromQL semantics (range vectors, aggregate operators, -//! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved -//! `ColumnRef`s — the same DAG shape -//! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to -//! canonical, positional `QueryExpr`. The structural decisions a +//! label matchers) and emits `UnresolvedOp` / `UnresolvedScalar` nodes with +//! unresolved `ColumnRef`s — the same tree shape +//! [`resolve_root`](asap_frontend_common::resolve_root) later binds to the +//! positional [`OperatorNode`](asap_types::ir::OperatorNode) DAG. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter //! `topk` recognition, the `PerEntity`/`Reduce` reduction choice, //! `without(...)` grouping) are made right here, since a front end @@ -38,9 +37,11 @@ //! | `increase(m[w])` | `Aggregate{[Increase], TimeRange{w}}` | //! | `changes`/`delta`/`idelta`/`deriv`/`resets`/`predict_linear`/`double_exponential_smoothing`(`m[w]`, …) | `Aggregate{[Changes/Delta/…], TimeRange{w}}` — per-series counter-derivative intents (issue #44) | //! | `absent(v)` / `absent_over_time(m[w])` / `present_over_time(m[w])` | `Aggregate{[Absent/AbsentOverTime/PresentOverTime]}` — presence intents; the empty→synthesized-sample logic is a post-ASAP concern (issue #47) | -//! | `abs`/`ceil`/`sqrt`/`ln`/`clamp*`/`round`/trig(`v`), `pi()` | `Aggregate{[Math(f)]}` element-wise transform (issue #45); `pi()` → a `PromqlScalarBridge` leaf | -//! | `time()` / `timestamp`/`hour`/`day_of_week`/… (`v`) | `EvalTimestamp` leaf / `Aggregate{[TimeFn(f)]}` (issue #46) | -//! | `vector(s)` / `scalar(v)` | `PromqlVectorFromScalar` / `PromqlScalarFromVector` — the scalar⇄vector bridges (issue #48) | +//! | `abs`/`ceil`/`sqrt`/`ln`/`clamp*`/`round`/trig(`v`), `pi()` | typed scalar `Project` (issue #45); `pi()` → a `ScalarExpr::Literal` root | +//! | `time()` / `timestamp`/`hour`/`day_of_week`/… (`v`) | `ScalarExpr::EvalTimestamp` root / `Aggregate{[TimeFn(f)]}` (issue #46) | +//! | `vector(s)` / `scalar(v)` | `PromqlVectorFromScalar(s)` / `ScalarExpr::PromqlScalarFromVector(v)` — the scalar⇄vector bridges (issue #48) | +//! | ` op ` (`time() - 1`, `1 < bool 2`, `-time()`) | `ScalarExpr::{Arithmetic, Case, Negative}` — a scalar expression, never an operator | +//! | `v op `, `a op bool b`, `v > bool 0` | `Project`/`Filter` with owned scalar expressions; vector/vector uses `BinaryOp{return_bool}` | //! | `label_replace(v,…)` / `label_join(v,…)` | `PromqlRelabel{dst, value}` — per-series label rewrite; value unchanged (issue #50) | //! | `info(v, [selector])` | `PromqlInfoEnrich{selector}` — label-enrichment join against the info metric(s); join keys resolved during post-ASAP binding (issue #84) | //! | `group` / `offset` / `@` / `info` | **rejected** — distinct semantics with no intent-algebra representation yet (`info` label-join → #84) | @@ -50,7 +51,7 @@ //! | `limitk(k, v)` / `limit_ratio(r, v)` | `PromqlSeriesSample{LimitK(k) \| LimitRatio(r)}` — series-sampling selection, whole series kept unchanged (issue #86) | //! | `topk(k, count_over_time(…))` / `topk(k, sum_over_time(…))` | `Aggregate{[TopK{k}]}` (heavy-hitter intent) over the explicit inner `Aggregate{[Count/Sum]}` | //! | `topk(k, )` / `bottomk(k, …)` | `Sort{value} → Limit{k}` | -//! | `m{f}` | `Scan{predicates}` | +//! | `m{f}` / `m{f}[w]` | `TimeRange{ingestion, Instant, Scan{predicates}}` / `TimeRange{w, Range, Scan}` | //! | `a OP b` | `BinaryOp{vector_match}` | //! | `expr[r:res]` | `PromqlSubquery{r, res}` | //! | ` offset ` / ` @ `/`start()`/`end()` | `TimeShift{shift}` over the selector's `Scan` — pass-through schema; a ranged selector shifts under its `TimeRange` (issue #40) | @@ -59,22 +60,30 @@ use std::rc::Rc; use std::time::{Duration, SystemTime}; use promql_parser::label::{MatchOp, Matcher}; +use promql_parser::parser::value::ValueType; use promql_parser::parser::{ self, token, AggregateExpr, AtModifier as ParserAtModifier, BinaryExpr, Call, Expr, LabelModifier, Offset, VectorMatchCardinality, VectorSelector, }; -use asap_types::pre_asap::agg_intent::{topk, AggIntent, MathFunc, TimeFunc}; -use asap_types::pre_asap::query_expr::{ - AtModifier, BinaryOpKind, GroupKeys, GroupSide, Predicate, PromQLVectorSetOpKind, Reduction, - SortKey, Source, TimeShift, UnresolvedQueryExpr as Unresolved, VectorGrouping, VectorMatch, - VectorMatchKind, +use asap_frontend_common::{ + UnresolvedOp as Unresolved, UnresolvedPredicate, UnresolvedScalar as Scalar, UnresolvedSortKey, }; +use asap_types::ir::operator_properties::{ + AtModifier, BinaryOpKind, GroupKeys, GroupSide, PromQLVectorSetOpKind, Reduction, Source, + TimeShift, VectorGrouping, VectorMatch, VectorMatchKind, +}; +use asap_types::ir::{BinaryOperator, ExprSemantics, TimeRangeKind}; +use asap_types::pre_asap::agg_intent::{topk, AggIntent, TimeFunc}; + use asap_types::pre_asap::{ ArithmeticOpKind, ColumnRef, CompareOpKind, InfoMatcher, SampleKind, ScalarValue, }; use asap_types::types::AccuracyTarget; +/// Every scalar expression this front end builds follows PromQL's numeric rules. +const PROMQL: ExprSemantics = ExprSemantics::Promql; + use crate::error::PromqlError as LoweringError; type Result = std::result::Result; @@ -158,7 +167,7 @@ enum InnerFunc { struct Inner { metric: String, - matchers: Vec, + matchers: Vec, window: Option, func: Option, /// `offset` / `@` on the selector, carried to the `Source` (issue #40). @@ -172,16 +181,35 @@ struct Inner { const MAX_DEPTH: usize = 256; impl PromqlLowerer { - pub(crate) fn lower_with_ingestion_interval( + pub(crate) fn lower_query_with_ingestion_interval( query: &str, accuracy: &AccuracyTarget, interval: Duration, - ) -> Result { + ) -> Result { let _guard = AccuracyGuard::install(accuracy.clone()); let _interval = IngestionIntervalGuard::install(interval); let ast = parser::parse(query).map_err(LoweringError::Parse)?; check_depth(&ast, MAX_DEPTH)?; - walk(&ast) + let mut metrics = Vec::new(); + collect_metric_names(&ast, &mut metrics); + if metrics.iter().any(|metric| { + crate::histogram::current_kind_of(metric) + == Some(crate::histogram::HistogramKind::Native) + }) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + + if ast.value_type() == ValueType::Scalar { + Ok(asap_types::ir::QueryRoot::Scalar( + asap_frontend_common::resolve_scalar_root(&lower_scalar(&ast)?)?, + )) + } else { + Ok(asap_types::ir::QueryRoot::Operator( + asap_frontend_common::resolve_root(&walk(&ast)?)?, + )) + } } } @@ -273,6 +301,13 @@ fn check_depth(expr: &Expr, budget: usize) -> Result<()> { } fn walk(expr: &Expr) -> Result { + // A scalar-typed expression (`5`, `time() - 1`, `scalar(v)`, `1 < bool 2`) + // is a scalar expression at an operator position, never an operator tree. + if expr.value_type() == ValueType::Scalar { + return Err(LoweringError::UnsupportedFeature( + "scalar root requires query-root lowering".into(), + )); + } match expr { Expr::Aggregate(agg) => walk_aggregate(agg), Expr::Call(call) if call.func.name.starts_with("histogram_") => walk_histogram(call), @@ -282,31 +317,22 @@ fn walk(expr: &Expr) -> Result { Expr::Call(call) if is_typeconv_fn(call.func.name) => walk_typeconv(call), Expr::Call(call) if is_label_fn(call.func.name) => walk_label(call), Expr::Call(call) if is_sort_fn(call.func.name) => walk_sort(call), - // A bare `min_of`/`max_of(consts…)` scalar query folds to a `PromqlScalarBridge` - // leaf; a non-constant argument makes `num_expr` fail → rejected (#89). - Expr::Call(call) if is_scalar_reducer_fn(call.func.name) => { - Ok(Unresolved::promql_scalar(num_expr(expr)?)) - } Expr::Call(call) if call.func.name == "info" => walk_info(call), Expr::Call(call) => walk_call(call), Expr::Binary(bin) => walk_binary(bin), Expr::Paren(p) => walk(&p.expr), // `UnaryExpr` is built only by negation (`Neg`); unary `+` is folded to - // identity and `-` to a negated `NumberLiteral`, so this wraps a - // sub-expression whose samples must be sign-flipped. Now that a scalar - // operand exists (#35), express it as `x * -1` — a constant-foldable - // operand (`-(10*1024)`) collapses to a negated `PromqlScalarBridge` leaf; anything - // else is a vector, sign-flipped by a `Mul` against `PromqlScalarBridge(-1)`. `Mul` - // is commutative, so operand order carries no hazard (#36). - Expr::Unary(u) => match num_expr(&u.expr) { - Ok(v) => Ok(Unresolved::promql_scalar(-v)), - Err(_) => Ok(Unresolved::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(walk(&u.expr)?), - rhs: Rc::new(Unresolved::promql_scalar(-1.0)), - vector_match: None, - }), - }, + // identity and `-` to a negated `NumberLiteral`. A scalar + // operand was dispatched to `lower_scalar` above (→ `Negative`), so this + // is a vector projection. Unary negation retains the metric name. + Expr::Unary(u) => Ok(Unresolved::PromqlMap { + child: Rc::new(walk(&u.expr)?), + sample: Scalar::Negative { + expr: Box::new(Scalar::Column(ColumnRef::SampleValue)), + semantics: ExprSemantics::Promql, + }, + drop_metric_name: false, + }), Expr::Subquery(sq) => { let subquery = Unresolved::PromqlSubquery { range: sq.range, @@ -332,13 +358,14 @@ fn walk(expr: &Expr) -> Result { let (metric, matchers, shift) = vs_parts(&ms.vs)?; Ok(Unresolved::TimeRange { range: ms.range, + kind: TimeRangeKind::Range, child: Rc::new(filtered_source(metric, matchers, shift)), }) } - // A number literal is a scalar leaf (`v > 5`, or a bare scalar query - // `5`). String literals only appear as function args (`label_replace`, - // …), which are not supported, so reject them (issue #35). - Expr::NumberLiteral(n) => Ok(Unresolved::promql_scalar(n.val)), + // Scalar-typed, dispatched above; kept for exhaustiveness. String + // literals only appear as function args (`label_replace`, …), so a + // bare one is rejected (issue #35). + Expr::NumberLiteral(_) => unreachable!("scalar handled above"), Expr::StringLiteral(_) => Err(LoweringError::UnsupportedFeature( "bare string literal".into(), )), @@ -348,6 +375,98 @@ fn walk(expr: &Expr) -> Result { } } +/// Lower a scalar-typed PromQL expression to a scalar expression. A constant +/// sub-expression folds to one `Literal` (as `num_expr` always did); anything +/// else keeps its structure: `-time()` → `Negative`, `time() - 1` → +/// `Arithmetic`, `scalar(v)` → `PromqlScalarFromVector`, and a `bool` +/// comparison → `Case(Compare → 1, else 0)` (PromQL yields `0`/`1`). +fn lower_scalar(expr: &Expr) -> Result { + if let Ok(v) = num_expr(expr) { + return Ok(Scalar::Literal(ScalarValue::Float64(v))); + } + match expr { + Expr::Paren(p) => lower_scalar(&p.expr), + Expr::Unary(u) => Ok(Scalar::Negative { + expr: Box::new(lower_scalar(&u.expr)?), + semantics: PROMQL, + }), + Expr::Binary(bin) => lower_scalar_binary(bin), + Expr::Call(call) => match call.func.name { + "time" => Ok(Scalar::EvalTimestamp), + "pi" => Ok(Scalar::Literal(ScalarValue::Float64(std::f64::consts::PI))), + "scalar" => Ok(Scalar::PromqlScalarFromVector(Rc::new(walk(arg( + call, 0, + )?)?))), + // `min_of`/`max_of` fold only over constants (#89); the fold above + // failed, so surface its error for the non-constant argument. + name if is_scalar_reducer_fn(name) => Err(num_expr(expr).unwrap_err()), + other => Err(LoweringError::UnsupportedFunction(other.to_string())), + }, + other => Err(LoweringError::UnsupportedFeature(format!( + "scalar expression `{other}`" + ))), + } +} + +/// ` op `: arithmetic is an `Arithmetic` expression; a +/// comparison needs the `bool` modifier (PromQL has no scalar filter) and +/// becomes `Case(Compare → 1.0, else 0.0)`. The parser already rejects both a +/// bool-less scalar comparison and a scalar set op; both are re-checked here. +fn lower_scalar_binary(bin: &BinaryExpr) -> Result { + let left = Box::new(lower_scalar(&bin.lhs)?); + let right = Box::new(lower_scalar(&bin.rhs)?); + match binop(bin.op.id())? { + BinaryOpKind::Arithmetic(op) => Ok(Scalar::Arithmetic { + op, + left, + right, + semantics: PROMQL, + }), + BinaryOpKind::Compare(op) | BinaryOpKind::CompareBool(op) => { + if !bin.return_bool() { + return Err(LoweringError::InvalidParameter( + "a comparison between two scalars requires the `bool` modifier".into(), + )); + } + let compare = Scalar::Compare { + left, + op, + right, + semantics: PROMQL, + }; + Ok(Scalar::Case { + operand: None, + branches: vec![(compare, Scalar::Literal(ScalarValue::Float64(1.0)))], + else_expr: Some(Box::new(Scalar::Literal(ScalarValue::Float64(0.0)))), + }) + } + BinaryOpKind::Set(_) => Err(LoweringError::UnsupportedFeature( + "set operator between two scalars".into(), + )), + } +} + +/// A binary operation over two vectors. +fn vector_binary( + kind: BinaryOpKind, + vector_match: Option, + return_bool: bool, + lhs: Unresolved, + rhs: Unresolved, +) -> Unresolved { + Unresolved::BinaryOp { + operator: BinaryOperator { + kind, + vector_match, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), + } +} + /// Lower a bare function call (`rate(m[5m])`, `max_over_time(m[5m])`, …). /// /// The common case routes through the flat `lower_inner_call` template. The one @@ -576,7 +695,7 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// build this node) decides `PerEntity` vs `Reduce(by)` *without* knowing /// about `without` yet — it only ever sees `by`-mode keys, since `without`'s /// excluded-labels list is applied here, after the fact, exactly like the -/// pre-#179 legacy `relational::QueryExpr` DAG's own `mark_without` did (its +/// pre-#179 legacy relational tree's own `mark_without` did (its /// converter read `without` only after this front-end step had already set /// it). Whether /// `reduction_for` picked `PerEntity` (only possible when `keys` was empty) @@ -652,24 +771,36 @@ fn build_over_sub_dag(outer: Outer, keys: Vec, child: Unresolved) -> child, )); } - let sorted = Unresolved::Sort { - keys: vec![SortKey { - expr: Unresolved::Column(ColumnRef::SampleValue), - ascending: !descending, - nulls_first: false, - }], - partition_by: keys.into(), - child: Rc::new(child), - }; - Unresolved::Limit { - n: k as usize, - offset: 0, - child: Rc::new(sorted), - } + ranked_by_value(keys, k, descending, child) } }) } +/// Generic `topk`/`bottomk`: `Limit{k} → Sort{value, partition_by: keys}` over +/// `child` — an order-by-value ranking, not a heavy-hitter intent. +fn ranked_by_value( + keys: Vec, + k: u64, + descending: bool, + child: Unresolved, +) -> Unresolved { + let sorted = Unresolved::Sort { + keys: vec![UnresolvedSortKey { + expr: Scalar::Column(ColumnRef::SampleValue), + ascending: !descending, + nulls_first: false, + }], + partition_by: keys.into(), + child: Rc::new(child), + }; + Unresolved::Limit { + n: Some(k as usize), + offset: 0, + partition_by: GroupKeys::none(), + child: Rc::new(sorted), + } +} + /// The `histogram_*` function family (issues #43, histogram_quantile). /// /// `histogram_quantile(φ, )` lowers `` in full — preserving any @@ -696,7 +827,7 @@ fn walk_histogram(call: &Call) -> Result { // The true signal is the argument's sample type: a declared // `HistogramKind` (issue #79) drives the choice when available, else we // fall back to the structural `by (le)`/`_bucket` heuristic (issue #43). - if !histogram_arg_is_sketchable(arg_expr) { + if !histogram_arg_is_sketchable(arg_expr)? { return Ok(classic_histogram_quantile(phi, "", walk(arg_expr)?)); } let func = AggIntent::Quantile { @@ -706,23 +837,9 @@ fn walk_histogram(call: &Call) -> Result { }; return Ok(outer_aggregate(vec![], func, walk(arg_expr)?)); } - // (histogram_quantile handled above; accessors below) - let (func, vec_idx) = match call.func.name { - "histogram_count" => (AggIntent::HistogramCount, 0), - "histogram_sum" => (AggIntent::HistogramSum, 0), - "histogram_avg" => (AggIntent::HistogramAvg, 0), - "histogram_stddev" => (AggIntent::HistogramStdDev, 0), - "histogram_stdvar" => (AggIntent::HistogramStdVar, 0), - "histogram_fraction" => ( - AggIntent::HistogramFraction { - lower: num_arg(call, 0)?, - upper: num_arg(call, 1)?, - }, - 2, - ), - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - Ok(outer_aggregate(vec![], func, walk(arg(call, vec_idx)?)?)) + Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )) } /// Classic-bucket `histogram_quantile(φ, child)`. One histogram is the set of @@ -769,7 +886,7 @@ fn walk_histogram_quantiles(call: &Call) -> Result { )); } // The bucket-vs-native choice is a property of the argument, not of φ. - let sketchable = histogram_arg_is_sketchable(vec_expr); + let sketchable = histogram_arg_is_sketchable(vec_expr)?; let branches = (2..call.args.args.len()) .map(|i| { let phi = bounded_quantile_param(num_arg(call, i)?)?; @@ -796,9 +913,7 @@ fn walk_histogram_quantiles(call: &Call) -> Result { }; Ok(Unresolved::PromqlRelabel { dst: label.clone(), - value: Rc::new(Unresolved::Literal(ScalarValue::Utf8(open_metrics_float( - phi, - )))), + value: Scalar::Literal(ScalarValue::Utf8(open_metrics_float(phi))), child: Rc::new(quantile), }) }) @@ -852,12 +967,12 @@ fn open_metrics_float(v: f64) -> String { } } -/// The time / calendar functions (issue #46). +/// The calendar functions (issue #46); `time()` is scalar-typed and lowers in +/// `lower_scalar`. fn is_time_fn(name: &str) -> bool { matches!( name, - "time" - | "timestamp" + "timestamp" | "minute" | "hour" | "day_of_week" @@ -869,33 +984,31 @@ fn is_time_fn(name: &str) -> bool { ) } -/// `time()` → the `EvalTimestamp` leaf. `timestamp(v)` and the calendar accessors → -/// `Aggregate{[TimeFn(f)]}` over the argument vector, or over `EvalTimestamp` for the +/// `timestamp(v)` and the calendar accessors → `Aggregate{[TimeFn(f)]}` over +/// the argument vector, or over `PromqlVectorFromScalar(EvalTimestamp)` for the /// no-argument calendar forms (`hour()`, `day_of_week()`, …). Issue #46. fn walk_time(call: &Call) -> Result { - if call.func.name == "time" { - return Ok(Unresolved::EvalTimestamp); + // timestamp() reads the selected sample's timestamp, not its value. + if call.func.name == "timestamp" { + return Ok(outer_aggregate( + vec![], + AggIntent::TimeFn(TimeFunc::Timestamp), + walk(arg(call, 0)?)?, + )); } - let func = match call.func.name { - "timestamp" => TimeFunc::Timestamp, - "minute" => TimeFunc::Minute, - "hour" => TimeFunc::Hour, - "day_of_week" => TimeFunc::DayOfWeek, - "day_of_month" => TimeFunc::DayOfMonth, - "day_of_year" => TimeFunc::DayOfYear, - "month" => TimeFunc::Month, - "year" => TimeFunc::Year, - "days_in_month" => TimeFunc::DaysInMonth, - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - // A calendar function with no argument reads the evaluation time; otherwise - // it maps over each sample's timestamp in the argument vector. - let inner = if call.args.args.is_empty() { - Unresolved::EvalTimestamp + let child = if call.args.args.is_empty() { + Unresolved::PromqlVectorFromScalar(Scalar::EvalTimestamp) } else { walk(arg(call, 0)?)? }; - Ok(outer_aggregate(vec![], AggIntent::TimeFn(func), inner)) + Ok(Unresolved::PromqlMap { + child: Rc::new(child), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args: vec![Scalar::Column(ColumnRef::SampleValue)], + }, + drop_metric_name: true, + }) } /// The presence functions (issue #47). @@ -919,24 +1032,19 @@ fn walk_presence(call: &Call) -> Result { Ok(outer_aggregate(vec![], func, walk(arg(call, 0)?)?)) } -/// The scalar⇄vector type-conversion functions (issue #48). `info` is *not* -/// here: it is a label-enrichment join against info metrics, not a type -/// conversion, so it falls through to the `UnsupportedFunction` path (#84). +/// The scalar→vector conversion (issue #48); `scalar(v)` is scalar-typed and +/// lowers in `lower_scalar`. `info` is *not* here: it is a label-enrichment +/// join, not a type conversion (#84). fn is_typeconv_fn(name: &str) -> bool { - matches!(name, "vector" | "scalar") + name == "vector" } -/// `vector(s)` — promote a scalar to a label-less instant vector. `scalar(v)` -/// — collapse a single-element vector to its value. Both are honest bridge -/// nodes in the IR; the "exactly one element → NaN otherwise" runtime rule of -/// `scalar` is a post-ASAP/runtime concern (issue #48). +/// `vector(s)` — promote a scalar to a label-less instant vector carrying the +/// scalar expression `s` (issue #48). fn walk_typeconv(call: &Call) -> Result { - let inner = walk(arg(call, 0)?)?; - Ok(match call.func.name { - "vector" => Unresolved::PromqlVectorFromScalar(Rc::new(inner)), - "scalar" => Unresolved::PromqlScalarFromVector(Rc::new(inner)), - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }) + Ok(Unresolved::PromqlVectorFromScalar(lower_scalar(arg( + call, 0, + )?)?)) } /// The instant-vector reordering functions (issue #51). @@ -960,13 +1068,13 @@ fn walk_sort(call: &Call) -> Result { "sort_by_label_desc" => (false, false), other => return Err(LoweringError::UnsupportedFunction(other.to_string())), }; - let sort_key = |expr| SortKey { + let sort_key = |expr| UnresolvedSortKey { expr, ascending, nulls_first: false, }; let keys = if by_value { - vec![sort_key(Unresolved::Column(ColumnRef::SampleValue))] + vec![sort_key(Scalar::Column(ColumnRef::SampleValue))] } else { // `sort_by_label(v, "l1", "l2", …)` — one key per label arg, in order. if call.args.args.len() < 2 { @@ -976,7 +1084,7 @@ fn walk_sort(call: &Call) -> Result { } (1..call.args.args.len()) .map(|i| { - Ok(sort_key(Unresolved::Column(ColumnRef::Named(str_arg( + Ok(sort_key(Scalar::Column(ColumnRef::Named(str_arg( call, i, )?)))) }) @@ -1050,19 +1158,15 @@ fn walk_label(call: &Call) -> Result { let replacement = str_arg(call, 2)?; let src = str_arg(call, 3)?; let regex = str_arg(call, 4)?; - let value = Unresolved::FunctionCall { + let value = Scalar::FunctionCall { name: "label_replace".into(), args: vec![ - Unresolved::Column(ColumnRef::Named(src)), - Unresolved::Literal(ScalarValue::Utf8(regex)), - Unresolved::Literal(ScalarValue::Utf8(replacement)), + Scalar::Column(ColumnRef::Named(src)), + Scalar::Literal(ScalarValue::Utf8(regex)), + Scalar::Literal(ScalarValue::Utf8(replacement)), ], }; - Ok(Unresolved::PromqlRelabel { - dst, - value: Rc::new(value), - child, - }) + Ok(Unresolved::PromqlRelabel { dst, value, child }) } "label_join" => { // label_join(v, dst, sep, src_1, …, src_n) — needs ≥1 source label. @@ -1073,19 +1177,15 @@ fn walk_label(call: &Call) -> Result { } let dst = str_arg(call, 1)?; let sep = str_arg(call, 2)?; - let mut args = vec![Unresolved::Literal(ScalarValue::Utf8(sep))]; + let mut args = vec![Scalar::Literal(ScalarValue::Utf8(sep))]; for i in 3..call.args.args.len() { - args.push(Unresolved::Column(ColumnRef::Named(str_arg(call, i)?))); + args.push(Scalar::Column(ColumnRef::Named(str_arg(call, i)?))); } - let value = Unresolved::FunctionCall { + let value = Scalar::FunctionCall { name: "label_join".into(), args, }; - Ok(Unresolved::PromqlRelabel { - dst, - value: Rc::new(value), - child, - }) + Ok(Unresolved::PromqlRelabel { dst, value, child }) } other => Err(LoweringError::UnsupportedFunction(other.to_string())), } @@ -1118,7 +1218,6 @@ fn is_math_fn(name: &str) -> bool { | "atanh" | "deg" | "rad" - | "pi" | "round" | "clamp" | "clamp_min" @@ -1127,59 +1226,24 @@ fn is_math_fn(name: &str) -> bool { } /// A math / trig function — a per-series element-wise value transform, lowered -/// to a per-series `Aggregate{[Math(f)]}` over the (instant) argument vector. -/// `pi()` is the constant π, lowered to a `PromqlScalarBridge` leaf (issue #45). +/// to a typed scalar projection over the instant-vector argument. +/// `pi()` is scalar-typed and lowers in `lower_scalar` (issue #45). fn walk_math(call: &Call) -> Result { - if call.func.name == "pi" { - return Ok(Unresolved::promql_scalar(std::f64::consts::PI)); + let mut args = vec![Scalar::Column(ColumnRef::SampleValue)]; + for index in 1..call.args.args.len() { + args.push(lower_scalar(arg(call, index)?)?); } - let func = match call.func.name { - "abs" => MathFunc::Abs, - "ceil" => MathFunc::Ceil, - "floor" => MathFunc::Floor, - "exp" => MathFunc::Exp, - "ln" => MathFunc::Ln, - "log2" => MathFunc::Log2, - "log10" => MathFunc::Log10, - "sqrt" => MathFunc::Sqrt, - "sgn" => MathFunc::Sgn, - "sin" => MathFunc::Sin, - "cos" => MathFunc::Cos, - "tan" => MathFunc::Tan, - "asin" => MathFunc::Asin, - "acos" => MathFunc::Acos, - "atan" => MathFunc::Atan, - "sinh" => MathFunc::Sinh, - "cosh" => MathFunc::Cosh, - "tanh" => MathFunc::Tanh, - "asinh" => MathFunc::Asinh, - "acosh" => MathFunc::Acosh, - "atanh" => MathFunc::Atanh, - "deg" => MathFunc::Deg, - "rad" => MathFunc::Rad, - // `round(v)` defaults the step to 1; `round(v, to)` reads arg 1. - "round" => MathFunc::Round { - to_nearest: if call.args.args.len() >= 2 { - num_arg(call, 1)? - } else { - 1.0 - }, - }, - "clamp" => MathFunc::Clamp { - min: num_arg(call, 1)?, - max: num_arg(call, 2)?, - }, - "clamp_min" => MathFunc::ClampMin { - min: num_arg(call, 1)?, - }, - "clamp_max" => MathFunc::ClampMax { - max: num_arg(call, 1)?, + if call.func.name == "round" && args.len() == 1 { + args.push(Scalar::Literal(ScalarValue::Float64(1.0))); + } + Ok(Unresolved::PromqlMap { + child: Rc::new(walk(arg(call, 0)?)?), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args, }, - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - // The value being transformed is always arg 0 (a vector). - let inner = walk(arg(call, 0)?)?; - Ok(outer_aggregate(vec![], AggIntent::Math(func), inner)) + drop_metric_name: true, + }) } /// Whether `expr` is a **classic cumulative-bucket** `histogram_quantile` @@ -1202,15 +1266,31 @@ fn walk_math(call: &Call) -> Result { /// declared `RawSamples`) and the false-negative (a suffix-less classic /// histogram declared `ClassicBucket`) of the structural heuristic. With no /// declaration, fall back to the structural `by (le)`/`_bucket` heuristic. -fn histogram_arg_is_sketchable(arg: &Expr) -> bool { +fn histogram_arg_is_sketchable(arg: &Expr) -> Result { let mut metrics = Vec::new(); collect_metric_names(arg, &mut metrics); - for metric in &metrics { - if let Some(kind) = crate::histogram::current_kind_of(metric) { - return kind.is_sketchable(); + let kinds = metrics + .iter() + .filter_map(|metric| crate::histogram::current_kind_of(metric)) + .collect::>(); + if kinds.contains(&crate::histogram::HistogramKind::Native) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + if let Some(kind) = kinds.first() { + if kinds.iter().any(|other| other != kind) { + return Err(LoweringError::UnsupportedFeature( + "mixed histogram sample contracts".into(), + )); } + return Ok(kind.is_sketchable()); + } + if is_classic_bucket_arg(arg) { + Ok(false) + } else { + Err(LoweringError::UnsupportedFeature("histogram_quantile requires classic buckets; use quantile for float samples or explicitly declare the RawSamples extension".into())) } - !is_classic_bucket_arg(arg) } /// Collect the metric names of every vector/matrix selector reachable in `expr` @@ -1282,9 +1362,28 @@ fn selector_is_bucket(vs: &VectorSelector) -> bool { || vs.matchers.matchers.iter().any(|m| m.name == "le") } +/// A binary op with at least one vector operand (a scalar/scalar op is +/// scalar-typed and never reaches here). A scalar side lowers to a +/// scalar expression; mixed operations resolve to Project or Filter. fn walk_binary(bin: &BinaryExpr) -> Result { - let lhs = scalar_or_vector(&bin.lhs)?; - let rhs = scalar_or_vector(&bin.rhs)?; + let op = binop(bin.op.id())?; + let scalar_left = bin.lhs.value_type() == ValueType::Scalar; + if scalar_left || bin.rhs.value_type() == ValueType::Scalar { + let (scalar, vector) = if scalar_left { + (&bin.lhs, &bin.rhs) + } else { + (&bin.rhs, &bin.lhs) + }; + return Ok(Unresolved::PromqlScalarOp { + child: Rc::new(walk(vector)?), + scalar: lower_scalar(scalar)?, + op, + scalar_left, + return_bool: bin.return_bool(), + }); + } + let lhs = walk(&bin.lhs)?; + let rhs = walk(&bin.rhs)?; // `VectorMatch` has no fill field; dropping fill would change which series // are emitted and their values, so the query must fall back to exact // execution instead. @@ -1295,10 +1394,6 @@ fn walk_binary(bin: &BinaryExpr) -> Result { ))); } } - let op = match (binop(bin.op.id())?, bin.return_bool()) { - (BinaryOpKind::Compare(op), true) => BinaryOpKind::CompareBool(op), - (op, _) => op, - }; let vector_match = bin.modifier.as_ref().map(|m| { let (kind, labels) = match &m.matching { Some(LabelModifier::Include(ls)) => (VectorMatchKind::On, ls.labels.clone()), @@ -1329,12 +1424,7 @@ fn walk_binary(bin: &BinaryExpr) -> Result { grouping, } }); - Ok(Unresolved::BinaryOp { - op, - lhs: Rc::new(lhs), - rhs: Rc::new(rhs), - vector_match, - }) + Ok(vector_binary(op, vector_match, bin.return_bool(), lhs, rhs)) } fn lower_inner(expr: &Expr) -> Result { @@ -1601,20 +1691,7 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result Some(intent) => windowed_aggregate(inner, vec![], intent), None => instant_source(inner.metric, inner.matchers, inner.shift), }; - let sorted = Unresolved::Sort { - keys: vec![SortKey { - expr: Unresolved::Column(ColumnRef::SampleValue), - ascending: !descending, - nulls_first: false, - }], - partition_by: keys.into(), - child: Rc::new(base), - }; - Ok(Unresolved::Limit { - n: k as usize, - offset: 0, - child: Rc::new(sorted), - }) + Ok(ranked_by_value(keys, k, descending, base)) } } } @@ -1650,19 +1727,12 @@ fn windowed_aggregate( let child = match inner.window { Some(w) => Unresolved::TimeRange { range: w, + kind: TimeRangeKind::Range, child: Rc::new(base), }, - None => base, + None => ingestion_lookback(base), }; let reduction = reduction_for(&keys, inner.window.is_some() || intent.is_per_series()); - let child = if inner.window.is_none() { - Unresolved::TimeRange { - range: current_ingestion_interval(), - child: Rc::new(child), - } - } else { - child - }; Unresolved::Aggregate { reduction, measures: vec![intent], @@ -1715,13 +1785,10 @@ fn per_series_aggregate( } } -fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { +fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { let scan = Unresolved::Scan { source: Source::TimeSeries { metric }, - predicates: matchers - .into_iter() - .map(|m| Predicate(Rc::new(m))) - .collect(), + predicates: matchers.into_iter().map(UnresolvedPredicate).collect(), // Usage-derived (PromQL is schemaless) — the SchemaResolver fills this in. schema: None, }; @@ -1735,10 +1802,17 @@ fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) } } -fn instant_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { +/// An instant selector: the latest sample per series within the workload's +/// ingestion interval, so the lookback is an `Instant` `TimeRange`. +fn instant_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { + ingestion_lookback(filtered_source(metric, matchers, shift)) +} + +fn ingestion_lookback(child: Unresolved) -> Unresolved { Unresolved::TimeRange { range: current_ingestion_interval(), - child: Rc::new(filtered_source(metric, matchers, shift)), + kind: TimeRangeKind::Instant, + child: Rc::new(child), } } @@ -1881,7 +1955,7 @@ fn resolve_group(agg: &AggregateExpr) -> Result<(Vec, bool)> { // ── Free helpers ────────────────────────────────────────────────────────────── -fn vs_parts(vs: &VectorSelector) -> Result<(String, Vec, TimeShift)> { +fn vs_parts(vs: &VectorSelector) -> Result<(String, Vec, TimeShift)> { // A non-equality `__name__` matcher (`=~` / `!~` / `!=`) selects *across* // metric names. `Source::TimeSeries { metric }` carries a single concrete // metric name, so there is no representation for a regex/negated name @@ -1961,21 +2035,22 @@ fn system_time_ms(t: SystemTime) -> Result { }) } -fn matcher_to_compare(m: &Matcher) -> Unresolved { +fn matcher_to_compare(m: &Matcher) -> Scalar { let op = match &m.op { MatchOp::Equal => CompareOpKind::Eq, MatchOp::NotEqual => CompareOpKind::Ne, MatchOp::Re(_) => CompareOpKind::Regex, MatchOp::NotRe(_) => CompareOpKind::NotRegex, }; - Unresolved::Compare { - left: Rc::new(Unresolved::Column(ColumnRef::Named(m.name.clone()))), + Scalar::Compare { + left: Box::new(Scalar::Column(ColumnRef::Named(m.name.clone()))), op, - right: Rc::new(Unresolved::Literal(ScalarValue::Utf8(m.value.clone()))), + right: Box::new(Scalar::Literal(ScalarValue::Utf8(m.value.clone()))), + semantics: PROMQL, } } -fn extract_matrix(expr: &Expr) -> Result<(String, Vec, Duration, TimeShift)> { +fn extract_matrix(expr: &Expr) -> Result<(String, Vec, Duration, TimeShift)> { match expr { Expr::MatrixSelector(ms) => { let (metric, matchers, shift) = vs_parts(&ms.vs)?; @@ -2017,6 +2092,7 @@ fn num_expr(expr: &Expr) -> Result { match expr { Expr::NumberLiteral(n) => Ok(n.val), Expr::Paren(p) => num_expr(&p.expr), + Expr::Unary(u) => Ok(-num_expr(&u.expr)?), // Constant-fold a pure scalar arithmetic expression — the parser does // not fold `10*1024*1024` / `24 * 3600`. A `modifier` (vector matching) // or a non-arithmetic operator means it is not a pure scalar. @@ -2074,15 +2150,6 @@ fn is_scalar_reducer_fn(name: &str) -> bool { matches!(name, "min_of" | "max_of") } -/// A `BinaryOp` operand: fold a pure-scalar expression (`5`, `10*1024*1024`) to -/// a `PromqlScalarBridge` leaf, otherwise walk it as a vector (issue #35). -fn scalar_or_vector(expr: &Expr) -> Result { - match num_expr(expr) { - Ok(v) => Ok(Unresolved::promql_scalar(v)), - Err(_) => walk(expr), - } -} - /// `topk`/`bottomk` count parameter — a non-negative integer. Rejects /// fractional / negative / non-finite values rather than silently truncating /// or saturating them via `as u64` (`topk(2.7, …)` ≠ `topk(2, …)`). diff --git a/crates/frontend-promql/tests/count_planning.rs b/crates/frontend-promql/tests/count_planning.rs index 3d3b65651..4609258bf 100644 --- a/crates/frontend-promql/tests/count_planning.rs +++ b/crates/frontend-promql/tests/count_planning.rs @@ -1,19 +1,19 @@ //! Query text through summary selection: counts use observations, never value weights. -use std::rc::Rc; - use asap_aware_mapping::accuracy::DefaultAccuracyModel; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_aware_mapping::{ - default_strategies, search_workload_with_targets, Replacement, ReplacementStrategy, - SketchAlgorithmStrategy, TargetSubDAG, + default_strategies, search_workload_with_targets, ASAPStrategies, Replacement, + ReplacementStrategy, TargetSubDAG, }; mod support; +use asap_types::ir::physical_export::PhysicalASAPOperatorPayload; +use asap_types::ir::{ASAPOp, Operator}; use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, FieldDataType, NonNegativeWeightProof, - PostAsapOperatorPayload, SketchAlgorithm, SummaryExpr, SummaryInputExpr, WeightDomain, + ExactKind, FieldDataType, NonNegativeWeightProof, SketchAlgorithm, SummaryInputExpr, + WeightDomain, }; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, post_asap_dag}; #[test] fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { @@ -21,7 +21,7 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { epsilon: 0.01, delta: 0.01, }; - let root = Rc::new(lower_promql("count by(job)(up)", target.clone()).unwrap()); + let root = lower_promql("count by(job)(up)", target.clone()).unwrap(); let space = search_workload_with_targets( vec![("count", root, Some(target))], &default_strategies(), @@ -51,14 +51,14 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { #[test] fn exact_counts_select_count_accumulators() { for query in ["count(up)", "count by(job)(up)", "count_over_time(up[5m])"] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact).unwrap()); + let root = lower_promql(query, AccuracyTarget::Exact).unwrap(); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); assert!( candidates.iter().any(|candidate| { - matches!(&candidate.replacement, Replacement::Summary(node) - if matches!(&node.expr, SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Count, _), .. })) + matches!(&candidate.replacement, Replacement::SubDAG(node) + if matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Count, _), .. }))) }), "{query}: {candidates:?}" ); @@ -69,22 +69,23 @@ fn exact_counts_select_count_accumulators() { #[test] fn frequency_count_candidates_use_unit_weights() { for query in ["count_over_time(up[5m])", "count(up)"] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Epsilon(0.02)).unwrap()); + let root = lower_promql(query, AccuracyTarget::Epsilon(0.02)).unwrap(); let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let mut algorithms = Vec::new(); for candidate in &candidates { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { continue; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator + else { continue; }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), input, .. - } = &summary_input.expr + }) = &summary_input.operator else { continue; }; @@ -98,11 +99,11 @@ fn frequency_count_candidates_use_unit_weights() { ) { continue; } - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); assert!( dag.nodes.iter().any(|node| matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { input: actual, .. } if actual == input + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { input: actual, .. }) if actual == input )), "post-ASAP DAG must preserve the count update contract" ); @@ -135,25 +136,26 @@ fn frequency_count_candidates_use_unit_weights() { // This narrow test oracle interprets the emitted aggregate, not Prometheus ingestion, // staleness, or scrape scheduling. Unsupported plan shapes fail explicitly. fn aggregate_fixture(query: &str, series: &[Vec]) -> Vec { - use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction}; + use asap_types::ir::NonASAPOp; + use asap_types::pre_asap::{AggIntent, Reduction}; let root = lower_promql(query, AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &root + } = root.expect_non_asap() else { panic!("expected aggregate: {root:?}"); }; - match child.as_ref() { - QueryExpr::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), - QueryExpr::TimeRange { range, child } => { + match child.expect_non_asap() { + NonASAPOp::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), + NonASAPOp::TimeRange { range, child, .. } => { assert!(matches!(range.as_secs(), 1 | 300)); if range.as_secs() == 1 { assert!(series.iter().all(|samples| samples.len() == 1)); } - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } other => panic!("unsupported fixture input: {other:?}"), } @@ -225,22 +227,20 @@ fn count_over_time_counts_scrapes_not_sample_values() { #[test] fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { use asap_types::pre_asap::ColumnRef; - let root = - Rc::new(lower_promql("count_over_time(up[5m])", AccuracyTarget::Epsilon(0.02)).unwrap()); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql("count_over_time(up[5m])", AccuracyTarget::Epsilon(0.02)).unwrap(); + let candidates = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let dag = candidates .iter() .find_map(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return None; }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); dag.nodes .iter() .any(|node| { matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::Cms) }) .then_some(dag) @@ -250,7 +250,7 @@ fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { .nodes .iter() .find_map(|node| match &node.payload { - PostAsapOperatorPayload::SummaryAgg { input, .. } => Some(input), + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { input, .. }) => Some(input), _ => None, }) .unwrap(); diff --git a/crates/frontend-promql/tests/histogram_metadata.rs b/crates/frontend-promql/tests/histogram_metadata.rs index 55f35ddeb..aba373804 100644 --- a/crates/frontend-promql/tests/histogram_metadata.rs +++ b/crates/frontend-promql/tests/histogram_metadata.rs @@ -7,16 +7,17 @@ use asap_frontend_promql::{HistogramCatalog, HistogramKind}; mod support; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::ir::{NonASAPOp, OperatorNode}; +use asap_types::pre_asap::AggIntent; use asap_types::types::AccuracyTarget; use support::{lower_promql, lower_promql_with_histograms}; /// The histogram/quantile intent kind in the lowered DAG: `"HQ"` for the /// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. -fn quantile_kind(qe: &QueryExpr) -> &'static str { - fn walk(e: &QueryExpr) -> Option<&'static str> { - match e { - QueryExpr::Aggregate { +fn quantile_kind(qe: &OperatorNode) -> &'static str { + fn walk(e: &OperatorNode) -> Option<&'static str> { + match e.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => measures .iter() @@ -26,12 +27,12 @@ fn quantile_kind(qe: &QueryExpr) -> &'static str { _ => None, }) .or_else(|| walk(child)), - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Project { child, .. } => walk(child), + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } + | NonASAPOp::Project { child, .. } => walk(child), _ => None, } } @@ -48,27 +49,26 @@ fn with_meta(q: &str, catalog: HistogramCatalog) -> &'static str { #[test] fn heuristic_baseline_is_unchanged_without_a_catalog() { - // Classic `by (le)`-bucket form → HistogramQuantile; anything else → Quantile. + // Classic buckets are represented; undeclared native samples are rejected. assert_eq!( heuristic( "histogram_quantile(0.9, sum by (le) (rate(http_request_duration_seconds_bucket[5m])))" ), "HQ" ); - assert_eq!(heuristic("histogram_quantile(0.9, native_latency)"), "Q"); + assert!(lower_promql( + "histogram_quantile(0.9, native_latency)", + AccuracyTarget::Exact + ) + .is_err()); } #[test] fn declared_classic_bucket_fixes_the_false_negative() { // A classic histogram exposed WITHOUT the `_bucket` suffix and queried with - // no `le` grouping/matcher: the heuristic wrongly routes it to the - // sketch-able Quantile. Declaring it `ClassicBucket` corrects it. + // no `le` grouping/matcher requires an explicit sample-type declaration. let q = "histogram_quantile(0.9, latency_seconds)"; - assert_eq!( - heuristic(q), - "Q", - "heuristic mis-routes the suffix-less classic histogram" - ); + assert!(lower_promql(q, AccuracyTarget::Exact).is_err()); assert_eq!( with_meta( q, @@ -80,7 +80,7 @@ fn declared_classic_bucket_fixes_the_false_negative() { } #[test] -fn declared_raw_or_native_fixes_the_false_positive() { +fn declared_raw_extension_and_native_gap_override_the_heuristic() { // A metric merely NAMED `…_bucket` that actually holds raw samples / a native // histogram: the heuristic wrongly routes it to bucket interpolation. let q = "histogram_quantile(0.9, foo_bucket)"; @@ -97,14 +97,9 @@ fn declared_raw_or_native_fixes_the_false_positive() { "Q", "raw samples are sketch-able" ); - assert_eq!( - with_meta( - q, - HistogramCatalog::new().with("foo_bucket", HistogramKind::Native) - ), - "Q", - "native histograms are sketch-able" - ); + let catalog = HistogramCatalog::new().with("foo_bucket", HistogramKind::Native); + assert!(lower_promql_with_histograms(q, AccuracyTarget::Exact, catalog.clone()).is_err()); + assert!(lower_promql_with_histograms("foo_bucket", AccuracyTarget::Exact, catalog).is_err()); } #[test] @@ -119,10 +114,12 @@ fn undeclared_metric_falls_back_to_the_heuristic() { ), "HQ" ); - assert_eq!( - with_meta("histogram_quantile(0.9, native_thing)", catalog), - "Q" - ); + assert!(lower_promql_with_histograms( + "histogram_quantile(0.9, native_thing)", + AccuracyTarget::Exact, + catalog + ) + .is_err()); } #[test] diff --git a/crates/frontend-promql/tests/maintained_population_horizon.rs b/crates/frontend-promql/tests/maintained_population_horizon.rs index 88b1c88fe..b4130bf86 100644 --- a/crates/frontend-promql/tests/maintained_population_horizon.rs +++ b/crates/frontend-promql/tests/maintained_population_horizon.rs @@ -1,37 +1,33 @@ mod support; use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator}; use asap_types::post_asap::maintained_population::PopulationInput; -use asap_types::post_asap::{SummaryExpr, ValueOperation}; use asap_types::types::AccuracyTarget; -use std::rc::Rc; // A population for a one-second selector must expire members after one second. #[test] fn population_preserves_selector_horizon() { - let root = Rc::new(support::lower_promql("sum(a)", AccuracyTarget::Exact).unwrap()); + let root = support::lower_promql("sum(a)", AccuracyTarget::Exact).unwrap(); let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &candidate.expr else { + // The evaluation sits over the maintained population. + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &candidate.operator else { panic!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!() }; let PopulationInput::CurrentSeries(spec) = &population.input else { panic!() }; assert_eq!(spec.lookback_ms, 1_000); - asap_types::post_asap::compile_post_asap_dag(&candidate).unwrap(); - let asap_types::pre_asap::QueryExpr::Aggregate { child: source, .. } = root.as_ref() else { + support::post_asap_dag(&candidate); + let NonASAPOp::Aggregate { child: source, .. } = root.expect_non_asap() else { panic!() }; - assert!(spec.matches_input(source)); + assert!(spec.matches_node(source)); let mut wrong = spec.clone(); wrong.lookback_ms = 300_000; - assert!(!wrong.matches_input(source)); + assert!(!wrong.matches_node(source)); } diff --git a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs index 1b37cdee4..5e78e261d 100644 --- a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs +++ b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs @@ -29,7 +29,10 @@ use asap_frontend_promql::PromqlError as LoweringError; #[path = "../support.rs"] mod support; -use asap_types::pre_asap::{AggIntent, BinaryOpKind, CompareOpKind, QueryExpr, Reduction}; +use std::rc::Rc; + +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ScalarExpr}; +use asap_types::pre_asap::{AggIntent, BinaryOpKind, CompareOpKind, Reduction, ScalarValue}; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -44,78 +47,29 @@ fn queries() -> impl Iterator { } /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } -/// Every `AggIntent` in the DAG. -fn intents(e: &QueryExpr) -> Vec { +/// Every `AggIntent` in the tree. `AggIntent` only ever lives in +/// `Aggregate.measures`, never in a scalar position (issue #205); +/// `children()` also descends into the operators a scalar position reads. +fn intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } - // `AggIntent` only ever lives in `Aggregate.measures`, never in a - // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + fn go(e: &OperatorNode, out: &mut Vec) { + if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { + out.extend(measures.iter().cloned()); + } + for child in e.children() { + go(child, out); } } go(e, &mut out); out } -fn has bool>(e: &QueryExpr, p: F) -> bool { +fn has bool>(e: &OperatorNode, p: F) -> bool { intents(e).iter().any(p) } @@ -180,15 +134,21 @@ fn vector_vs_vector_comparison_lowers_to_binaryop() { // Both operands are instant vectors → a `BinaryOp{Compare}` of two // ingestion-interval-bounded scans. let qe = ok("node_hwmon_temp_celsius > node_hwmon_temp_max_celsius"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); assert!( - matches!(rhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } @@ -197,8 +157,8 @@ fn kube_replica_mismatch_comparison_lowers() { // Kubernetes: `kube_replicaset_spec_replicas != kube_replicaset_status_ready_replicas`. let qe = ok("kube_replicaset_spec_replicas != kube_replicaset_status_ready_replicas"); assert!(matches!( - &qe, - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) + qe.expect_non_asap(), + NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) )); } @@ -225,7 +185,11 @@ fn error_ratio_core_lowers() { // threshold: `sum(rate(failed[5m])) / sum(rate(total[5m]))` → a `BinaryOp(Div)` // of two cross-series sums over per-series rates. let qe = ok("sum(rate(litellm_proxy_failed_requests_metric_total[5m])) / sum(rate(litellm_proxy_total_requests_metric_total[5m]))"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert!(matches!(op, BinaryOpKind::Arithmetic(_))); @@ -250,11 +214,11 @@ fn all_targets_missing_core_lowers() { // Prometheus self-monitoring `sum by (job) (up)` (the corpus query is // `… == 0`). Cross-series sum grouped positionally on `job`. let qe = ok("sum by (job) (up)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; @@ -273,19 +237,17 @@ fn all_targets_missing_core_lowers() { #[test] fn scalar_threshold_comparisons_lower_to_binaryop_scalar() { // ~822/949 corpus queries are ` `. The numeric - // threshold is now a `PromqlScalarBridge` operand of the `BinaryOp` (issue + // threshold is now a `ScalarExpr` operand of the `BinaryOp` (issue // #35) — the single biggest unblock for real alerts. for q in [ "prometheus_config_last_reload_successful != 1", "increase(prometheus_tsdb_compactions_failed_total[1m]) > 0", "rate(alertmanager_notifications_failed_total[3m]) > 0.05", ] { - let QueryExpr::BinaryOp { rhs, .. } = ok(q) else { - panic!("expected a BinaryOp for {q:?}"); - }; + let qe = ok(q); assert!( - matches!(rhs.as_ref(), QueryExpr::PromqlScalarBridge(_)), - "scalar threshold operand for {q:?}, got {rhs:?}" + matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. }), + "{q}" ); } } @@ -336,12 +298,12 @@ fn vector_literal_lowers_to_a_labelless_vector() { // `vector(1)` — used in dead-man's-switch ("always firing") alerts. Now // lowers to a `PromqlVectorFromScalar` over the scalar `1` (issue #48). let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; - assert_eq!(inner.as_promql_scalar(), Some(1.0)); + assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); // The result is a vector: it carries a time index (unlike a bare scalar). - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(qe.schema.time_index.is_some()); } #[test] @@ -352,14 +314,14 @@ fn without_grouping_lowers_to_the_exclusion_form() { // labels are stored and the kept set is runtime-resolved (issue #39). let qe = ok(r#"(min without (cpu) (rate(node_cpu_seconds_total{mode="idle"}[1h]))) > 0.8"#); // Top level is the `> 0.8` comparison; the `min without (cpu)` is its LHS. - let QueryExpr::BinaryOp { lhs, .. } = &qe else { + let NonASAPOp::Filter { child: lhs, .. } = qe.expect_non_asap() else { panic!("expected a comparison BinaryOp, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = lhs.as_ref() + } = lhs.expect_non_asap() else { panic!("expected a `min without` Aggregate on the LHS, got {lhs:?}"); }; diff --git a/crates/frontend-promql/tests/observability/metrics_observability.rs b/crates/frontend-promql/tests/observability/metrics_observability.rs index 87d653ef2..d40df6236 100644 --- a/crates/frontend-promql/tests/observability/metrics_observability.rs +++ b/crates/frontend-promql/tests/observability/metrics_observability.rs @@ -6,15 +6,14 @@ use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{retain_exact, RealizationError}; use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_frontend_promql::PromqlError; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -64,19 +63,18 @@ fn queries(corpus: &str) -> impl Iterator { .filter(|line| !line.is_empty() && !line.starts_with('#')) } -fn post_asap_candidate(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn post_asap_candidate(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. }) => Ok(node), - _ => keep_pre_asap(&root), + _ => retain_exact(root), } } @@ -99,9 +97,8 @@ fn benchmark_corpora_are_total_and_report_coverage() { Ok(expr) => { lowered += 1; match post_asap_candidate(&expr) { - Ok(node) if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) => { - post_asap_candidates += 1 - } + // An ASAP operator bound somewhere below the root. + Ok(node) if node.contains_asap() => post_asap_candidates += 1, Ok(_) => { post_asap_unchanged += 1; if std::env::var_os("METRICS_OBSERVABILITY_REPORT").is_some() { diff --git a/crates/frontend-promql/tests/observability/promql_corpus.rs b/crates/frontend-promql/tests/observability/promql_corpus.rs index 1bc7e2166..45afdeb5f 100644 --- a/crates/frontend-promql/tests/observability/promql_corpus.rs +++ b/crates/frontend-promql/tests/observability/promql_corpus.rs @@ -15,37 +15,35 @@ use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{retain_exact, RealizationError}; use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_frontend_promql::PromqlError as LoweringError; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; -/// This crate has no "bind me one DAG" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// This crate has no "bind me one dag" public API any more — +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so [`bind_tally`] /// gets one representative `Result` per query, matching what a totality /// check over the whole corpus wants. -fn bind(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn bind(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. }) => Ok(node), - _ => keep_pre_asap(&root), + _ => retain_exact(root), } } @@ -77,7 +75,10 @@ impl Tally { fn tally(corpus: &str) -> Tally { let mut t = Tally::default(); for q in queries(corpus) { - match lower_promql(q, AccuracyTarget::Exact) { + match asap_frontend_promql::lower_promql_query_workload( + &support::workload(q, AccuracyTarget::Exact), + 0, + ) { Ok(_) => t.lowered += 1, Err(LoweringError::Parse(_)) => t.unparseable += 1, Err(_) => t.rejected += 1, @@ -93,9 +94,10 @@ fn tally(corpus: &str) -> Tally { /// arm). #[derive(Default, Debug)] struct BindTally { - /// Root bound to `SummaryAgg`/`SummaryEstimate` — the pass did something. + /// An ASAP operator was bound somewhere below the root — the pass did + /// something. transformed: usize, - /// Root stayed `KeepPreAsap` — the pass left the query untouched. + /// The kept pre-ASAP dag — the pass left the query untouched. unchanged: usize, /// [`bind`] returned `Err` (schema derivation failed). errored: usize, @@ -108,7 +110,7 @@ fn bind_tally(corpus: &str, accuracy: AccuracyTarget) -> BindTally { continue; }; match bind(&dag) { - Ok(bound) if matches!(bound.expr, SummaryExpr::KeepPreAsap(_)) => t.unchanged += 1, + Ok(bound) if !bound.contains_asap() => t.unchanged += 1, Ok(_) => t.transformed += 1, Err(_) => t.errored += 1, } @@ -148,23 +150,17 @@ fn lowering_is_total_over_the_entire_corpus() { "testdata corpus unexpectedly small: {td:?}" ); - // Coverage tripwire: a code change that breaks lowering for a large slice of - // real PromQL trips this. Current numbers on the private promql-parser `asap` - // branch: docs 48 lowered / 1 rejected, testdata 1512 lowered / 76 rejected / - // 235 unparseable. The floors sit ~1% under those, so they guard regressions - // rather than pin an exact count — ratchet them up as coverage lands. - // - // The 235 unparseable are parser-fork gaps (issue #108); the rejections are - // lowering gaps (#109). Both shrink over time, so these floors normally only - // rise. Exception: the testdata floor was lowered to the measured 1485 when - // the 44 `fill` vector-matching queries became rejected rather than - // silently lowered without their fill semantics. + // Coverage tripwire after rejecting unrepresented native histogram samples: + // docs 48 lowered / 1 rejected; testdata 1121 lowered / 469 rejected / + // 233 parser gaps. Earlier coverage counted native histogram operations + // incorrectly treated as float quantiles. Keep the rejection cases in the + // corpus: accepting them requires a native histogram sample representation. assert!( docs.lowered >= 47, "docs lowering coverage regressed: {docs:?}" ); assert!( - td.lowered >= 1485, + td.lowered >= 1121, "testdata lowering coverage regressed: {td:?}" ); } diff --git a/crates/frontend-promql/tests/promql_binding_regressions.rs b/crates/frontend-promql/tests/promql_binding_regressions.rs index 26d1be06d..416b6680d 100644 --- a/crates/frontend-promql/tests/promql_binding_regressions.rs +++ b/crates/frontend-promql/tests/promql_binding_regressions.rs @@ -33,9 +33,10 @@ fn irate_and_rate_have_distinct_canonical_intents() { /// PromQL count counts series even when two sample values are equal. #[test] fn count_is_row_count_not_distinct_sample_value_count() { - use asap_types::pre_asap::{AggIntent, QueryExpr}; - let dag = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { measures, .. } = dag else { + use asap_types::ir::NonASAPOp; + use asap_types::pre_asap::AggIntent; + let tree = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); + let NonASAPOp::Aggregate { measures, .. } = tree.expect_non_asap() else { panic!("expected aggregate") }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); diff --git a/crates/frontend-promql/tests/promql_conformance.rs b/crates/frontend-promql/tests/promql_conformance.rs index 6f932879e..2f3fa2bf8 100644 --- a/crates/frontend-promql/tests/promql_conformance.rs +++ b/crates/frontend-promql/tests/promql_conformance.rs @@ -31,22 +31,26 @@ // `__GAP`-suffixed test names intentionally SHOUT the documented divergences. #![allow(non_snake_case)] +use std::rc::Rc; use std::time::Duration; use asap_frontend_promql::PromqlError as LoweringError; mod support; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, +}; use asap_types::pre_asap::schema::DataType; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, MathFunc, - PromQLVectorSetOpKind, QueryExpr, Reduction, SampleKind, Source, TimeFunc, + AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, PromQLVectorSetOpKind, + Reduction, SampleKind, ScalarValue, Source, TimeFunc, }; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, promql_scalar}; // ── harness helpers ───────────────────────────────────────────────────────────── /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } @@ -59,72 +63,30 @@ fn rejected(q: &str) -> LoweringError { } } -/// Every `AggIntent` anywhere in the DAG, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { +/// Every `AggIntent` anywhere in the tree, root-to-leaf. +fn intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); collect(e, &mut out); out } -fn collect(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - collect(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => collect(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { - collect(lhs, out); - collect(rhs, out); - } - QueryExpr::Join { left, right, .. } | QueryExpr::SetOp { left, right, .. } => { - collect(left, out); - collect(right, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| collect(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - collect(inner, out) - } - // `AggIntent` only ever lives in `Aggregate.measures`, never in a - // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} +/// `AggIntent` only ever lives in `Aggregate.measures`, never in a scalar +/// position (issue #205); `children()` also descends into the operators a +/// scalar position reads (`scalar(v)`). +fn collect(e: &OperatorNode, out: &mut Vec) { + if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { + out.extend(measures.iter().cloned()); + } + for child in e.children() { + collect(child, out); } } /// The first `Scan` reached by descending single-child nodes, with its metric /// name and predicate count. -fn first_scan(e: &QueryExpr) -> (String, usize) { - match e { - QueryExpr::Scan { +fn first_scan(e: &OperatorNode) -> (String, usize) { + match e.expect_non_asap() { + NonASAPOp::Scan { source, predicates, .. } => { let name = match source { @@ -133,45 +95,32 @@ fn first_scan(e: &QueryExpr) -> (String, usize) { }; (name, predicates.len()) } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_scan(child), + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_scan(child), other => panic!("no Scan reachable from {other:?}"), } } -fn has bool>(e: &QueryExpr, pred: F) -> bool { +fn has bool>(e: &OperatorNode, pred: F) -> bool { intents(e).iter().any(pred) } -/// Whether the DAG contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary +/// Whether the tree contains a `Mul`-by-`ScalarExpr(-1)` anywhere — the shape unary /// negation lowers to (issue #36). -fn negates_via_scalar(e: &QueryExpr) -> bool { - let is_neg_one = |q: &QueryExpr| { - q.as_promql_scalar() - .is_some_and(|v| (v + 1.0).abs() < 1e-12) - }; - match e { - QueryExpr::BinaryOp { op, lhs, rhs, .. } => { - (*op == BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul) - && (is_neg_one(lhs) || is_neg_one(rhs))) - || negates_via_scalar(lhs) - || negates_via_scalar(rhs) - } - QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Project { child, .. } => negates_via_scalar(child), - _ => false, +fn negates_via_scalar(e: &OperatorNode) -> bool { + fn negative(expr: &ScalarExpr) -> bool { + matches!(expr, ScalarExpr::Negative { .. }) || expr.children().iter().any(|e| negative(e)) } + e.expect_non_asap() + .scalar_exprs() + .iter() + .any(|e| negative(e)) + || e.children().iter().any(|e| negates_via_scalar(e)) } // ───────────────────────────────────────────────────────────────────────────── @@ -193,10 +142,10 @@ fn promql_scan_schema_is_open() { // runtime-only, so the binding schema lists only the (ts, value) floor + // referenced labels and may be a subset of the runtime row. let qe = ok("node_cpu_seconds_total"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected a TimeRange for a bare selector, got {qe:?}"); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let NonASAPOp::Scan { schema, .. } = child.expect_non_asap() else { panic!("expected a Scan inside the TimeRange, got {qe:?}"); }; assert!( @@ -241,7 +190,7 @@ fn range_vector_selector_is_time_range() { // SEMANTICS: `[5m]` turns an instant vector into a range vector, // represented in the canonical DAG as a dedicated `TimeRange` node. let qe = ok("node_cpu_seconds_total[5m]"); - let QueryExpr::TimeRange { range, .. } = &qe else { + let NonASAPOp::TimeRange { range, .. } = qe.expect_non_asap() else { panic!("expected TimeRange for a range-vector selector, got {qe:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -252,19 +201,44 @@ fn range_vector_selector_is_time_range() { // functions.test) // ───────────────────────────────────────────────────────────────────────────── +#[test] +fn selector_time_ranges_carry_their_kind() { + // SEMANTICS: an instant selector reads the latest sample within the + // ingestion interval (`Instant`); `m[5m]` is a range selection (`Range`). + // Same length is not the same shape: `m` and `m[1s]` stay distinct. + assert!(matches!( + ok("node_cpu_seconds_total").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Instant, + .. + } + )); + assert!(matches!( + ok("node_cpu_seconds_total[5m]").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); + assert_ne!( + ok("node_cpu_seconds_total"), + ok("node_cpu_seconds_total[1s]") + ); +} + #[test] fn rate_range_lives_in_time_range_node() { // SEMANTICS: per-second average rate; the temporal range lives on the // enclosing `TimeRange` node, not inside the intent. let qe = ok("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -281,14 +255,14 @@ fn irate_maps_to_its_own_intent() { #[test] fn increase_range_lives_in_time_range_node() { let qe = ok("increase(http_requests_total[1h])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -303,7 +277,7 @@ fn increase_range_lives_in_time_range_node() { fn sum_collapses_all_series() { // SEMANTICS: `sum(v)` → one output series. No grouping → no Partition. let qe = ok("sum(node_filesystem_size_bytes)"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -314,12 +288,12 @@ fn sum_by_groups_via_positional_aggregate() { // name-based Partition). SchemaResolver leaf = [ts, value, instance, job] (referenced // keys appended sorted), so the keys resolve to columns [2, 3]. let qe = ok("sum by(job, instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected positional Aggregate for `by(...)`, got {qe:?}"); }; @@ -330,7 +304,7 @@ fn sum_by_groups_via_positional_aggregate() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } @@ -369,11 +343,11 @@ fn sum_without_groups_by_the_complement() { // the runtime: the grouping is the exclusion form and the output schema // stays OPEN (unlike `by`, which freezes to closed). let qe = ok("sum without(instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -385,7 +359,7 @@ fn sum_without_groups_by_the_complement() { assert_eq!(by.keys().len(), 1, "the one excluded label (instance)"); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - !qe.output_schema().unwrap().closed, + !qe.schema.clone().closed, "a `without` result keeps an open schema (kept label set is runtime-only)" ); } @@ -418,16 +392,16 @@ fn group_aggregator_lowers_to_a_distinct_intent() { fn sum_of_rate_is_two_levels() { // SEMANTICS: per-series rate, THEN cross-series sum. Both must survive. let qe = ok("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -436,12 +410,12 @@ fn sum_by_of_rate_groups_outer_level() { // Outer cross-series Sum grouped on positional `Aggregate.by` over the // label-preserving inner Rate. Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (rate(node_network_receive_bytes_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by instance, got {qe:?}"); }; @@ -449,8 +423,8 @@ fn sum_by_of_rate_groups_outer_level() { assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // child is the inner per-series Rate aggregate. assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -461,26 +435,29 @@ fn sum_by_of_over_time_groups_outer_level() { // preserving, so the key resolves positionally just like the rate case (no // name-based Partition). Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (avg_over_time(node_cpu_seconds_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by instance, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // child is the inner per-series reduction: Aggregate{Avg} over TimeRange. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (per-series avg_over_time) under the Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ───────────────────────────────────────────────────────────────────────────── @@ -501,7 +478,7 @@ fn over_time_functions_reduce_over_time_range() { ] { let qe = ok(q); assert!( - matches!(&qe, QueryExpr::Aggregate { .. }), + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. }), "{q}: expected Aggregate" ); let matched = intents(&qe).iter().any(|i| match want { @@ -519,7 +496,7 @@ fn over_time_functions_reduce_over_time_range() { #[test] fn quantile_over_time_is_aggregate_over_time_range() { let qe = ok("quantile_over_time(0.9, request_latency_seconds[5m])"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has( &qe, |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) @@ -536,7 +513,7 @@ fn histogram_quantile_over_rate() { // φ-quantile from bucket rates. The `_bucket` metric marks the classic // cumulative-bucket form → `HistogramQuantile` (even without `sum by (le)`). let qe = ok("histogram_quantile(0.9, rate(demo_api_request_duration_seconds_bucket[5m]))"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected Aggregate{{HistogramQuantile}}, got {qe:?}"); }; assert!( @@ -553,9 +530,9 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { let qe = ok( "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", ); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; @@ -566,11 +543,11 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { )); // `sum by(le)` now survives as a positional Aggregate (by = [2], `le`), over // the inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected `sum by(le)` as a positional Aggregate, got {child:?}"); }; @@ -586,7 +563,11 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { #[test] fn vector_arithmetic() { let qe = ok("node_memory_MemFree_bytes + node_memory_Cached_bytes"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Add)); @@ -597,14 +578,17 @@ fn on_matching_with_group_left() { // SEMANTICS: many-to-one matching on a label subset. let qe = ok("rate(demo_cpu_usage_seconds_total[1m]) / on(instance, job) group_left demo_num_cpus"); - let QueryExpr::BinaryOp { - op, vector_match, .. - } = &qe - else { + let NonASAPOp::BinaryOp { operator, .. } = qe.expect_non_asap() else { panic!("expected BinaryOp, got {qe:?}"); }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); - let vm = vector_match.as_ref().expect("on(...) group_left present"); + assert_eq!( + operator.kind, + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + ); + let vm = operator + .vector_match + .as_ref() + .expect("on(...) group_left present"); assert_eq!(vm.labels, vec!["instance".to_string(), "job".to_string()]); assert!( vm.grouping.is_some(), @@ -617,15 +601,51 @@ fn vector_comparison_filters() { // SEMANTICS: `>` between two vectors keeps the LHS series where it holds. let qe = ok("go_goroutines > go_threads"); assert!( - matches!(&qe, QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) + matches!(qe.expect_non_asap(), NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) ); } +#[test] +fn comparison_bool_modifier_returns_zero_or_one() { + // SEMANTICS (operators.test): `bool` turns a filtering comparison into a + // 0/1-valued one. On a vector operand it is `return_bool` on the + // `BinaryOp`; between two scalars it is a `Case(Compare → 1, else 0)` + // scalar expression under PromQL numeric rules — and a scalar comparison + // without `bool` is not a PromQL expression at all. + let bool_flag = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { return_bool, .. } => *return_bool, + NonASAPOp::Project { .. } => true, + NonASAPOp::Filter { .. } => false, + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert!(bool_flag("go_goroutines > bool go_threads")); + assert!(bool_flag("go_goroutines > bool 0")); + assert!(!bool_flag("go_goroutines > go_threads")); + assert!(!bool_flag("go_goroutines > 0")); + + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { branches, .. } = &qe else { + panic!("expected a scalar Case, got {qe:?}"); + }; + assert!(matches!( + branches.as_slice(), + [( + ScalarExpr::Compare { + op: CompareOpKind::Lt, + semantics: ExprSemantics::Promql, + .. + }, + _ + )] + )); + rejected("1 < 2"); +} + #[test] fn unary_negation_lowers_as_multiply_by_minus_one() { // SEMANTICS (PromQL, issue #36): `-expr` flips the sign of every sample. // Now that a scalar operand exists (#35), it lowers as `expr * -1` — a `Mul` - // BinaryOp of the (label-preserving) vector against `PromqlScalarBridge(-1)`. These are + // BinaryOp of the (label-preserving) vector against `ScalarExpr(-1)`. These are // the five cases the old `__GAP` test pinned as rejected. for q in [ "-rate(http_errors_total[5m])", @@ -635,93 +655,38 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { "sum(-node_cpu_seconds_total)", ] { let qe = ok(q); - // A `Mul`-by-`-1` against a `PromqlScalarBridge(-1)` appears somewhere in every DAG. + // A `Mul`-by-`-1` against a `ScalarExpr(-1)` appears somewhere in every tree. assert!( negates_via_scalar(&qe), "no `* -1` negation found in {q}: {qe:?}" ); } - // `-some_metric` at the root: `Scan * PromqlScalarBridge(-1)`, schema follows the vector. - let QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, - } = &ok("-some_metric") - else { - panic!("expected a BinaryOp for `-some_metric`"); - }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), - "vector on the left" - ); - assert!( - rhs.as_promql_scalar() - .is_some_and(|v| (v + 1.0).abs() < 1e-12), - "negation multiplies by PromqlScalarBridge(-1), got {rhs:?}" - ); - assert!( - vector_match.is_none(), - "scalar negation carries no vector match" - ); - // Label-preserving: the schema is the vector operand's, unchanged. - let schema = ok("-some_metric").output_schema().unwrap(); - assert_eq!( - schema - .fields - .iter() - .map(|c| c.name.as_str()) - .collect::>(), - vec!["ts", "value"], - ); - - // `sum(-m)` — the negation lowers inside the aggregate argument (issue #27 - // nesting), so the outer node is the `Sum` aggregate over the `Mul`. - let QueryExpr::Aggregate { - measures, child, .. - } = &ok("sum(-node_cpu_seconds_total)") - else { - panic!("expected an outer Aggregate for `sum(-m)`"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!( - child.as_ref(), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - .. - } - )); + let negated = ok("-some_metric"); + assert!(negates_via_scalar(&negated)); + assert!(negated.schema.has_promql_series_identity()); + assert!(negated.schema.time_index.is_some()); + let summed = ok("sum(-node_cpu_seconds_total)"); + assert!(has(&summed, |i| matches!(i, AggIntent::Sum { .. }))); + assert!(negates_via_scalar(&summed)); } #[test] fn unary_negation_of_constant_folds_to_scalar() { // `-(10*1024*1024)` — the operand is constant-foldable, so negation collapses - // to a single negated `PromqlScalarBridge` leaf (no `BinaryOp`), just like a bare literal. - assert!(ok("-(10*1024*1024)") - .as_promql_scalar() + // to a single negated `ScalarExpr` leaf (no `BinaryOp`), just like a bare literal. + assert!(promql_scalar(&support::scalar_root("-(10*1024*1024)")) .is_some_and(|v| (v + 10_485_760.0).abs() < 1e-6)); } #[test] fn double_unary_negation_nests() { - // `- -some_metric` — negation of a negation: `(m * -1) * -1`. Both levels - // lower; the value is unchanged but the structure is faithfully nested. - let QueryExpr::BinaryOp { op, lhs, .. } = &ok("- -some_metric") else { - panic!("expected outer BinaryOp for `- -some_metric`"); + let qe = ok("- -some_metric"); + let NonASAPOp::Project { child, .. } = qe.expect_non_asap() else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!( - matches!( - lhs.as_ref(), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - .. - } - ), - "inner negation nests under the outer one" - ); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); + assert!(negates_via_scalar(child)); } #[test] @@ -754,39 +719,22 @@ fn count_maps_to_count_and_inherits_accuracy() { #[test] fn scalar_literal_operand_lowers_as_binaryop_scalar() { - // Issue #35: ` op ` — the numeric threshold is a - // `PromqlScalarBridge` operand of the `BinaryOp`, and constant arithmetic - // (`10*1024*1024`) is folded. The output schema is the vector side's. let qe = ok("node_filesystem_avail_bytes > 10*1024*1024"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Compare { op, right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); - assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), - "vector on the left" - ); - assert!( - rhs.as_promql_scalar() - .is_some_and(|v| (v - 10_485_760.0).abs() < 1e-6), - "folded scalar threshold on the right, got {rhs:?}" - ); - // Schema derivation follows the vector side (a scalar contributes no labels). - assert!(qe.output_schema().is_ok()); + assert_eq!(*op, CompareOpKind::Gt); + assert_eq!(promql_scalar(right), Some(10_485_760.0)); } #[test] fn scalar_arithmetic_scales_the_vector() { - // `rate(m[5m]) * 100` — a unit conversion. Arithmetic BinaryOp of the vector - // with a `PromqlScalarBridge(100)`. let qe = ok("rate(m[5m]) * 100"); - let QueryExpr::BinaryOp { op, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Arithmetic { op, right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!(rhs - .as_promql_scalar() - .is_some_and(|v| (v - 100.0).abs() < 1e-9)); + assert_eq!(*op, ArithmeticOpKind::Mul); + assert_eq!(promql_scalar(right), Some(100.0)); } // ───────────────────────────────────────────────────────────────────────────── @@ -797,12 +745,22 @@ fn scalar_arithmetic_scales_the_vector() { #[test] fn set_ops_lower_to_binaryop() { // SEMANTICS: or = union of label sets; and = intersection; unless = difference. - assert!(matches!(&ok("up{job=\"a\"} or up{job=\"b\"}"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Or))); - assert!(matches!(&ok("node_network_mtu_bytes and node_up"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::And))); - assert!(matches!(&ok("node_network_mtu_bytes unless node_down"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Unless))); + let set_op = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { operator, .. } => operator.kind.clone(), + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert_eq!( + set_op("up{job=\"a\"} or up{job=\"b\"}"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Or) + ); + assert_eq!( + set_op("node_network_mtu_bytes and node_up"), + BinaryOpKind::Set(PromQLVectorSetOpKind::And) + ); + assert_eq!( + set_op("node_network_mtu_bytes unless node_down"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) + ); } // ───────────────────────────────────────────────────────────────────────────── @@ -824,7 +782,7 @@ fn topk_over_count_is_heavy_hitter() { fn bottomk_is_generic_sort_limit() { // SEMANTICS: bottom-k → generic ascending order + limit (no sketch). let qe = ok("bottomk(3, count_over_time(http_requests_total[5m]))"); - assert!(matches!(&qe, QueryExpr::Limit { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); } #[test] @@ -833,9 +791,9 @@ fn topk_over_nested_sum_preserves_weighted_topk_accuracy() { // The final rates are query-time values. Their ordering does not establish // frequency-sketch membership semantics. let qe = ok("topk(3, sum by(instance) (rate(node_cpu_seconds_total[5m])))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected weighted TopK aggregate, got {qe:?}"); }; @@ -861,18 +819,18 @@ fn outer_aggregate_over_nested_aggregate_nests() { // flat two-level template rejected. Each level survives into the // canonical DAG (issue #27). let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner `sum by (job)` Aggregate, got {child:?}"); }; @@ -895,12 +853,12 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { // the query lowers with the provably-absent key dropped, exactly // `sum(sum by (group)(…))`. let qe = ok(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -910,7 +868,7 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { "absent `job` key dropped → global aggregate" ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { reduction, .. } = child.expect_non_asap() else { panic!("expected inner `sum by (group)` Aggregate, got {child:?}"); }; assert_eq!( @@ -927,16 +885,16 @@ fn outer_group_key_present_after_inner_aggregate_still_resolves() { // resolving positionally — the absent-key drop only fires on provable // absence, never on a resolvable key. let qe = ok("sum(sum by (job, group)(http_requests)) by (job)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate, got {child:?}"); }; @@ -957,9 +915,9 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // still resolve. Each `or` side is bound independently against its own // sub-DAG, so the key is seeded as an inherited column on both sides. let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -969,12 +927,12 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { 1, "grouped by the one `__name__` key" ); - let QueryExpr::BinaryOp { lhs, rhs, .. } = child.as_ref() else { + let NonASAPOp::BinaryOp { lhs, rhs, .. } = child.expect_non_asap() else { panic!("expected a BinaryOp child, got {child:?}"); }; // Both independently-bound sides carry `__name__` at the same position, so // the outer group key is consistent across the union. - let (ls, rs) = (lhs.output_schema().unwrap(), rhs.output_schema().unwrap()); + let (ls, rs) = (lhs.schema.clone(), rhs.schema.clone()); assert_eq!(ls.column_id("__name__"), rs.column_id("__name__")); assert_eq!( ls.column_id("__name__"), @@ -983,8 +941,8 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // The general case (a plain label, not just `__name__`) also lowers. assert!(matches!( - ok("sum by (job)(metric_a or metric_b)"), - QueryExpr::Aggregate { .. } + ok("sum by (job)(metric_a or metric_b)").expect_non_asap(), + NonASAPOp::Aggregate { .. } )); } @@ -994,15 +952,15 @@ fn aggregate_over_binary_op_nests() { // op over two range vectors. The old template only accepted a single inner // selector/call; now the binary op lowers and the outer sum wraps it. let qe = ok("sum(rate(a[5m]) + rate(b[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::BinaryOp { .. }), + matches!(child.expect_non_asap(), NonASAPOp::BinaryOp { .. }), "argument lowers as a BinaryOp, got {child:?}" ); } @@ -1015,7 +973,10 @@ fn aggregate_over_binary_op_nests() { fn subquery_wraps_inner_query() { // SEMANTICS: `[range:res]` evaluates the inner query across a range. let qe = ok("rate(demo_api_request_duration_seconds_count[5m])[1h:]"); - assert!(matches!(&qe, QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); } @@ -1026,12 +987,12 @@ fn over_time_of_subquery_reduces_per_series() { // then `max_over_time` takes the max of those samples *per series*. It lowers // to a per-series `Max` reduction over a `PromqlSubquery` (issue #27). let qe = ok("max_over_time(rate(demo_api_request_duration_seconds_count[5m])[1h:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate at the root, got {qe:?}"); }; @@ -1044,7 +1005,7 @@ fn over_time_of_subquery_reduces_per_series() { // The reduction rides directly on the sub-query (the structural range marker // that keeps it label-preserving), which wraps the inner `rate`. assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), "the `Max` reduces over a PromqlSubquery, got {child:?}" ); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Rate))); @@ -1055,16 +1016,19 @@ fn quantile_over_time_of_subquery_carries_phi() { // The `quantile_over_time` φ parameter is read from arg 0; the sub-query is // arg 1. It lowers to a per-series `Quantile(φ)` over the `PromqlSubquery`. let qe = ok("quantile_over_time(0.9, rate(demo[5m])[1h:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.9).abs() < 1e-9) ); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); } #[test] @@ -1074,12 +1038,12 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { // survives for the OUTER cross-series `sum by (job)` to group on. If the // inner `Max` collapsed labels, `job` would not resolve here. let qe = ok("sum by (job) (max_over_time(rate(demo{job=\"api\"}[5m])[1h:]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1089,20 +1053,20 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner node is the per-series `max_over_time` reduction over the subquery. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, measures: inner_measures, child: inner_child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate, got {child:?}"); }; assert_eq!(inner_reduction, &Reduction::PerEntity); assert!(matches!(inner_measures.as_slice(), [AggIntent::Max { .. }])); assert!(matches!( - inner_child.as_ref(), - QueryExpr::PromqlSubquery { .. } + inner_child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } )); } @@ -1124,66 +1088,66 @@ fn nested_subquery_from_prometheus_docs() { // the label-preserving `[ts, value]`. let qe = ok("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected `max_over_time` Aggregate at the root, got {qe:?}"); }; assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range, resolution, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the outer `[10m:]` PromqlSubquery, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(600)); assert_eq!(*resolution, None, "`[10m:]` keeps the default resolution"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the `deriv` Aggregate, got {child:?}"); }; assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Deriv])); - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range, resolution, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the inner `[30s:5s]` PromqlSubquery, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(30)); assert_eq!(*resolution, Some(Duration::from_secs(5))); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the `rate` Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected the `[5s]` TimeRange under rate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(5)); // Per-series end to end: the schema keeps the (ts, value) floor and stays open. - let schema = qe.output_schema().expect("schema derivation"); + let schema = qe.schema.clone(); assert_eq!( schema .fields @@ -1205,21 +1169,22 @@ fn offset_modifier_lowers_to_a_time_shift() { // past — a `TimeShift` wrapper over the selector (signed ms; a negative // offset shifts forward). Schema is unchanged (the shift only moves *when*). let qe = ok("http_requests_total offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.offset_ms, 300_000); assert!(shift.at.is_none()); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); // `offset -5m` shifts forward → negative ms. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total offset -5m") else { + let qe = ok("http_requests_total offset -5m"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift"); }; assert_eq!(shift.offset_ms, -300_000); @@ -1230,29 +1195,31 @@ fn at_modifier_lowers_to_a_time_shift() { // SEMANTICS (PromQL, issue #40): `@ ` pins the evaluation to an absolute // instant (PromQL seconds → IR milliseconds); `@ start()` / `@ end()` anchor // to the query range bounds. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ 1609746000") else { + let qe = ok("http_requests_total @ 1609746000"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift for `@ `"); }; assert_eq!(shift.at, Some(AtModifier::Timestamp(1_609_746_000_000))); assert_eq!(shift.offset_ms, 0); - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ start()") else { + let qe = ok("http_requests_total @ start()"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift for `@ start()`"); }; assert_eq!(shift.at, Some(AtModifier::Start)); // Offset and `@` compose: `@ end() offset 5m` carries both. let qe = ok("http_requests_total @ end() offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.at, Some(AtModifier::End)); @@ -1265,21 +1232,21 @@ fn offset_on_a_ranged_selector_wraps_inside_the_time_range() { // `TimeShift` sits *under* the `TimeRange` (the 5m window is taken at the // shifted time), and the whole thing under the per-series `Rate` (#40). let qe = ok("rate(http_requests_total[5m] offset 1h)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected the rate Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { child, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { child, .. } = child.expect_non_asap() else { panic!("expected a TimeRange under rate, got {child:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("expected a TimeShift under the TimeRange, got {child:?}"); }; assert_eq!(shift.offset_ms, 3_600_000); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } // ───────────────────────────────────────────────────────────────────────────── @@ -1313,7 +1280,7 @@ fn count_over_time_value_column_is_float64() { // #69: a per-series range reduction produces a PromQL sample value, which is // always float64. `count_over_time`'s `Count` intent types `Int64`, but the // derived `value` column must be `Float64` like every other range reducer. - let schema = ok("count_over_time(m[5m])").output_schema().unwrap(); + let schema = ok("count_over_time(m[5m])").schema.clone(); let value = schema .fields .iter() @@ -1335,12 +1302,12 @@ fn counter_derivative_functions_lower_to_distinct_intents() { ("resets(m[1h])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate for {q:?}, got {qe:?}"); }; @@ -1355,7 +1322,7 @@ fn counter_derivative_functions_lower_to_distinct_intents() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { .. }), + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. }), "{q}: reduction rides on a TimeRange, got {child:?}" ); } @@ -1366,9 +1333,9 @@ fn predict_linear_carries_horizon_seconds() { // `predict_linear(v[w], t)` — the 2nd (scalar) arg is the prediction horizon // in seconds; it must be carried in the intent (it changes the result). let qe = ok("predict_linear(node_filesystem_avail_bytes[3h], 86400)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -1376,7 +1343,10 @@ fn predict_linear_carries_horizon_seconds() { measures.as_slice(), &[AggIntent::PredictLinear { seconds: 86400.0 }] ); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -1394,12 +1364,12 @@ fn aggregation_over_counter_derivative_keeps_labels() { // A counter-derivative is per-series (label-preserving), so an outer // `sum by (job)` can group on a label the inner `changes` preserved. let qe = ok(r#"sum by (job) (changes(m{job="api"}[15m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1420,12 +1390,12 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { // grouped outer (`avg by (dc)`) must resolve its key against the labels the // inner reduction preserved, threading any scalar param (predict horizon). let qe = ok("avg by (dc) (predict_linear(m[3h], 3600))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1434,11 +1404,11 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { "outer `avg by (dc)` groups on a label" ); assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, measures: inner_measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner per-series Aggregate, got {child:?}"); }; @@ -1458,11 +1428,14 @@ fn topk_over_counter_derivative_is_generic_sort_limit() { // `topk(k, deriv(...))` ranks the per-series derivative values — a generic // `Sort + Limit`, NOT a heavy-hitter `TopK` (that's only `count_over_time`). let qe = ok("topk(3, deriv(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - assert!(matches!(child.as_ref(), QueryExpr::Sort { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Sort { .. })); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Deriv))); assert!( !intents(&qe) @@ -1477,28 +1450,37 @@ fn counter_derivative_composes_in_binary_ops() { // As a vector operand: `delta(a[5m]) / delta(b[5m])` is a BinaryOp of two // per-series Delta reductions. let ratio = ok("delta(a[5m]) / delta(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &ratio else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = ratio.expect_non_asap() + else { panic!("expected BinaryOp, got {ratio:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); // Under an aggregate over a binary op mixing a counter-derivative with // another per-series function: `sum(rate(m[5m]) + changes(m[5m]))`. let mixed = ok("sum(rate(m[5m]) + changes(m[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &mixed + } = mixed.expect_non_asap() else { panic!("expected Aggregate, got {mixed:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::BinaryOp { .. } + )); assert!(intents(&mixed).iter().any(|i| matches!(i, AggIntent::Rate))); assert!(intents(&mixed) .iter() @@ -1522,12 +1504,12 @@ fn range_functions_over_a_subquery_reduce_per_series() { ("resets(sum(m)[5m:])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("{q}: expected an Aggregate, got {qe:?}"); }; @@ -1542,7 +1524,7 @@ fn range_functions_over_a_subquery_reduce_per_series() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), "{q}: reduces directly over the PromqlSubquery (no TimeRange), got {child:?}" ); } @@ -1570,8 +1552,8 @@ fn predict_linear_and_double_exp_over_a_subquery_carry_params() { #[test] fn histogram_quantile_classic_bucket_vs_native() { // Two lowerings of `histogram_quantile(φ, …)`: the classic cumulative-bucket - // form → exact `HistogramQuantile`; a native-histogram / raw-samples argument - // → the generic (sketch-able) `Quantile`. The classic form is recognised by + // form → exact `HistogramQuantile`; native samples require a new type. + // The classic form is recognised by // `by (le)`, a `_bucket` metric, or an `le` matcher (issue #43). for classic in [ "histogram_quantile(0.9, sum by (le) (rate(x_bucket[5m])))", @@ -1595,65 +1577,27 @@ fn histogram_quantile_classic_bucket_vs_native() { "histogram_quantile(0.9, my_native_histogram)", "histogram_quantile(0.9, request_duration_seconds)", // raw samples (your extension) ] { - let qe = ok(native); - assert!( - has( - &qe, - |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) - ), - "native/raw form → generic Quantile: {native}" - ); - assert!( - !has(&qe, |i| matches!(i, AggIntent::HistogramQuantile { .. })), - "{native}" - ); + rejected(native); } } #[test] -fn histogram_accessors_lower_to_per_series_intents() { - // `histogram_(v)` extracts a float per series from a native - // histogram — a per-series `Aggregate{[accessor]}` directly over the - // (instant) argument, no grouping. (`histogram_quantile` has its own two - // lowerings — see `histogram_quantile_classic_bucket_vs_native`.) - for (q, want) in [ - ("histogram_count(v)", AggIntent::HistogramCount), - ("histogram_sum(v)", AggIntent::HistogramSum), - ("histogram_avg(v)", AggIntent::HistogramAvg), - ("histogram_stddev(v)", AggIntent::HistogramStdDev), - ("histogram_stdvar(v)", AggIntent::HistogramStdVar), +fn native_histogram_accessors_are_explicit_gaps() { + // Native histogram samples have no typed representation yet. + for q in [ + "histogram_count(v)", + "histogram_sum(v)", + "histogram_avg(v)", + "histogram_stddev(v)", + "histogram_stdvar(v)", ] { - let qe = ok(q); - let QueryExpr::Aggregate { - reduction, - measures, - .. - } = &qe - else { - panic!("{q}: expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); - assert_eq!( - measures.as_slice(), - std::slice::from_ref(&want), - "{q}: wrong intent" - ); + rejected(q); } } #[test] -fn histogram_fraction_carries_its_bounds() { - // `histogram_fraction(lower, upper, v)` — bounds from args 0/1, vector arg 2. - let qe = ok("histogram_fraction(0, 0.2, v)"); - assert!(intents(&qe).iter().any(|i| matches!( - i, - AggIntent::HistogramFraction { lower, upper } - if *lower == 0.0 && (*upper - 0.2).abs() < 1e-9 - ))); +fn histogram_fraction_is_an_explicit_gap() { + rejected("histogram_fraction(0, 0.2, v)"); } // ───────────────────────────────────────────────────────────────────────────── @@ -1661,68 +1605,42 @@ fn histogram_fraction_carries_its_bounds() { // ───────────────────────────────────────────────────────────────────────────── #[test] -fn math_functions_lower_to_per_series_math_intents() { - // Each `f(v)` is a per-series element-wise value transform — a per-series - // `Aggregate{[Math(f)]}` over the (instant) argument, no grouping. - for (q, want) in [ - ("abs(v)", MathFunc::Abs), - ("ceil(v)", MathFunc::Ceil), - ("floor(v)", MathFunc::Floor), - ("sqrt(v)", MathFunc::Sqrt), - ("ln(v)", MathFunc::Ln), - ("log2(v)", MathFunc::Log2), - ("sgn(v)", MathFunc::Sgn), - ("sin(v)", MathFunc::Sin), - ("atanh(v)", MathFunc::Atanh), - ("deg(v)", MathFunc::Deg), - ("rad(v)", MathFunc::Rad), +fn math_functions_lower_to_typed_scalar_projections() { + for name in [ + "abs", "ceil", "floor", "sqrt", "ln", "log2", "sgn", "sin", "atanh", "deg", "rad", ] { - let qe = ok(q); - let QueryExpr::Aggregate { - reduction, - measures, - .. - } = &qe - else { - panic!("{q}: expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); + let query = ok(&format!("{name}(v)")); assert!( - matches!(measures.as_slice(), [AggIntent::Math(m)] if *m == want), - "{q}: wrong intent, got {measures:?}" + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) ); + query.validate_structure().unwrap(); } } #[test] fn clamp_and_round_carry_their_params() { - assert!(intents(&ok("clamp(v, 0, 100)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Clamp { min, max }) if *min == 0.0 && *max == 100.0) - )); - assert!(intents(&ok("clamp_min(v, 1)")) - .iter() - .any(|i| matches!(i, AggIntent::Math(MathFunc::ClampMin { min }) if *min == 1.0))); - assert!(intents(&ok("clamp_max(v, 5)")) - .iter() - .any(|i| matches!(i, AggIntent::Math(MathFunc::ClampMax { max }) if *max == 5.0))); - // `round(v)` defaults the step to 1; `round(v, 5)` reads it. - assert!(intents(&ok("round(v)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Round { to_nearest }) if *to_nearest == 1.0) - )); - assert!(intents(&ok("round(v, 5)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Round { to_nearest }) if *to_nearest == 5.0) - )); + for (query, params) in [ + ("clamp(v,0,100)", vec![0.0, 100.0]), + ("clamp_min(v,1)", vec![1.0]), + ("clamp_max(v,5)", vec![5.0]), + ("round(v)", vec![1.0]), + ("round(v,5)", vec![5.0]), + ] { + let node = ok(query); + let ScalarExpr::FunctionCall { args, .. } = support::sample_expression(&node) else { + panic!() + }; + assert_eq!( + args.iter().skip(1).map(promql_scalar).collect::>(), + params.into_iter().map(Some).collect::>() + ); + } } #[test] fn pi_lowers_to_a_scalar_constant() { - // `pi()` is the constant π — a `PromqlScalarBridge` leaf, not a `Math` intent. - assert!(ok("pi()") - .as_promql_scalar() + // `pi()` is the constant π — a `ScalarExpr` leaf, not a `Math` intent. + assert!(promql_scalar(&support::scalar_root("pi()")) .is_some_and(|v| (v - std::f64::consts::PI).abs() < 1e-12)); } @@ -1747,7 +1665,7 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { // `absent(v)` synthesizes its output labels from `v`'s equality matchers, so // those labels must survive into the schema — here `job` from `{job="x"}`. let qe = ok(r#"absent(up{job="x"})"#); - let cols = qe.output_schema().unwrap(); + let cols = qe.schema.clone(); assert!( cols.fields.iter().any(|c| c.name == "job"), "matcher label `job` kept, got {:?}", @@ -1761,73 +1679,55 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { #[test] fn time_lowers_to_the_eval_time_scalar() { - // SEMANTICS: `time()` is the query evaluation timestamp as a scalar — a leaf, - // not an aggregate over any series. - assert!(matches!(ok("time()"), QueryExpr::EvalTimestamp)); - // …and it is scalar-shaped: a single float `value`, no time index. - let sch = ok("time()").output_schema().unwrap(); - assert_eq!(sch.fields.len(), 1); - assert_eq!(sch.fields[0].name, "value"); - assert!(sch.time_index.is_none()); + assert!(matches!( + support::scalar_root("time()"), + ScalarExpr::EvalTimestamp + )); } #[test] fn time_minus_vector_is_the_uptime_pattern() { - // `time() - process_start_time_seconds` — the canonical uptime expression. - // The scalar `time()` broadcasts against the vector; the result takes the - // vector's schema. let qe = ok("time() - process_start_time_seconds"); - let QueryExpr::BinaryOp { lhs, op, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); - }; - assert!(matches!(lhs.as_ref(), QueryExpr::EvalTimestamp)); - assert!(matches!( - op, - BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub) - )); - assert!(qe.output_schema().is_ok()); + assert!( + matches!(support::sample_expression(&qe), ScalarExpr::Arithmetic { op: ArithmeticOpKind::Sub, left, .. } if matches!(left.as_ref(), ScalarExpr::EvalTimestamp)) + ); + assert!(qe.schema.time_index.is_some()); } #[test] fn calendar_functions_lower_to_time_fn_intents() { - // SEMANTICS: each of these is a per-series float transform of its argument's - // timestamp (or, for `timestamp`, the sample's own time). functions.test. - for (q, want) in [ - ("timestamp(up)", TimeFunc::Timestamp), - ("minute(v)", TimeFunc::Minute), - ("hour(v)", TimeFunc::Hour), - ("day_of_week(v)", TimeFunc::DayOfWeek), - ("day_of_month(v)", TimeFunc::DayOfMonth), - ("day_of_year(v)", TimeFunc::DayOfYear), - ("month(v)", TimeFunc::Month), - ("year(v)", TimeFunc::Year), - ("days_in_month(v)", TimeFunc::DaysInMonth), + assert!(has(&ok("timestamp(up)"), |i| *i + == AggIntent::TimeFn(TimeFunc::Timestamp))); + for name in [ + "minute", + "hour", + "day_of_week", + "day_of_month", + "day_of_year", + "month", + "year", + "days_in_month", ] { - let qe = ok(q); + let query = ok(&format!("{name}(v)")); assert!( - has(&qe, |i| *i == AggIntent::TimeFn(want)), - "{q} → TimeFn({want:?}), got {:?}", - intents(&qe) + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) ); } } #[test] fn no_arg_calendar_function_reads_the_eval_time() { - // `day_of_week()` with no argument computes over the evaluation time itself, - // so it is a `TimeFn` aggregate whose child is the `EvalTimestamp` scalar. - let qe = ok("day_of_week()"); - let QueryExpr::Aggregate { - measures, child, .. - } = &qe - else { - panic!("expected an Aggregate, got {qe:?}"); + let query = ok("day_of_week()"); + let NonASAPOp::Project { child, .. } = query.expect_non_asap() else { + panic!() }; assert!(matches!( - measures.as_slice(), - [AggIntent::TimeFn(TimeFunc::DayOfWeek)] + child.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(ScalarExpr::EvalTimestamp) )); - assert!(matches!(child.as_ref(), QueryExpr::EvalTimestamp)); + assert!( + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name,.. } if name=="promql_day_of_week") + ); } #[test] @@ -1848,30 +1748,23 @@ fn vector_promotes_a_scalar_to_a_vector() { // SEMANTICS: `vector(s)` is the scalar→instant-vector bridge — a label-less // single series carrying the scalar's value. let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; - assert_eq!(inner.as_promql_scalar(), Some(1.0)); + assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); // Vector-typed: schema has a time index (a scalar leaf has none). - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.time_index.is_some()); assert!(sch.fields.iter().any(|c| c.name == "value")); } #[test] fn scalar_collapses_a_vector_to_a_scalar() { - // SEMANTICS: `scalar(v)` is the instant-vector→scalar bridge. - let qe = ok("scalar(node_load1)"); - let QueryExpr::PromqlScalarFromVector(inner) = &qe else { - panic!("expected PromqlScalarFromVector, got {qe:?}"); + let qe = support::scalar_root("scalar(node_load1)"); + let ScalarExpr::PromqlScalarFromVector(inner) = &qe else { + panic!() }; - let (metric, _) = first_scan(inner); - assert_eq!(metric, "node_load1"); - // PromqlScalarBridge-typed: single `value` column, no time index. - let sch = qe.output_schema().unwrap(); - assert!(sch.time_index.is_none()); - assert_eq!(sch.fields.len(), 1); - assert_eq!(sch.fields[0].name, "value"); + assert_eq!(first_scan(inner).0, "node_load1"); } #[test] @@ -1880,26 +1773,32 @@ fn vector_zero_is_a_vector_operand_of_a_set_op() { // vectors, so `vector(0)` must be a vector (a `PromqlVectorFromScalar`), never a // folded scalar operand. let qe = ok("up or vector(0)"); - let QueryExpr::BinaryOp { rhs, op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected a BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)); - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlVectorFromScalar(_))); + assert!(matches!( + rhs.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(_) + )); } #[test] fn scalar_of_a_vector_feeds_a_threshold_comparison() { - // `node_load1 > scalar(node_cpu_count)` — `scalar(...)` is a scalar operand, - // so the BinaryOp output takes the vector (lhs) side's schema. let qe = ok("node_load1 > scalar(node_cpu_count)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Compare { right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlScalarFromVector(_))); - // The BinaryOp output schema follows the vector (lhs) side, not the scalar. - let (metric, _) = first_scan(lhs); - assert_eq!(metric, "node_load1"); - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(matches!( + right.as_ref(), + ScalarExpr::PromqlScalarFromVector(_) + )); + assert!(qe.schema.time_index.is_some()); } #[test] @@ -1909,13 +1808,13 @@ fn info_lowers_to_a_label_enrichment_join() { // (issue #84). The value/time axis pass through; the enriched labels are // runtime, so the schema stays the child's. let qe = ok("info(rate(http_requests_total[5m]))"); - let QueryExpr::PromqlInfoEnrich { selector, child } = &qe else { + let NonASAPOp::PromqlInfoEnrich { selector, child } = qe.expect_non_asap() else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert!(selector.is_empty(), "no selector → default target_info"); // The child is the untouched input (a per-series rate reduction here). assert!(has(child, |i| *i == AggIntent::Rate)); - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(qe.schema.clone().time_index.is_some()); } #[test] @@ -1925,7 +1824,7 @@ fn info_selector_carries_the_info_side_matchers() { // matchers are kept symbolically (not run through the single-metric selector // path). let qe = ok(r#"info(build_info, {__name__=~".+_info", another_data=~".+"})"#); - let QueryExpr::PromqlInfoEnrich { selector, .. } = &qe else { + let NonASAPOp::PromqlInfoEnrich { selector, .. } = qe.expect_non_asap() else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert_eq!( @@ -1949,12 +1848,12 @@ fn info_composes_under_an_aggregation_and_over_a_time_shift() { // `offset` / `@` on the input now lower to a `TimeShift` under the info-join // (issue #40) — the enrichment composes over the shifted selector. assert!(matches!( - ok("info(metric @ 60)"), - QueryExpr::PromqlInfoEnrich { .. } + ok("info(metric @ 60)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } )); assert!(matches!( - ok("info(metric offset 1m)"), - QueryExpr::PromqlInfoEnrich { .. } + ok("info(metric offset 1m)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } )); } @@ -1967,12 +1866,12 @@ fn group_lowers_to_a_constant_group_intent() { // SEMANTICS: `group(v)` yields a constant 1 per group — a distinct intent, // NOT folded onto `sum` (which would return the value sum instead of 1). let qe = ok("group(up)"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Group])); // Output column is the constant-1 `group` value. - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "group")); } @@ -1980,7 +1879,7 @@ fn group_lowers_to_a_constant_group_intent() { fn group_by_keeps_the_grouping_keys() { // `group by (job) (up)` — the grouping keys ride on `Aggregate.by`. let qe = ok("group by (job) (up)"); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "job")); assert!(has(&qe, |i| *i == AggIntent::Group)); } @@ -1991,13 +1890,13 @@ fn count_values_groups_by_value_and_synthesizes_a_label() { // value, counts each distinct value, and emits that value as a new label // `l`. The intent carries the label; schema gains a `Utf8` `l` column. let qe = ok(r#"count_values("version", build_version)"#); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::CountValues { label }] if label == "version") ); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); let version = sch .fields .iter() @@ -2023,7 +1922,7 @@ fn count_values_accepts_a_parenthesised_label_and_by_grouping() { &qe, |i| matches!(i, AggIntent::CountValues { label } if label == "v") )); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "job")); assert!(sch.fields.iter().any(|c| c.name == "v")); } @@ -2034,7 +1933,7 @@ fn count_values_label_colliding_with_a_group_key_is_not_duplicated() { // with a group-by key. PromQL's synthesized label takes precedence; the // output must carry a single `job` column, never two. let qe = ok(r#"count_values by (job) ("job", version)"#); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); let jobs = sch.fields.iter().filter(|c| c.name == "job").count(); assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.fields); assert!(sch.fields.iter().any(|c| c.name == "count")); @@ -2046,18 +1945,18 @@ fn limitk_and_limit_ratio_lower_to_series_sampling() { // series kept unchanged (NOT a ranking), so they lower to the dedicated // `PromqlSeriesSample` node, never `topk`'s `Sort → Limit` (issue #86). assert!(matches!( - ok("limitk(2, http_requests)"), - QueryExpr::PromqlSeriesSample { + ok("limitk(2, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitK(2), .. } )); assert!(matches!( - ok("limit_ratio(0.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 + ok("limit_ratio(0.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 )); // Series-preserving: the output schema equals the input's (ts, value). - let sch = ok("limitk(2, http_requests)").output_schema().unwrap(); + let sch = ok("limitk(2, http_requests)").schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some()); } @@ -2067,12 +1966,12 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { // A negative ratio selects the complementary fraction — it must survive, not // be normalised away. Out-of-range magnitudes clamp to [-1, 1] (Prometheus). assert!(matches!( - ok("limit_ratio(-0.5, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 + ok("limit_ratio(-0.5, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 )); assert!(matches!( - ok("limit_ratio(1.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 + ok("limit_ratio(1.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 )); } @@ -2080,7 +1979,7 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { fn limitk_by_carries_the_grouping_and_composes_in_a_set_op() { // `limitk by (group)` samples per group; the grouping label is seeded. let qe = ok("limitk by (group) (2, http_requests)"); - let QueryExpr::PromqlSeriesSample { by, .. } = &qe else { + let NonASAPOp::PromqlSeriesSample { by, .. } = qe.expect_non_asap() else { panic!("expected a PromqlSeriesSample, got {qe:?}"); }; assert!(!by.is_empty(), "grouped sampling keeps its `by` keys"); @@ -2106,20 +2005,20 @@ fn dynamic_and_non_finite_sample_params_are_rejected() { // ───────────────────────────────────────────────────────────────────────────── /// Descend single-child nodes to the first `PromqlRelabel`. -fn first_relabel(e: &QueryExpr) -> &QueryExpr { - match e { - QueryExpr::PromqlRelabel { .. } => e, - QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } => first_relabel(child), +fn first_relabel(e: &OperatorNode) -> &OperatorNode { + match e.expect_non_asap() { + NonASAPOp::PromqlRelabel { .. } => e, + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } => first_relabel(child), other => panic!("no PromqlRelabel reachable from {other:?}"), } } /// True when `value` is a `FunctionCall` with the given name. -fn is_fn_named(value: &QueryExpr, name: &str) -> bool { - matches!(value, QueryExpr::FunctionCall { name: n, .. } if n == name) +fn is_fn_named(value: &ScalarExpr, name: &str) -> bool { + matches!(value, ScalarExpr::FunctionCall { name: n, .. } if n == name) } #[test] @@ -2127,7 +2026,7 @@ fn label_replace_is_a_relabel_over_the_vector() { // SEMANTICS: `label_replace(v, dst, repl, src, regex)` rewrites the `dst` // label per series from a regex over `src`; the sample value is untouched. let qe = ok(r#"label_replace(up, "host", "$1", "instance", "(.+):.*")"#); - let QueryExpr::PromqlRelabel { dst, value, child } = &qe else { + let NonASAPOp::PromqlRelabel { dst, value, child } = qe.expect_non_asap() else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "host"); @@ -2137,7 +2036,7 @@ fn label_replace_is_a_relabel_over_the_vector() { // The value expression is a `label_replace` fn reading the `src` label. assert!(is_fn_named(value, "label_replace")); // Output: the child's columns + the synthesized `host` label; value & ts kept. - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "host")); assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some(), "the vector's time axis survives"); @@ -2148,12 +2047,12 @@ fn label_join_concatenates_source_labels() { // SEMANTICS: `label_join(v, dst, sep, src…)` joins the source labels with // `sep` into `dst`. let qe = ok(r#"label_join(up, "combined", "-", "job", "instance")"#); - let QueryExpr::PromqlRelabel { dst, value, .. } = &qe else { + let NonASAPOp::PromqlRelabel { dst, value, .. } = qe.expect_non_asap() else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "combined"); assert!(is_fn_named(value, "label_join")); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "combined")); } @@ -2164,9 +2063,11 @@ fn label_replace_composes_under_an_aggregation() { let qe = ok(r#"sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))"#); // A PromqlRelabel sits below the outer Sum. let relabel = first_relabel(&qe); - assert!(matches!(relabel, QueryExpr::PromqlRelabel { dst, .. } if dst == "host")); + assert!( + matches!(relabel.expect_non_asap(), NonASAPOp::PromqlRelabel { dst, .. } if dst == "host") + ); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "host")); } @@ -2191,7 +2092,7 @@ fn extra_over_time_reducers_lower_to_per_series_intents() { assert!(has(&qe, |i| *i == want), "{q}: {:?}", intents(&qe)); // Per-series: the range window survives as a `TimeRange`. assert!( - matches!(&qe, QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeRange { .. })), + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. })), "{q} keeps its range as a TimeRange" ); } @@ -2215,13 +2116,13 @@ fn sort_and_sort_desc_reorder_by_value_without_a_limit() { ("sort_desc(http_requests)", false), ] { let qe = ok(q); - let QueryExpr::Sort { keys, child, .. } = &qe else { + let NonASAPOp::Sort { keys, child, .. } = qe.expect_non_asap() else { panic!("{q}: expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 1); assert_eq!(keys[0].ascending, ascending, "{q}"); // No Limit above the Sort — every series is preserved. - assert!(!matches!(&qe, QueryExpr::Limit { .. })); + assert!(!matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); // The value column is what it ranks on: descend to the scan. let (metric, _) = first_scan(child); assert_eq!(metric, "http_requests"); @@ -2233,12 +2134,12 @@ fn sort_by_label_orders_on_each_label_in_turn() { // `sort_by_label(v, "group", "instance", "job")` — one ascending sort key per // label, in argument order; the labels are seeded into the schema. let qe = ok(r#"sort_by_label(http_requests, "group", "instance", "job")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { panic!("expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 3, "one key per label"); assert!(keys.iter().all(|k| k.ascending)); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); for label in ["group", "instance", "job"] { assert!(sch.fields.iter().any(|c| c.name == label), "{label} seeded"); } @@ -2247,7 +2148,7 @@ fn sort_by_label_orders_on_each_label_in_turn() { #[test] fn sort_by_label_desc_is_descending() { let qe = ok(r#"sort_by_label_desc(http_requests, "instance")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { panic!("expected a Sort, got {qe:?}"); }; assert!(keys.iter().all(|k| !k.ascending)); @@ -2256,28 +2157,43 @@ fn sort_by_label_desc_is_descending() { #[test] fn min_of_max_of_fold_constant_scalars() { // `min_of`/`max_of` are n-ary scalar reducers. When every argument is a - // constant they constant-fold to a `PromqlScalarBridge` leaf, just like scalar + // constant they constant-fold to a `ScalarExpr` leaf, just like scalar // arithmetic (#35) — the only form the intent algebra can hold (#89). - assert_eq!(ok("min_of(3, 5)").as_promql_scalar(), Some(3.0)); - assert_eq!(ok("max_of(3, 5)").as_promql_scalar(), Some(5.0)); - assert_eq!(ok("min_of(-2, -5)").as_promql_scalar(), Some(-5.0)); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(3, 5)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, 5)")), + Some(5.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(-2, -5)")), + Some(-5.0) + ); // Nested folds and use as a threshold operand. assert_eq!( - ok("max_of(min_of(2, 3), 10)").as_promql_scalar(), + promql_scalar(&support::scalar_root("max_of(min_of(2, 3), 10)")), Some(10.0) ); let qe = ok("up > max_of(1, 2)"); - let QueryExpr::BinaryOp { rhs, .. } = &qe else { + let ScalarExpr::Compare { right: rhs, .. } = support::sample_expression(&qe) else { panic!("{qe:?}") }; - assert_eq!(rhs.as_promql_scalar(), Some(2.0)); + assert_eq!(promql_scalar(rhs), Some(2.0)); } #[test] fn min_of_max_of_ignore_nan_like_the_min_max_aggregators() { // A NaN argument is skipped (Prometheus `min`/`max` NaN semantics). - assert_eq!(ok("max_of(3, NaN)").as_promql_scalar(), Some(3.0)); - assert_eq!(ok("min_of(NaN, 3)").as_promql_scalar(), Some(3.0)); + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, NaN)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(NaN, 3)")), + Some(3.0) + ); } #[test] diff --git a/crates/frontend-promql/tests/promql_equivalence.rs b/crates/frontend-promql/tests/promql_equivalence.rs index 9d0cab2ad..ac3c5a06f 100644 --- a/crates/frontend-promql/tests/promql_equivalence.rs +++ b/crates/frontend-promql/tests/promql_equivalence.rs @@ -17,12 +17,14 @@ #![allow(non_snake_case)] +use std::rc::Rc; + mod support; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; -fn lo(q: &str) -> QueryExpr { +fn lo(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("{q:?} should lower: {e}")) } diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 243bf6d18..540f426f5 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -1,10 +1,13 @@ //! End-to-end tests for PromQL → unresolved → canonical DAG lowering. +use std::rc::Rc; use std::time::Duration; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, +}; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, QueryExpr, Reduction, ScalarValue, - Source, + AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, Reduction, ScalarValue, Source, }; use asap_types::types::AccuracyTarget; use asap_types::workload::{ @@ -16,7 +19,7 @@ use asap_frontend_promql::{lower_promql_workload, PromqlError as LoweringError}; mod support; use support::lower_promql; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } @@ -77,12 +80,12 @@ fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { #[test] fn bare_selector_is_scan_with_predicates() { let qe = lower(r#"http_requests_total{env="prod",status!="500"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let NonASAPOp::Scan { source, predicates, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Scan, got {qe:?}"); }; @@ -92,29 +95,32 @@ fn bare_selector_is_scan_with_predicates() { assert_eq!(predicates.len(), 2); assert!(predicates .iter() - .all(|p| matches!(p.0.as_ref(), QueryExpr::Compare { .. }))); + .all(|p| matches!(&p.0, ScalarExpr::Compare { .. }))); } #[test] fn regex_matcher_lowers_to_regex_compareop() { let qe = lower(r#"http_requests_total{path=~"/api/.*"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let NonASAPOp::Scan { predicates, schema, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Scan, got {qe:?}"); }; - let QueryExpr::Compare { left, op, right } = predicates[0].0.as_ref() else { + let ScalarExpr::Compare { + left, op, right, .. + } = &predicates[0].0 + else { panic!("expected Compare, got {:?}", predicates[0].0); }; assert_eq!(*op, CompareOpKind::Regex); // The label matcher's column is resolved positionally against the scan schema. let path_id = schema.column_id("path").expect("path in scan schema"); - assert!(matches!(left.as_ref(), QueryExpr::Column(id) if *id == path_id)); - assert!(matches!(right.as_ref(), QueryExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); + assert!(matches!(left.as_ref(), ScalarExpr::Column(id) if *id == path_id)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); } // ── *_over_time → Aggregate over TimeRange ────────────────────────────────────── @@ -122,12 +128,12 @@ fn regex_matcher_lowers_to_regex_compareop() { #[test] fn quantile_over_time_is_time_range_aggregate() { let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; @@ -135,12 +141,14 @@ fn quantile_over_time_is_time_range_aggregate() { assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.99).abs() < 1e-9) ); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); // The label matcher folded onto the Scan. - assert!(matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1)); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) + ); } #[test] @@ -151,39 +159,42 @@ fn outer_sum_by_over_quantile_over_time_groups_positionally() { // a name-based Partition. Leaf = [ts, value, host, service] (referenced // names appended sorted) → host = col 2. let qe = lower(r#"sum by (host) (quantile_over_time(0.99, latency{service="web"}[5m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by host, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner: Aggregate{Quantile} over TimeRange (per-series over_time reduction). - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (quantile_over_time) under the outer Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Quantile { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] fn avg_over_time_maps_to_avg_intent() { let qe = lower("avg_over_time(cpu_seconds_total[10m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(600)); @@ -192,9 +203,9 @@ fn avg_over_time_maps_to_avg_intent() { #[test] fn stddev_and_stdvar_over_time() { let qe = lower("stddev_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; @@ -205,12 +216,15 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); let qe = lower("stdvar_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; @@ -221,7 +235,10 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -230,32 +247,33 @@ fn histogram_quantile_wraps_inner_in_quantile() { // not squashed away. The `_bucket` metric + `le` matcher mark the classic // form → `HistogramQuantile` over `Aggregate{Rate}` over Scan. let qe = lower(r#"histogram_quantile(0.95, rate(http_duration_seconds_bucket{le="0.5"}[5m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.95).abs() < 1e-9) ); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { + let NonASAPOp::TimeRange { range, child: tr_child, - } = child.as_ref() + .. + } = child.expect_non_asap() else { panic!("expected TimeRange under Rate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(tr_child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1) + matches!(tr_child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) ); } @@ -266,9 +284,9 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { // `sum by (le)` aggregate; now the `le` grouping survives into the // canonical DAG. let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; @@ -278,11 +296,11 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { ); // `sum by (le)` survives as a positional Aggregate (by = [2], `le`) over the // inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected `sum by (le)` as a positional Aggregate, got {child:?}"); }; @@ -292,12 +310,12 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { /// The classic `histogram_quantile` aggregate: its `without` keys, `le` /// column, and output column names. -fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { - let QueryExpr::Aggregate { +fn classic_histogram(qe: &OperatorNode) -> (Vec, usize, Vec) { + let NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, .. - } = qe + } = qe.expect_non_asap() else { panic!("expected a reducing Aggregate, got {qe:?}"); }; @@ -305,13 +323,7 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { panic!("expected HistogramQuantile, got {measures:?}"); }; assert!(by.is_without(), "histogram_quantile groups without (le)"); - let names = qe - .output_schema() - .unwrap() - .fields - .iter() - .map(|c| c.name.clone()) - .collect(); + let names = qe.schema.fields.iter().map(|c| c.name.clone()).collect(); (by.keys().to_vec(), *le, names) } @@ -321,10 +333,10 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { fn classic_histogram_quantile_groups_without_le() { let qe = lower("histogram_quantile(0.9, rate(http_duration_seconds_bucket[5m]))"); let (keys, le, names) = classic_histogram(&qe); - let QueryExpr::Aggregate { child, .. } = &qe else { + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { unreachable!() }; - let child = child.output_schema().unwrap(); + let child = &child.schema; assert_eq!(child.fields[le].name, "le"); assert_eq!(keys, vec![le]); assert_eq!(names, vec!["histogram_quantile"]); @@ -349,14 +361,16 @@ fn classic_histogram_quantile_keeps_out_of_range_quantiles() { ("histogram_quantile(-1, x_bucket)", -1.), ("histogram_quantile(2, x_bucket)", 2.), ] { - let QueryExpr::Aggregate { measures, .. } = lower(query) else { + let root = lower(query); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { panic!("{query}"); }; assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if *q == expected) ); } - let QueryExpr::Aggregate { measures, .. } = lower("histogram_quantile(NaN, x_bucket)") else { + let root = lower("histogram_quantile(NaN, x_bucket)"); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { panic!("NaN"); }; assert!(matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if q.is_nan())); @@ -379,14 +393,14 @@ fn classic_histogram_quantile_rejects_an_argument_without_le() { #[test] fn rate_has_time_range_child_not_window() { let qe = lower("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate for rate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child (not Window), got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -395,14 +409,14 @@ fn rate_has_time_range_child_not_window() { #[test] fn increase_maps_to_increase_intent() { let qe = lower("increase(errors_total[1h])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate for increase, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -415,21 +429,24 @@ fn sum_over_rate_keeps_both_levels() { // Regression: `sum(rate(m[w]))` — the most common PromQL shape — must keep // the cross-series Sum, not collapse to a bare per-series Rate. let qe = lower("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -438,20 +455,20 @@ fn sum_by_over_rate_groups_the_outer_sum() { // on a positional `Aggregate.by` (the same shape SQL produces) over the // label-preserving inner Rate. Leaf = [ts, value, job] → by = [2]. let qe = lower("sum by (job) (rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by job, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -459,16 +476,16 @@ fn sum_by_over_rate_groups_the_outer_sum() { fn count_over_rate_keeps_both_levels() { // The `Outer::Count` sibling of the `sum(rate(...))` bug. let qe = lower("count(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Count}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -487,12 +504,12 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ), ] { let dag = lower(query); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, reduction: actual, child, .. - } = &dag + } = dag.expect_non_asap() else { panic!("expected outer Count: {dag:?}"); }; @@ -501,12 +518,12 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { "{query}: {dag:?}" ); assert_eq!(actual, &reduction, "{query}"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, reduction, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner per-series Cardinality: {dag:?}"); }; @@ -516,7 +533,7 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ); assert_eq!(reduction, &Reduction::PerEntity, "{query}"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if range.as_secs() == 300) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, .. } if range.as_secs() == 300) ); } } @@ -552,14 +569,17 @@ fn count_never_lowers_to_distinct_sample_values() { #[test] fn count_over_time_is_count_intent() { let qe = lower("count_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -568,26 +588,29 @@ fn outer_count_counts_series() { // over the window (label-preserving), outer cross-series row count grouped // on a positional `Aggregate.by`. Leaf = [ts, value, symbol] → symbol = col 2. let qe = lower("count by (symbol) (count_over_time(financial_last_trade_price[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by symbol, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); // Inner: Aggregate{Count} over TimeRange (per-series count_over_time). - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (count_over_time) under the outer count, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ── topk / bottomk ──────────────────────────────────────────────────────────── @@ -596,12 +619,12 @@ fn outer_count_counts_series() { fn topk_over_count_is_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (10, count_over_time(requests{env="prod"}[1m]))"#); // Heavy-hitter: Aggregate{TopK} with grouping resolved to positional ids. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate with TopK, got {qe:?}"); }; @@ -612,29 +635,29 @@ fn topk_over_count_is_heavy_hitter_topk() { [AggIntent::TopK { k: 10, .. }] )); // The count_over_time under the TopK is a TimeRange-backed aggregate. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (count_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected TimeRange under Count aggregate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(60)); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } #[test] fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (5, sum_over_time(requests{env="prod"}[1m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate with TopK, got {qe:?}"); }; @@ -643,29 +666,38 @@ fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { measures.as_slice(), [AggIntent::TopK { k: 5, .. }] )); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (sum_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] fn topk_over_avg_is_generic_sort_limit() { let qe = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { n, offset, child } = &qe else { + let NonASAPOp::Limit { + n: Some(n), + offset, + child, + .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 5); assert_eq!(*offset, 0); - let QueryExpr::Sort { + let NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort under Limit, got {child:?}"); }; @@ -678,7 +710,7 @@ fn topk_over_avg_is_generic_sort_limit() { // Underneath: the label-preserving windowed avg aggregate (by: []), no // intervening Partition. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, measures, .. } + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, measures, .. } if reduction == &Reduction::PerEntity && matches!(measures.as_slice(), [AggIntent::Avg { .. }])), "expected bare per-series Avg aggregate under Sort, got {child:?}" ); @@ -687,7 +719,7 @@ fn topk_over_avg_is_generic_sort_limit() { #[test] fn ungrouped_topk_over_sum_is_heavy_hitter() { let qe = lower("topk(5, sum_over_time(m[5m]))"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has_intent(&qe, |i| matches!(i, AggIntent::Sum { .. }))); assert!(has_intent(&qe, |i| matches!( i, @@ -699,11 +731,14 @@ fn ungrouped_topk_over_sum_is_heavy_hitter() { fn bottomk_over_count_is_generic_sort_ascending() { // `bottomk` is never a heavy-hitter (descending=false), even over count. let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -715,11 +750,14 @@ fn bottomk_over_count_is_generic_sort_ascending() { #[test] fn bottomk_is_always_generic_sort_ascending() { let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -731,12 +769,12 @@ fn topk_count_output_schema_carries_group_key() { // (`service`) flows through to the outer TopK's `by` column. Leaf schema = // [ts, value, service] → TopK groups on service (col 2). let qe = lower("topk by (service) (5, count_over_time(m[1m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate{{TopK}}, got {qe:?}"); }; @@ -750,14 +788,17 @@ fn topk_count_output_schema_carries_group_key() { [AggIntent::TopK { k: 5, .. }] )); // Inner Count aggregate is visible with its TimeRange child. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Count}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ── binary ops ──────────────────────────────────────────────────────────────── @@ -765,22 +806,32 @@ fn topk_count_output_schema_carries_group_key() { #[test] fn binary_op_division() { let qe = lower("rate(a[5m]) / rate(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); } #[test] fn binary_op_with_on_grouping() { let qe = lower("a / on(host) b"); - let QueryExpr::BinaryOp { vector_match, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { vector_match, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; let vm = vector_match.as_ref().expect("vector_match present"); @@ -793,18 +844,38 @@ fn binary_op_with_on_grouping() { // carry it. #[test] fn bool_comparisons_are_distinct() { - let op = |q: &str| match lower(q) { - QueryExpr::BinaryOp { op, .. } => op, + let op = |q: &str| match lower(q).expect_non_asap() { + NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } => (operator.kind.clone(), *return_bool), + NonASAPOp::Filter { + pred: asap_types::ir::Predicate(ScalarExpr::Compare { op, .. }), + .. + } => (BinaryOpKind::Compare(op.clone()), false), + NonASAPOp::Project { cols, .. } => { + let ScalarExpr::Case { branches, .. } = &cols[1].expr else { + panic!() + }; + let ScalarExpr::Compare { op, .. } = &branches[0].0 else { + panic!() + }; + (BinaryOpKind::Compare(op.clone()), true) + } other => panic!("expected BinaryOp, got {other:?}"), }; - assert_eq!(op("a > 1"), BinaryOpKind::Compare(CompareOpKind::Gt)); + assert_eq!( + op("a > 1"), + (BinaryOpKind::Compare(CompareOpKind::Gt), false) + ); assert_eq!( op("a > bool 1"), - BinaryOpKind::CompareBool(CompareOpKind::Gt) + (BinaryOpKind::Compare(CompareOpKind::Gt), true) ); assert_eq!( op("a == bool on(job) b"), - BinaryOpKind::CompareBool(CompareOpKind::Eq) + (BinaryOpKind::Compare(CompareOpKind::Eq), true) ); } @@ -814,7 +885,7 @@ fn binary_op_binds_each_branch_against_its_own_schema() { // single root schema threaded to both branches, the left scan would leak the // right's group key (and vice-versa). Per-branch binding keeps them separate. let qe = lower("count by (job) (a) / count by (region) (b)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { lhs, rhs, .. } = qe.expect_non_asap() else { panic!("expected BinaryOp, got {qe:?}"); }; let lcols = scan_columns(lhs); @@ -829,26 +900,26 @@ fn binary_op_binds_each_branch_against_its_own_schema() { ); } -/// Collect every `AggIntent` in the DAG, root-to-leaf. -fn all_intents(e: &QueryExpr) -> Vec { +/// Collect every `AggIntent` in the dag, root-to-leaf. +fn all_intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); collect_intents(e, &mut out); out } -fn collect_intents(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { +fn collect_intents(e: &OperatorNode, out: &mut Vec) { + match e.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); collect_intents(child, out); } - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => collect_intents(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => collect_intents(child, out), + NonASAPOp::BinaryOp { lhs, rhs, .. } => { collect_intents(lhs, out); collect_intents(rhs, out); } @@ -856,20 +927,20 @@ fn collect_intents(e: &QueryExpr, out: &mut Vec) { } } -/// True if any `AggIntent` anywhere in the DAG satisfies `pred`. -fn has_intent bool>(e: &QueryExpr, pred: F) -> bool { +/// True if any `AggIntent` anywhere in the dag satisfies `pred`. +fn has_intent bool>(e: &OperatorNode, pred: F) -> bool { all_intents(e).iter().any(pred) } /// Field names on the first `Scan` reachable by descending single-child nodes. -fn scan_columns(e: &QueryExpr) -> Vec { - match e { - QueryExpr::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), - QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_columns(child), +fn scan_columns(e: &OperatorNode) -> Vec { + match e.expect_non_asap() { + NonASAPOp::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => scan_columns(child), _ => vec![], } } @@ -883,12 +954,12 @@ fn without_grouping_lowers_to_the_exclusion_form() { // label is stored positionally (the SchemaResolver seeds it), the grouping is the // `without` form, and the output schema stays open. let qe = lower("sum without (instance) (rate(m[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -899,10 +970,10 @@ fn without_grouping_lowers_to_the_exclusion_form() { // The inner per-series rate is preserved (label-preserving) under the outer // cross-series `without` reduction. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); - assert!(!qe.output_schema().unwrap().closed); + assert!(!qe.schema.clone().closed); } // ── parameter validation (reject rather than silently truncate/garble) ────────── @@ -920,7 +991,7 @@ fn out_of_range_quantile_phi_is_accepted() { for query in [ "quantile(1.5, up)", "quantile_over_time(1.5, m[5m])", - "histogram_quantile(2.0, rate(b[5m]))", + "histogram_quantile(2.0, rate(b_bucket[5m]))", ] { assert!( lower_promql(query, AccuracyTarget::Exact).is_ok(), @@ -979,7 +1050,7 @@ fn accuracy_target_flows_into_quantile_intent() { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; assert!(matches!( @@ -998,10 +1069,10 @@ fn aggregate_output_schema_preserves_time_axis_and_labels() { // predicate columns) to the scan schema, so `env` appears as a column // even though it is only used as a filter. // per_series_reduction_schema preserves the time axis and all label columns. - let QueryExpr::Aggregate { .. } = &qe else { + let NonASAPOp::Aggregate { .. } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; - let schema = qe.output_schema().expect("aggregate schema"); + let schema = &qe.schema; let names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); assert_eq!(names, vec!["ts", "value", "env"]); assert_eq!( @@ -1016,16 +1087,16 @@ fn scan_schema_carries_ts_value_and_group_keys() { // `service` is a group key → the SchemaResolver lands it in the self-contained // Scan schema (positional). `env` is only a filter, so it is not a column. let qe = lower("count by (service) (count_over_time(requests[1m]))"); - fn find_scan(n: &QueryExpr) -> &QueryExpr { - match n { - QueryExpr::Scan { .. } => n, - QueryExpr::TimeRange { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } => find_scan(child), + fn find_scan(n: &OperatorNode) -> &OperatorNode { + match n.expect_non_asap() { + NonASAPOp::Scan { .. } => n, + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } => find_scan(child), other => panic!("unexpected node {other:?}"), } } - let QueryExpr::Scan { schema, .. } = find_scan(&qe) else { + let NonASAPOp::Scan { schema, .. } = find_scan(&qe).expect_non_asap() else { unreachable!() }; let mut names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); @@ -1113,19 +1184,23 @@ fn reducing_group_by_lowers_to_aggregate_by() { // Cross-series reduce, no keys → bare `Aggregate { reduction: Reduce([]) }`. let q = lower("sum(http_requests_total)"); assert!( - matches!(q, QueryExpr::Aggregate { ref reduction, .. } if reduction == &Reduction::by(vec![])) + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::by(vec![])) ); // Cross-series reduce grouped by a label → `Aggregate.reduction`. let q = lower("sum by (job) (http_requests_total)"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } - if reduction.expect_reduce().len() == 1)); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); // Reduce over a label-preserving `rate` grouped by a label → still // `Aggregate.reduction` (the keys resolve against rate's preserved schema). let q = lower("sum by (job) (rate(http_requests_total[5m]))"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } - if reduction.expect_reduce().len() == 1)); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); } #[test] @@ -1134,20 +1209,20 @@ fn generic_topk_grouping_lowers_to_sort_partition_by() { // reducing → the grouping rides on `Sort.partition_by`, and the windowed // reduction beneath stays label-preserving (`by: []`). No `Partition` node. let q = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { child, .. } = &q else { + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let NonASAPOp::Sort { partition_by, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; assert_eq!(partition_by, &vec![2], "host is col 2 in [ts, value, host]"); assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) ); } @@ -1160,15 +1235,18 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // Partition→Sort.partition_by reframe in #12). Expected: // Limit{3} → Sort{value desc, partition_by:[job]} → Scan let q = lower("topk(3, http_requests_total) by (job)"); - let QueryExpr::Limit { n, child, .. } = &q else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = q.expect_non_asap() + else { panic!("expected Limit, got {q:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { + let NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; @@ -1177,7 +1255,7 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // No implicit reducing aggregate — the selector is label-preserving, so the // sort is directly over the selector horizon (the `job` label survives to partition by). assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })), "ranking is over the bare selector horizon, not a reducing Aggregate, got {child:?}" ); assert!( @@ -1191,20 +1269,20 @@ fn topk_over_bare_selector_ranks_raw_samples() { // Even without `by`, `topk(3, m)` ranks the raw instant-vector samples — it // does not sum them. The sort sits directly over the Scan, partition empty. let q = lower("topk(3, http_requests_total)"); - let QueryExpr::Limit { child, .. } = &q else { + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let NonASAPOp::Sort { partition_by, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; assert!(partition_by.is_empty(), "no `by` → global ranking"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); assert!(!has_intent(&q, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -1212,20 +1290,20 @@ fn topk_over_bare_selector_ranks_raw_samples() { // ── Issue #109: histogram_quantiles fans out into one branch per φ ────────── /// The `(label value, intent)` of each `histogram_quantiles` branch. -fn quantile_branches(q: &QueryExpr) -> Vec<(String, AggIntent)> { - let QueryExpr::Concat { children, .. } = q else { +fn quantile_branches(q: &OperatorNode) -> Vec<(String, AggIntent)> { + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected a Concat at the root, got {q:?}"); }; children .iter() .map(|c| { - let QueryExpr::PromqlRelabel { value, child, .. } = c else { + let NonASAPOp::PromqlRelabel { value, child, .. } = c.expect_non_asap() else { panic!("expected PromqlRelabel per branch, got {c:?}"); }; - let QueryExpr::Literal(ScalarValue::Utf8(v)) = value.as_ref() else { + let ScalarExpr::Literal(ScalarValue::Utf8(v)) = value else { panic!("expected a literal label value, got {value:?}"); }; - let QueryExpr::Aggregate { measures, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { measures, .. } = child.expect_non_asap() else { panic!("expected an Aggregate under the PromqlRelabel, got {child:?}"); }; (v.clone(), measures[0].clone()) @@ -1234,24 +1312,12 @@ fn quantile_branches(q: &QueryExpr) -> Vec<(String, AggIntent)> { } #[test] -fn histogram_quantiles_fans_out_over_native_histograms() { - // Raw / native-histogram argument → the sketch-able `Quantile` intent, - // exactly as the single-quantile `histogram_quantile` would choose. - let q = lower(r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#); - let branches = quantile_branches(&q); - assert_eq!(branches.len(), 3); - let labels: Vec<_> = branches.iter().map(|(l, _)| l.as_str()).collect(); - assert_eq!( - labels, - ["0.0", "0.25", "1.0"], - "OpenMetrics float formatting" - ); - for (_, intent) in &branches { - assert!( - matches!(intent, AggIntent::Quantile { .. }), - "native histogram → sketch-able Quantile, got {intent:?}" - ); - } +fn histogram_quantiles_rejects_unrepresented_native_histograms() { + assert!(lower_promql( + r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#, + AccuracyTarget::Exact + ) + .is_err()); } #[test] @@ -1270,15 +1336,15 @@ fn histogram_quantiles_over_classic_buckets_interpolates() { fn histogram_quantiles_branches_are_union_compatible() { // `Concat` derives its schema from the first child, so every branch must // agree on column names — the φ lives in the label, not the column name. - let q = lower(r#"histogram_quantiles(testhistogram3, "q", 0.5, 0.9)"#); - let QueryExpr::Concat { children, .. } = &q else { + let q = lower(r#"histogram_quantiles(testhistogram3_bucket, "q", 0.5, 0.9)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected Concat"); }; let shapes: Vec> = children .iter() .map(|c| { - c.output_schema() - .expect("branch schema") + c.schema + .clone() .fields .iter() .map(|c| c.name.clone()) @@ -1288,7 +1354,7 @@ fn histogram_quantiles_branches_are_union_compatible() { assert_eq!(shapes[0], shapes[1], "branches must be union-compatible"); assert_eq!(shapes[0], vec!["value".to_string(), "q".to_string()]); assert_eq!( - q.output_schema().expect("merged schema").fields.len(), + q.schema.fields.len(), 2, "the merged schema describes every branch" ); @@ -1296,11 +1362,11 @@ fn histogram_quantiles_branches_are_union_compatible() { #[test] fn histogram_quantiles_uses_the_given_label_name() { - let q = lower(r#"histogram_quantiles(h, "phi", 0.5)"#); - let QueryExpr::Concat { children, .. } = &q else { + let q = lower(r#"histogram_quantiles(h_bucket, "phi", 0.5)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected Concat"); }; - let QueryExpr::PromqlRelabel { dst, .. } = &children[0] else { + let NonASAPOp::PromqlRelabel { dst, .. } = children[0].expect_non_asap() else { panic!("expected PromqlRelabel"); }; assert_eq!(dst, "phi"); @@ -1309,7 +1375,7 @@ fn histogram_quantiles_uses_the_given_label_name() { #[test] fn histogram_quantiles_formats_small_quantiles_like_prometheus() { // `labels.FormatOpenMetricsFloat`: Go's %g, so exponent form below 1e-4. - let q = lower(r#"histogram_quantiles(h, "q", 0.00001)"#); + let q = lower(r#"histogram_quantiles(h_bucket, "q", 0.00001)"#); assert_eq!(quantile_branches(&q)[0].0, "1e-05"); } @@ -1317,9 +1383,9 @@ fn histogram_quantiles_formats_small_quantiles_like_prometheus() { fn histogram_quantiles_rejects_an_out_of_range_quantile() { // Same rule as `histogram_quantile(φ, …)` — one bad φ fails the whole call. for q in [ - r#"histogram_quantiles(h, "q", -0.1)"#, - r#"histogram_quantiles(h, "q", 1.01)"#, - r#"histogram_quantiles(h, "q", 0.5, NaN)"#, + r#"histogram_quantiles(h_bucket, "q", -0.1)"#, + r#"histogram_quantiles(h_bucket, "q", 1.01)"#, + r#"histogram_quantiles(h_bucket, "q", 0.5, NaN)"#, ] { assert!( lower_promql(q, AccuracyTarget::Exact).is_err(), @@ -1328,19 +1394,244 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { } } -// A subquery's `offset`/`@` shift the whole subquery, so the DAG keeps them. +// ── TimeRange.kind: instant vs range selectors ────────────────────────────────── + +#[test] +fn bare_instant_selector_is_an_instant_time_range() { + // `up` reads the latest sample per series within the workload's ingestion + // interval (1s in `support::workload`): an `Instant` lookback of that length. + let qe = lower("up"); + let NonASAPOp::TimeRange { range, kind, child } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + assert_eq!(*kind, TimeRangeKind::Instant); + assert_eq!(*range, Duration::from_secs(1)); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); +} + +#[test] +fn explicit_range_selector_is_a_range_time_range() { + // `m[5m]` keeps its own window and is a `Range` selection — both under a + // range function and as a bare matrix selector. + let qe = lower("rate(m[5m])"); + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { + panic!("expected Aggregate, got {qe:?}"); + }; + let NonASAPOp::TimeRange { range, kind, .. } = child.expect_non_asap() else { + panic!("expected TimeRange, got {child:?}"); + }; + assert_eq!(*kind, TimeRangeKind::Range); + assert_eq!(*range, Duration::from_secs(300)); + + let qe = lower("m[5m]"); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); +} + +#[test] +fn instant_and_range_selectors_of_equal_length_stay_distinct() { + // The kind is part of the shape: a 1s range selector is not the same dag as + // the 1s instant lookback injected around a bare selector. + assert_ne!(lower("up"), lower("up[1s]")); +} + +// ── the `bool` modifier → `return_bool` ───────────────────────────────────────── + +#[test] +fn vector_scalar_comparison_without_bool_filters() { + let qe = lower("up > 0"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. })); + assert!(matches!( + support::sample_expression(&qe), + ScalarExpr::Compare { + op: CompareOpKind::Gt, + .. + } + )); +} + +#[test] +fn vector_scalar_comparison_with_bool_sets_return_bool() { + let qe = lower("up > bool 0"); + assert!(matches!( + support::sample_expression(&qe), + ScalarExpr::Case { .. } + )); + assert_ne!(qe, lower("up > 0")); +} + +#[test] +fn vector_vector_comparison_with_bool_sets_return_bool() { + // `a > bool b` — the modifier lands on the vector/vector op itself, with + // the default (ignoring nothing) match. + let qe = lower("a > bool b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + lhs, + rhs, + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + assert_eq!(operator.kind, BinaryOpKind::Compare(CompareOpKind::Gt)); + assert!(matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(!lower("a > b").expect_non_asap().children().is_empty()); + assert_ne!(qe, lower("a > b")); +} + +#[test] +fn bool_modifier_composes_with_vector_matching() { + let qe = lower("a > bool on(job) b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + let vm = operator.vector_match.as_ref().expect("on(job) present"); + assert_eq!(vm.labels, vec!["job".to_string()]); +} + +// ── scalar expressions: negation, arithmetic, comparison ──────────────────────── + +#[test] +fn scalar_negation_of_time_is_a_negative_expression() { + // `-time()` is a scalar expression; its negation stays structural (the + // operand is not a constant to fold) and follows PromQL numeric rules. + let qe = support::scalar_root("-time()"); + let ScalarExpr::Negative { expr, semantics } = &qe else { + panic!("expected ScalarExpr(Negative), got {qe:?}"); + }; + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(expr.as_ref(), ScalarExpr::EvalTimestamp)); + // Scalar-shaped: no time index. +} + +#[test] +fn scalar_negation_of_a_constant_still_folds() { + // `-(2)` is constant: it folds to one literal rather than a `Negative`. + assert_eq!( + support::promql_scalar(&support::scalar_root("-(2)")), + Some(-2.0) + ); +} + +#[test] +fn scalar_arithmetic_carries_promql_semantics() { + let qe = support::scalar_root("time() - 1"); + let ScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } = &qe + else { + panic!("expected scalar(Arithmetic), got {qe:?}"); + }; + assert_eq!(*op, ArithmeticOpKind::Sub); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::EvalTimestamp)); + assert!(matches!( + right.as_ref(), + ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0 + )); +} + +#[test] +fn scalar_bool_comparison_is_a_zero_one_case_with_promql_semantics() { + // `1 < bool 2` → `Case(Compare(1 < 2) → 1.0, else 0.0)`: PromQL yields 0/1. + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { + operand, + branches, + else_expr, + } = &qe + else { + panic!("expected scalar(Case), got {qe:?}"); + }; + assert!(operand.is_none()); + let [(when, then)] = branches.as_slice() else { + panic!("expected one branch, got {branches:?}"); + }; + let ScalarExpr::Compare { + left, + op, + right, + semantics, + } = when + else { + panic!("expected a Compare condition, got {when:?}"); + }; + assert_eq!(*op, CompareOpKind::Lt); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 2.0)); + assert!(matches!(then, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!( + else_expr.as_deref(), + Some(ScalarExpr::Literal(ScalarValue::Float64(v))) if *v == 0.0 + )); +} + +#[test] +fn scalar_comparison_without_bool_is_rejected() { + // PromQL has no scalar filter: a scalar/scalar comparison needs `bool`. + for q in ["1 < 2", "time() > 0", "(1 + 1) == 2"] { + assert!( + lower_promql(q, AccuracyTarget::Exact).is_err(), + "{q} must be rejected without `bool`" + ); + } +} + +#[test] +fn label_matcher_predicates_carry_promql_semantics() { + let qe = lower(r#"up{job="api"}"#); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + let NonASAPOp::Scan { predicates, .. } = child.expect_non_asap() else { + panic!("expected Scan, got {child:?}"); + }; + assert!(matches!( + &predicates[0].0, + ScalarExpr::Compare { + semantics: ExprSemantics::Promql, + .. + } + )); +} + +// A subquery's `offset` / `@` modifier stays a `TimeShift` over the subquery. #[test] fn subquery_time_shift_is_retained() { - let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { - panic!("expected a range function"); + let root = lower("max_over_time(m[5m:1m] offset 1m)"); + let NonASAPOp::Aggregate { child, .. } = root.expect_non_asap() else { + panic!("expected a range function, got {root:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("subquery offset was dropped: {child:?}"); }; assert_eq!(shift.offset_ms, 60_000); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); assert!(matches!( - lower("max_over_time(m[5m:1m] @ 100)"), - QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeShift { .. }) + child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); + let root = lower("max_over_time(m[5m:1m] @ 100)"); + assert!(matches!( + root.expect_non_asap(), + NonASAPOp::Aggregate { child, .. } + if matches!(child.expect_non_asap(), NonASAPOp::TimeShift { .. }) )); } diff --git a/crates/frontend-promql/tests/unified_scalar_design.rs b/crates/frontend-promql/tests/scalar_design.rs similarity index 97% rename from crates/frontend-promql/tests/unified_scalar_design.rs rename to crates/frontend-promql/tests/scalar_design.rs index 4e3f3063e..e4f8038f8 100644 --- a/crates/frontend-promql/tests/unified_scalar_design.rs +++ b/crates/frontend-promql/tests/scalar_design.rs @@ -1,12 +1,11 @@ //! Scalar expressions never become constant-wrapper operators. -#[path = "unified_support.rs"] mod support; use asap_types::ir::{NonASAPOp, QueryRoot, ScalarExpr}; use asap_types::pre_asap::{ArithmeticOpKind, ScalarValue}; use asap_types::types::AccuracyTarget; fn root(query: &str) -> QueryRoot { - asap_frontend_promql::unified::lower_promql_query_workload( + asap_frontend_promql::lower_promql_query_workload( &support::workload(query, AccuracyTarget::Exact), 0, ) diff --git a/crates/frontend-promql/tests/support.rs b/crates/frontend-promql/tests/support.rs index 1f15b1ca2..1bd247d04 100644 --- a/crates/frontend-promql/tests/support.rs +++ b/crates/frontend-promql/tests/support.rs @@ -1,14 +1,17 @@ +use std::rc::Rc; + use asap_frontend_promql::{ lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, }; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; +use asap_types::pre_asap::ScalarValue; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; -fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { +pub fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -35,7 +38,11 @@ fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { } } -pub fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { +#[allow(dead_code)] +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, PromqlError> { let mut lowered = lower_promql_workload(&workload(query, accuracy), 0)?; Ok(lowered.remove(0)) } @@ -45,8 +52,60 @@ pub fn lower_promql_with_histograms( query: &str, accuracy: AccuracyTarget, histograms: HistogramCatalog, -) -> Result { +) -> Result, PromqlError> { let mut lowered = lower_promql_workload_with_histograms(&workload(query, accuracy), histograms, 0)?; Ok(lowered.remove(0)) } + +/// The value of a bare PromQL numeric literal / folded constant at an +/// scalar position (`Literal(Float64(v))`); `None` for any +/// other shape. +#[allow(dead_code)] +pub fn promql_scalar(node: &ScalarExpr) -> Option { + match node { + ScalarExpr::Literal(ScalarValue::Float64(v)) => Some(*v), + _ => None, + } +} + +/// Time `root` under the default (every summary maintained) lifecycle +/// assignment and export the post-ASAP DAG — the wire-6 export needs every +/// node timed first. +#[allow(dead_code)] +pub fn post_asap_dag(root: &Rc) -> asap_types::ir::physical_export::PhysicalASAPDAG { + use asap_types::ir::{apply_lifecycle_timings, LifecycleAssignment, TimingMemo}; + let timed = apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .expect("default lifecycle timings"); + asap_types::ir::physical_export::compile_physical_asap_dag(&timed) + .expect("post-ASAP DAG export") +} + +#[allow(dead_code)] +pub fn scalar_root(query: &str) -> ScalarExpr { + match asap_frontend_promql::lower_promql_query_workload( + &workload(query, AccuracyTarget::Exact), + 0, + ) + .unwrap() + .remove(0) + { + asap_types::ir::QueryRoot::Scalar(expr) => expr, + _ => panic!("expected scalar root: {query}"), + } +} + +#[allow(dead_code)] +pub fn sample_expression(node: &OperatorNode) -> &ScalarExpr { + match node.expect_non_asap() { + NonASAPOp::Project { cols, .. } => { + &cols[node.schema.column_id("value").unwrap_or(cols.len() - 1)].expr + } + NonASAPOp::Filter { pred, .. } => &pred.0, + other => panic!("expected sample expression, got {other:?}"), + } +} diff --git a/crates/frontend-promql/tests/unified_histogram_metadata.rs b/crates/frontend-promql/tests/unified_histogram_metadata.rs deleted file mode 100644 index 49fdffe02..000000000 --- a/crates/frontend-promql/tests/unified_histogram_metadata.rs +++ /dev/null @@ -1,138 +0,0 @@ -//! Type-driven `histogram_quantile` discrimination (issue #79). -//! -//! The structural heuristic (`by (le)` / `_bucket` / `le=` matcher) proxies the -//! argument's sample type. A declared [`HistogramKind`] overrides it, fixing the -//! heuristic's false-positive and false-negative cases. Undeclared metrics still -//! fall back to the heuristic. - -use asap_frontend_promql::unified::{HistogramCatalog, HistogramKind}; -#[path = "unified_support.rs"] -mod support; -use asap_types::ir::{NonASAPOp, OperatorNode}; -use asap_types::pre_asap::AggIntent; -use asap_types::types::AccuracyTarget; -use support::{lower_promql, lower_promql_with_histograms}; - -/// The histogram/quantile intent kind in the lowered tree: `"HQ"` for the -/// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. -fn quantile_kind(qe: &OperatorNode) -> &'static str { - fn walk(e: &OperatorNode) -> Option<&'static str> { - match e.expect_non_asap() { - NonASAPOp::Aggregate { - measures, child, .. - } => measures - .iter() - .find_map(|i| match i { - AggIntent::HistogramQuantile { .. } => Some("HQ"), - AggIntent::Quantile { .. } => Some("Q"), - _ => None, - }) - .or_else(|| walk(child)), - NonASAPOp::TimeRange { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } - | NonASAPOp::PromqlSubquery { child, .. } - | NonASAPOp::Project { child, .. } => walk(child), - _ => None, - } - } - walk(qe).expect("a HistogramQuantile or Quantile intent") -} - -fn heuristic(q: &str) -> &'static str { - quantile_kind(&lower_promql(q, AccuracyTarget::Exact).unwrap()) -} - -fn with_meta(q: &str, catalog: HistogramCatalog) -> &'static str { - quantile_kind(&lower_promql_with_histograms(q, AccuracyTarget::Exact, catalog).unwrap()) -} - -#[test] -fn heuristic_baseline_is_unchanged_without_a_catalog() { - // Classic buckets are represented; undeclared native samples are rejected. - assert_eq!( - heuristic( - "histogram_quantile(0.9, sum by (le) (rate(http_request_duration_seconds_bucket[5m])))" - ), - "HQ" - ); - assert!(lower_promql( - "histogram_quantile(0.9, native_latency)", - AccuracyTarget::Exact - ) - .is_err()); -} - -#[test] -fn declared_classic_bucket_fixes_the_false_negative() { - // A classic histogram exposed WITHOUT the `_bucket` suffix and queried with - // no `le` grouping/matcher requires an explicit sample-type declaration. - let q = "histogram_quantile(0.9, latency_seconds)"; - assert!(lower_promql(q, AccuracyTarget::Exact).is_err()); - assert_eq!( - with_meta( - q, - HistogramCatalog::new().with("latency_seconds", HistogramKind::ClassicBucket) - ), - "HQ", - "metadata routes it to exact bucket interpolation" - ); -} - -#[test] -fn declared_raw_extension_and_native_gap_override_the_heuristic() { - // A metric merely NAMED `…_bucket` that actually holds raw samples / a native - // histogram: the heuristic wrongly routes it to bucket interpolation. - let q = "histogram_quantile(0.9, foo_bucket)"; - assert_eq!( - heuristic(q), - "HQ", - "heuristic mis-routes on the `_bucket` name" - ); - assert_eq!( - with_meta( - q, - HistogramCatalog::new().with("foo_bucket", HistogramKind::RawSamples) - ), - "Q", - "raw samples are sketch-able" - ); - let catalog = HistogramCatalog::new().with("foo_bucket", HistogramKind::Native); - assert!(lower_promql_with_histograms(q, AccuracyTarget::Exact, catalog.clone()).is_err()); - assert!(lower_promql_with_histograms("foo_bucket", AccuracyTarget::Exact, catalog).is_err()); -} - -#[test] -fn undeclared_metric_falls_back_to_the_heuristic() { - // A catalog that doesn't mention the queried metric leaves the structural - // decision in place. - let catalog = HistogramCatalog::new().with("some_other_metric", HistogramKind::RawSamples); - assert_eq!( - with_meta( - "histogram_quantile(0.9, sum by (le) (x_bucket))", - catalog.clone() - ), - "HQ" - ); - assert!(lower_promql_with_histograms( - "histogram_quantile(0.9, native_thing)", - AccuracyTarget::Exact, - catalog - ) - .is_err()); -} - -#[test] -fn the_catalog_does_not_leak_across_calls() { - // The ambient catalog is scoped to the single `_with_histograms` call; a - // subsequent plain `lower_promql` sees no metadata (guards against a - // thread-local that isn't cleaned up). - let _ = with_meta( - "histogram_quantile(0.9, foo_bucket)", - HistogramCatalog::new().with("foo_bucket", HistogramKind::RawSamples), - ); - // `foo_bucket` would be sketch-able under that catalog, but with none it must - // revert to the heuristic (the `_bucket` name → HistogramQuantile). - assert_eq!(heuristic("histogram_quantile(0.9, foo_bucket)"), "HQ"); -} diff --git a/crates/frontend-promql/tests/unified_promql_conformance.rs b/crates/frontend-promql/tests/unified_promql_conformance.rs deleted file mode 100644 index 6ea89a0bb..000000000 --- a/crates/frontend-promql/tests/unified_promql_conformance.rs +++ /dev/null @@ -1,2208 +0,0 @@ -//! PromQL **semantic conformance** for the parse-to-canonical-tree lowering. -//! -//! We *lower* PromQL to the intent algebra; we do not *execute* it. So "same -//! semantic job as Prometheus" here means: for each canonical query, does the -//! canonical tree encode the **documented PromQL meaning** — and where we knowingly -//! diverge (reject, approximate, or drop a modifier), is that pinned by a test -//! so it stays visible? -//! -//! Sources for the queries + their semantics: -//! - PromQL basics (data types, selectors, offset/@/subquery): -//! -//! - PromLabs PromQL cheat sheet (common real-world queries by category): -//! -//! - Prometheus' own engine test corpus (these are *execution* tests — -//! load → eval → expect values — so they define semantics we mirror as -//! *structure*): -//! Relevant files, mapped to the sections below: selectors.test, -//! aggregators.test, functions.test, histograms.test, operators.test, -//! subquery.test, at_modifier.test, literals.test, limit.test -//! -//! Legend used in test names: -//! - (no suffix) — we lower it and the canonical intent matches PromQL. -//! - `__GAP` — a PromQL capability we don't *yet* support. It is **cleanly -//! rejected** (never silently mislowered), and pinned here so adding support -//! later flips the assertion deliberately. -//! -//! NOTE: the formerly-silent divergences (`group`→sum, dropped `offset`/`@`, -//! `changes`/`resets`→count) are now rejected rather than mislowered — see the -//! equivalence suite (`promql_equivalence.rs`) and section L below. - -// `__GAP`-suffixed test names intentionally SHOUT the documented divergences. -#![allow(non_snake_case)] - -use std::rc::Rc; -use std::time::Duration; - -use asap_frontend_promql::unified::PromqlError as LoweringError; -#[path = "unified_support.rs"] -mod support; -use asap_types::ir::{ - BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, -}; -use asap_types::pre_asap::schema::DataType; -use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, PromQLVectorSetOpKind, - Reduction, SampleKind, ScalarValue, Source, TimeFunc, -}; -use asap_types::types::AccuracyTarget; -use support::{lower_promql, promql_scalar}; - -// ── harness helpers ───────────────────────────────────────────────────────────── - -/// Lower, expecting success. -fn ok(q: &str) -> Rc { - lower_promql(q, AccuracyTarget::Exact) - .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) -} - -/// Lower, expecting a clean `LoweringError` (an unsupported capability). -fn rejected(q: &str) -> LoweringError { - match lower_promql(q, AccuracyTarget::Exact) { - Err(e) => e, - Ok(tree) => panic!("expected {q:?} to be rejected, but it lowered to: {tree:?}"), - } -} - -/// Every `AggIntent` anywhere in the tree, root-to-leaf. -fn intents(e: &OperatorNode) -> Vec { - let mut out = Vec::new(); - collect(e, &mut out); - out -} - -/// `AggIntent` only ever lives in `Aggregate.measures`, never in a scalar -/// position (issue #205); `children()` also descends into the operators a -/// scalar position reads (`scalar(v)`). -fn collect(e: &OperatorNode, out: &mut Vec) { - if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { - out.extend(measures.iter().cloned()); - } - for child in e.children() { - collect(child, out); - } -} - -/// The first `Scan` reached by descending single-child nodes, with its metric -/// name and predicate count. -fn first_scan(e: &OperatorNode) -> (String, usize) { - match e.expect_non_asap() { - NonASAPOp::Scan { - source, predicates, .. - } => { - let name = match source { - Source::TimeSeries { metric } => metric.clone(), - Source::Table { table_ref } => table_ref.clone(), - }; - (name, predicates.len()) - } - NonASAPOp::TimeRange { child, .. } - | NonASAPOp::TimeShift { child, .. } - | NonASAPOp::Aggregate { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } - | NonASAPOp::PromqlSubquery { child, .. } => first_scan(child), - other => panic!("no Scan reachable from {other:?}"), - } -} - -fn has bool>(e: &OperatorNode, pred: F) -> bool { - intents(e).iter().any(pred) -} - -/// Whether the tree contains a `Mul`-by-`ScalarExpr(-1)` anywhere — the shape unary -/// negation lowers to (issue #36). -fn negates_via_scalar(e: &OperatorNode) -> bool { - fn negative(expr: &ScalarExpr) -> bool { - matches!(expr, ScalarExpr::Negative { .. }) || expr.children().iter().any(|e| negative(e)) - } - e.expect_non_asap() - .scalar_exprs() - .iter() - .any(|e| negative(e)) - || e.children().iter().any(|e| negates_via_scalar(e)) -} - -// ───────────────────────────────────────────────────────────────────────────── -// A. Selectors & label matchers (basics §"Instant/Range Vector -// Selectors"; selectors.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn instant_vector_selector() { - // SEMANTICS: bare metric → instant vector (latest sample per series). - let (metric, preds) = first_scan(&ok("node_cpu_seconds_total")); - assert_eq!(metric, "node_cpu_seconds_total"); - assert_eq!(preds, 0, "no label matchers → no predicates"); -} - -#[test] -fn promql_scan_schema_is_open() { - // A schemaless PromQL leaf is *open*: the metric's full label set is - // runtime-only, so the binding schema lists only the (ts, value) floor + - // referenced labels and may be a subset of the runtime row. - let qe = ok("node_cpu_seconds_total"); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected a TimeRange for a bare selector, got {qe:?}"); - }; - let NonASAPOp::Scan { schema, .. } = child.expect_non_asap() else { - panic!("expected a Scan inside the TimeRange, got {qe:?}"); - }; - assert!( - !schema.closed, - "a schemaless PromQL scan has an open schema" - ); -} - -#[test] -fn label_matchers_become_scan_predicates() { - // SEMANTICS: `=`, `!=`, `=~`, `!~` filter series; one conjunct per matcher. - let (_, preds) = first_scan(&ok( - r#"http_requests_total{job!="x",path=~"/api/.*",env!~"dev"}"#, - )); - assert_eq!(preds, 3, "three matchers → three Scan predicates"); -} - -#[test] -fn name_label_selects_the_metric() { - // SEMANTICS: the metric name is the internal `__name__` label. - let (metric, preds) = first_scan(&ok(r#"{__name__="up"}"#)); - assert_eq!(metric, "up"); - assert_eq!(preds, 0, "__name__ is the metric, not a residual predicate"); -} - -#[test] -fn name_regex_matcher_is_rejected__GAP() { - // A `__name__=~` / `!~` / `!=` matcher selects *across* metric names, which - // the single-metric `Source::TimeSeries { metric }` can't represent. It is - // rejected (issue #67) rather than silently mislowered to a literal metric - // named after the pattern (`{__name__=~"node_.*"}` → `Source("node_.*")`). - // Full support needs a wildcard/regex `Source` in the IR. - let _ = rejected(r#"{__name__=~"node_.*"}"#); - let _ = rejected(r#"{__name__!~"x", job="y"}"#); - // Equality still names the metric (regression guard for the fix). - let (metric, _) = first_scan(&ok(r#"{__name__="up"}"#)); - assert_eq!(metric, "up"); -} - -#[test] -fn range_vector_selector_is_time_range() { - // SEMANTICS: `[5m]` turns an instant vector into a range vector, - // represented in the canonical tree as a dedicated `TimeRange` node. - let qe = ok("node_cpu_seconds_total[5m]"); - let NonASAPOp::TimeRange { range, .. } = qe.expect_non_asap() else { - panic!("expected TimeRange for a range-vector selector, got {qe:?}"); - }; - assert_eq!(*range, Duration::from_secs(300)); -} - -// ───────────────────────────────────────────────────────────────────────────── -// B. Counters: rate / irate / increase (cheat sheet "Rates of Increase"; -// functions.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn selector_time_ranges_carry_their_kind() { - // SEMANTICS: an instant selector reads the latest sample within the - // ingestion interval (`Instant`); `m[5m]` is a range selection (`Range`). - // Same length is not the same shape: `m` and `m[1s]` stay distinct. - assert!(matches!( - ok("node_cpu_seconds_total").expect_non_asap(), - NonASAPOp::TimeRange { - kind: TimeRangeKind::Instant, - .. - } - )); - assert!(matches!( - ok("node_cpu_seconds_total[5m]").expect_non_asap(), - NonASAPOp::TimeRange { - kind: TimeRangeKind::Range, - .. - } - )); - assert_ne!( - ok("node_cpu_seconds_total"), - ok("node_cpu_seconds_total[1s]") - ); -} - -#[test] -fn rate_range_lives_in_time_range_node() { - // SEMANTICS: per-second average rate; the temporal range lives on the - // enclosing `TimeRange` node, not inside the intent. - let qe = ok("rate(http_requests_total[5m])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { - panic!("expected TimeRange child, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(300)); -} - -#[test] -fn irate_maps_to_its_own_intent() { - assert!(has(&ok("irate(http_requests_total[1m])"), |i| matches!( - i, - AggIntent::IRate - ))); -} - -#[test] -fn increase_range_lives_in_time_range_node() { - let qe = ok("increase(http_requests_total[1h])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { - panic!("expected TimeRange child, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(3600)); -} - -// ───────────────────────────────────────────────────────────────────────────── -// C. Aggregation across series (cheat sheet "Aggregating Over -// Multiple Series"; aggregators.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn sum_collapses_all_series() { - // SEMANTICS: `sum(v)` → one output series. No grouping → no Partition. - let qe = ok("sum(node_filesystem_size_bytes)"); - assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); - assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); -} - -#[test] -fn sum_by_groups_via_positional_aggregate() { - // SEMANTICS: `by(job,instance)` keeps those labels; the grouping lives on a - // positional `Aggregate.by` — the same shape SQL `GROUP BY` produces (not a - // name-based Partition). SchemaResolver leaf = [ts, value, instance, job] (referenced - // keys appended sorted), so the keys resolve to columns [2, 3]. - let qe = ok("sum by(job, instance) (node_filesystem_size_bytes)"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected positional Aggregate for `by(...)`, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::by(vec![2, 3]), - "group keys resolve to positional ColumnIds" - ); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!( - matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) - ); -} - -#[test] -fn count_is_row_count() { - assert!(has(&ok("count(up)"), |i| matches!( - i, - AggIntent::Count { .. } - ))); -} - -#[test] -fn avg_min_max_stddev_stdvar_quantile_aggregators() { - assert!(has(&ok("avg(up)"), |i| matches!(i, AggIntent::Avg { .. }))); - assert!(has(&ok("min(up)"), |i| matches!(i, AggIntent::Min { .. }))); - assert!(has(&ok("max(up)"), |i| matches!(i, AggIntent::Max { .. }))); - assert!(has(&ok("stddev(up)"), |i| matches!( - i, - AggIntent::StdDev { .. } - ))); - assert!(has(&ok("stdvar(up)"), |i| matches!( - i, - AggIntent::Variance { .. } - ))); - assert!(has(&ok("quantile(0.5, up)"), |i| matches!( - i, - AggIntent::Quantile { .. } - ))); -} - -#[test] -fn sum_without_groups_by_the_complement() { - // SEMANTICS (issue #39): `without(instance)` = group by all labels EXCEPT - // instance. The complement can't be enumerated under the open usage-derived - // schema, so the excluded label is stored and the kept set is deferred to - // the runtime: the grouping is the exclusion form and the output schema - // stays OPEN (unlike `by`, which freezes to closed). - let qe = ok("sum without(instance) (node_filesystem_size_bytes)"); - let NonASAPOp::Aggregate { - reduction, - measures, - .. - } = qe.expect_non_asap() - else { - panic!("expected an Aggregate, got {qe:?}"); - }; - let by = reduction.expect_reduce(); - assert!( - by.is_without(), - "the grouping is the `without` exclusion form" - ); - assert_eq!(by.keys().len(), 1, "the one excluded label (instance)"); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!( - !qe.schema.clone().closed, - "a `without` result keeps an open schema (kept label set is runtime-only)" - ); -} - -#[test] -fn without_on_topk_is_rejected() { - // `without(...)` is modelled only for reducing aggregations; on topk/bottomk - // (a ranking, not a reduction) it would need without-partitioning, so it is - // rejected rather than silently lowered as a `by` (issue #39). - let e = rejected("topk without (job) (3, http_requests_total)"); - assert!(format!("{e}").contains("without"), "got {e}"); -} - -#[test] -fn group_aggregator_lowers_to_a_distinct_intent() { - // SEMANTICS (PromQL): `group(v)` returns a constant 1 per group (presence), - // NOT a sum. It now lowers to a distinct `Group` intent (never folded onto - // `Sum`) — see §S. Regression guard that it is not a `Sum`. - let qe = ok("group by (job) (up)"); - assert!(has(&qe, |i| *i == AggIntent::Group)); - assert!(!has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); -} - -// ───────────────────────────────────────────────────────────────────────────── -// D. Two-level: outer aggregation OVER an inner counter (the canonical -// `sum(rate(...))` shape; aggregators.test + functions.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn sum_of_rate_is_two_levels() { - // SEMANTICS: per-series rate, THEN cross-series sum. Both must survive. - let qe = ok("sum(rate(http_requests_total[5m]))"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) - )); -} - -#[test] -fn sum_by_of_rate_groups_outer_level() { - // Outer cross-series Sum grouped on positional `Aggregate.by` over the - // label-preserving inner Rate. Leaf = [ts, value, instance] → by = [2]. - let qe = ok("sum by(instance) (rate(node_network_receive_bytes_total[5m]))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate grouped by instance, got {qe:?}"); - }; - assert_eq!(reduction, &Reduction::by(vec![2])); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - // child is the inner per-series Rate aggregate. - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) - )); -} - -#[test] -fn sum_by_of_over_time_groups_outer_level() { - // Outer cross-series Sum grouped on positional `Aggregate.by` over an inner - // *per-series* `avg_over_time` — `Window { Aggregate{Avg} }` is label- - // preserving, so the key resolves positionally just like the rate case (no - // name-based Partition). Leaf = [ts, value, instance] → by = [2]. - let qe = ok("sum by(instance) (avg_over_time(node_cpu_seconds_total[5m]))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate grouped by instance, got {qe:?}"); - }; - assert_eq!(reduction, &Reduction::by(vec![2])); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - // child is the inner per-series reduction: Aggregate{Avg} over TimeRange. - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected Aggregate (per-series avg_over_time) under the Sum, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -// ───────────────────────────────────────────────────────────────────────────── -// E. Aggregation over time (per-series) (cheat sheet "Aggregating Over -// Time"; functions.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn over_time_functions_reduce_over_time_range() { - // SEMANTICS: reduce the samples WITHIN each series over the range → - // Aggregate over TimeRange (per-series, label-preserving). - for (q, want) in [ - ("avg_over_time(go_goroutines[5m])", "avg"), - ("max_over_time(process_resident_memory_bytes[1d])", "max"), - ("min_over_time(go_goroutines[5m])", "min"), - ("sum_over_time(go_goroutines[5m])", "sum"), - ("count_over_time(go_goroutines[5m])", "count"), - ] { - let qe = ok(q); - assert!( - matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. }), - "{q}: expected Aggregate" - ); - let matched = intents(&qe).iter().any(|i| match want { - "avg" => matches!(i, AggIntent::Avg { .. }), - "max" => matches!(i, AggIntent::Max { .. }), - "min" => matches!(i, AggIntent::Min { .. }), - "sum" => matches!(i, AggIntent::Sum { .. }), - "count" => matches!(i, AggIntent::Count { .. }), - _ => unreachable!(), - }); - assert!(matched, "{q}: missing {want} intent"); - } -} - -#[test] -fn quantile_over_time_is_aggregate_over_time_range() { - let qe = ok("quantile_over_time(0.9, request_latency_seconds[5m])"); - assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); - assert!(has( - &qe, - |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) - )); -} - -// ───────────────────────────────────────────────────────────────────────────── -// F. Histograms (cheat sheet "Quantiles from -// Histograms"; histograms.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn histogram_quantile_over_rate() { - // φ-quantile from bucket rates. The `_bucket` metric marks the classic - // cumulative-bucket form → `HistogramQuantile` (even without `sum by (le)`). - let qe = ok("histogram_quantile(0.9, rate(demo_api_request_duration_seconds_bucket[5m]))"); - let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { - panic!("expected Aggregate{{HistogramQuantile}}, got {qe:?}"); - }; - assert!( - matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.9).abs() < 1e-9) - ); - assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); -} - -#[test] -fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { - // SEMANTICS: the standard pattern — bucket rates summed by `le`, then the - // quantile. The `sum by (le)` aggregation must survive into the - // canonical tree. - let qe = ok( - "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", - ); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); - }; - // `by (le)` marks the classic cumulative-bucket form → `HistogramQuantile`. - assert!(matches!( - measures.as_slice(), - [AggIntent::HistogramQuantile { .. }] - )); - // `sum by(le)` now survives as a positional Aggregate (by = [2], `le`), over - // the inner Rate — no name-based Partition. - let NonASAPOp::Aggregate { - reduction, - measures, - .. - } = child.expect_non_asap() - else { - panic!("expected `sum by(le)` as a positional Aggregate, got {child:?}"); - }; - assert_eq!(reduction, &Reduction::by(vec![2])); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); -} - -// ───────────────────────────────────────────────────────────────────────────── -// G. Binary ops: math, matching, comparison (cheat sheet "Math Between -// Series" / "Filtering Series by Value"; operators.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn vector_arithmetic() { - let qe = ok("node_memory_MemFree_bytes + node_memory_Cached_bytes"); - let NonASAPOp::BinaryOp { - operator: BinaryOperator { kind: op, .. }, - .. - } = qe.expect_non_asap() - else { - panic!("expected BinaryOp, got {qe:?}"); - }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Add)); -} - -#[test] -fn on_matching_with_group_left() { - // SEMANTICS: many-to-one matching on a label subset. - let qe = - ok("rate(demo_cpu_usage_seconds_total[1m]) / on(instance, job) group_left demo_num_cpus"); - let NonASAPOp::BinaryOp { operator, .. } = qe.expect_non_asap() else { - panic!("expected BinaryOp, got {qe:?}"); - }; - assert_eq!( - operator.kind, - BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) - ); - let vm = operator - .vector_match - .as_ref() - .expect("on(...) group_left present"); - assert_eq!(vm.labels, vec!["instance".to_string(), "job".to_string()]); - assert!( - vm.grouping.is_some(), - "group_left should set the grouping side" - ); -} - -#[test] -fn vector_comparison_filters() { - // SEMANTICS: `>` between two vectors keeps the LHS series where it holds. - let qe = ok("go_goroutines > go_threads"); - assert!( - matches!(qe.expect_non_asap(), NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) - ); -} - -#[test] -fn comparison_bool_modifier_returns_zero_or_one() { - // SEMANTICS (operators.test): `bool` turns a filtering comparison into a - // 0/1-valued one. On a vector operand it is `return_bool` on the - // `BinaryOp`; between two scalars it is a `Case(Compare → 1, else 0)` - // scalar expression under PromQL numeric rules — and a scalar comparison - // without `bool` is not a PromQL expression at all. - let bool_flag = |q: &str| match ok(q).expect_non_asap() { - NonASAPOp::BinaryOp { return_bool, .. } => *return_bool, - NonASAPOp::Project { .. } => true, - NonASAPOp::Filter { .. } => false, - other => panic!("expected BinaryOp for {q}, got {other:?}"), - }; - assert!(bool_flag("go_goroutines > bool go_threads")); - assert!(bool_flag("go_goroutines > bool 0")); - assert!(!bool_flag("go_goroutines > go_threads")); - assert!(!bool_flag("go_goroutines > 0")); - - let qe = support::scalar_root("1 < bool 2"); - let ScalarExpr::Case { branches, .. } = &qe else { - panic!("expected a scalar Case, got {qe:?}"); - }; - assert!(matches!( - branches.as_slice(), - [( - ScalarExpr::Compare { - op: CompareOpKind::Lt, - semantics: ExprSemantics::Promql, - .. - }, - _ - )] - )); - rejected("1 < 2"); -} - -#[test] -fn unary_negation_lowers_as_multiply_by_minus_one() { - // SEMANTICS (PromQL, issue #36): `-expr` flips the sign of every sample. - // Now that a scalar operand exists (#35), it lowers as `expr * -1` — a `Mul` - // BinaryOp of the (label-preserving) vector against `ScalarExpr(-1)`. These are - // the five cases the old `__GAP` test pinned as rejected. - for q in [ - "-rate(http_errors_total[5m])", - "-some_metric", - "-metric_a or -metric_b", - "http_requests_total - -http_errors_total", - "sum(-node_cpu_seconds_total)", - ] { - let qe = ok(q); - // A `Mul`-by-`-1` against a `ScalarExpr(-1)` appears somewhere in every tree. - assert!( - negates_via_scalar(&qe), - "no `* -1` negation found in {q}: {qe:?}" - ); - } - - let negated = ok("-some_metric"); - assert!(negates_via_scalar(&negated)); - assert!(negated.schema.has_promql_series_identity()); - assert!(negated.schema.time_index.is_some()); - let summed = ok("sum(-node_cpu_seconds_total)"); - assert!(has(&summed, |i| matches!(i, AggIntent::Sum { .. }))); - assert!(negates_via_scalar(&summed)); -} - -#[test] -fn unary_negation_of_constant_folds_to_scalar() { - // `-(10*1024*1024)` — the operand is constant-foldable, so negation collapses - // to a single negated `ScalarExpr` leaf (no `BinaryOp`), just like a bare literal. - assert!(promql_scalar(&support::scalar_root("-(10*1024*1024)")) - .is_some_and(|v| (v + 10_485_760.0).abs() < 1e-6)); -} - -#[test] -fn double_unary_negation_nests() { - let qe = ok("- -some_metric"); - let NonASAPOp::Project { child, .. } = qe.expect_non_asap() else { - panic!() - }; - assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); - assert!(negates_via_scalar(child)); -} - -#[test] -fn count_maps_to_count_and_inherits_accuracy() { - // Counts preserve the workload accuracy target without counting distinct values. - let exact = lower_promql("count by (job) (up)", AccuracyTarget::Exact).unwrap(); - assert!( - has(&exact, |i| matches!( - i, - AggIntent::Count { - accuracy: AccuracyTarget::Exact - } - )), - "Count must stay Exact under AccuracyTarget::Exact, got {:?}", - intents(&exact) - ); - - let approx = lower_promql("count by (job) (up)", AccuracyTarget::Epsilon(0.01)).unwrap(); - assert!( - has(&approx, |i| matches!( - i, - AggIntent::Count { - accuracy: AccuracyTarget::Epsilon(e) - } if (*e - 0.01).abs() < 1e-9 - )), - "Count must carry the approximate target, got {:?}", - intents(&approx) - ); -} - -#[test] -fn scalar_literal_operand_lowers_as_binaryop_scalar() { - let qe = ok("node_filesystem_avail_bytes > 10*1024*1024"); - let ScalarExpr::Compare { op, right, .. } = support::sample_expression(&qe) else { - panic!() - }; - assert_eq!(*op, CompareOpKind::Gt); - assert_eq!(promql_scalar(right), Some(10_485_760.0)); -} - -#[test] -fn scalar_arithmetic_scales_the_vector() { - let qe = ok("rate(m[5m]) * 100"); - let ScalarExpr::Arithmetic { op, right, .. } = support::sample_expression(&qe) else { - panic!() - }; - assert_eq!(*op, ArithmeticOpKind::Mul); - assert_eq!(promql_scalar(right), Some(100.0)); -} - -// ───────────────────────────────────────────────────────────────────────────── -// H. Set operations (cheat sheet "Set Operations"; -// operators.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn set_ops_lower_to_binaryop() { - // SEMANTICS: or = union of label sets; and = intersection; unless = difference. - let set_op = |q: &str| match ok(q).expect_non_asap() { - NonASAPOp::BinaryOp { operator, .. } => operator.kind.clone(), - other => panic!("expected BinaryOp for {q}, got {other:?}"), - }; - assert_eq!( - set_op("up{job=\"a\"} or up{job=\"b\"}"), - BinaryOpKind::Set(PromQLVectorSetOpKind::Or) - ); - assert_eq!( - set_op("node_network_mtu_bytes and node_up"), - BinaryOpKind::Set(PromQLVectorSetOpKind::And) - ); - assert_eq!( - set_op("node_network_mtu_bytes unless node_down"), - BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) - ); -} - -// ───────────────────────────────────────────────────────────────────────────── -// I. Sorting / top-k (cheat sheet "Sorting"/topk; -// functions.test, limit.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn topk_over_count_is_heavy_hitter() { - // SEMANTICS: top-k by frequency → first-class heavy-hitter `TopK` intent. - let qe = ok("topk(10, count_over_time(http_requests_total[1m]))"); - assert!(has( - &qe, - |i| matches!(i, AggIntent::TopK { k, .. } if *k == 10) - )); -} - -#[test] -fn bottomk_is_generic_sort_limit() { - // SEMANTICS: bottom-k → generic ascending order + limit (no sketch). - let qe = ok("bottomk(3, count_over_time(http_requests_total[5m]))"); - assert!(matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); -} - -#[test] -fn topk_over_nested_sum_preserves_weighted_topk_accuracy() { - // SEMANTICS (PromQL): `topk(3, sum by(x)(rate(...)))` is extremely common. - // The final rates are query-time values. Their ordering does not establish - // frequency-sketch membership semantics. - let qe = ok("topk(3, sum by(instance) (rate(node_cpu_seconds_total[5m])))"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected weighted TopK aggregate, got {qe:?}"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 3, .. }] - )); - // The inner `sum by (instance)` survives as a cross-series Aggregate over the - // per-series rate — the nesting the old two-level template could not express. - assert!( - has(child, |i| matches!(i, AggIntent::Sum { .. })) - && has(child, |i| matches!(i, AggIntent::Rate)), - "inner sum-over-rate preserved, got {:?}", - intents(child) - ); - assert!(has(&qe, |i| matches!(i, AggIntent::TopK { .. }))); -} - -#[test] -fn outer_aggregate_over_nested_aggregate_nests() { - // `max(sum by (job) (rate(m[5m])))` — an outer cross-series reduction over a - // nested per-group reduction over a per-series rate: three stacked levels the - // flat two-level template rejected. Each level survives into the - // canonical tree (issue #27). - let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let NonASAPOp::Aggregate { - reduction, - measures, - .. - } = child.expect_non_asap() - else { - panic!("expected inner `sum by (job)` Aggregate, got {child:?}"); - }; - assert_eq!( - reduction, - &Reduction::by(vec![2]), - "job grouping survives on the inner aggregate" - ); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(has(&qe, |i| matches!(i, AggIntent::Rate)), "rate preserved"); -} - -#[test] -fn outer_group_key_absent_from_nested_aggregate_is_dropped() { - // SEMANTICS (PromQL, issue #53): aggregating `by` a label that no input - // series carries is valid — every series lands in one group and the - // (empty) label is omitted from the output. Here the inner `sum by (group)` - // collapses `job` away (its closed output schema is `[group, sum]`), so the - // outer `by (job)` groups everything into a single global partition: - // the query lowers with the provably-absent key dropped, exactly - // `sum(sum by (group)(…))`. - let qe = ok(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::by(vec![]), - "absent `job` key dropped → global aggregate" - ); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let NonASAPOp::Aggregate { reduction, .. } = child.expect_non_asap() else { - panic!("expected inner `sum by (group)` Aggregate, got {child:?}"); - }; - assert_eq!( - reduction, - &Reduction::by(vec![2]), - "inner grouping on `group` survives" - ); -} - -#[test] -fn outer_group_key_present_after_inner_aggregate_still_resolves() { - // The counterpart guard for #53: when the outer key IS in the inner - // aggregate's output (`by (job)` over `sum by (job, group)`), it must keep - // resolving positionally — the absent-key drop only fires on provable - // absence, never on a resolvable key. - let qe = ok("sum(sum by (job, group)(http_requests)) by (job)"); - let NonASAPOp::Aggregate { - reduction, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - let NonASAPOp::Aggregate { - reduction: inner_reduction, - .. - } = child.expect_non_asap() - else { - panic!("expected inner Aggregate, got {child:?}"); - }; - // Inner output schema is [group, job, sum] (keys in label-column order, - // labels alphabetical on the scan) → job = col 1. - assert_eq!( - reduction, - &Reduction::by(vec![1]), - "outer `job` resolves against the inner output" - ); - assert_eq!(inner_reduction.expect_reduce().len(), 2); -} - -#[test] -fn outer_group_key_over_binary_op_resolves_on_both_sides() { - // Issue #52: an outer aggregate's group key that appears in *neither* side of - // a binary op — the metric-name label `__name__`, or a plain `job` — must - // still resolve. Each `or` side is bound independently against its own - // sub-tree, so the key is seeded as an inherited column on both sides. - let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); - let NonASAPOp::Aggregate { - reduction, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - // `__name__` resolves to a single positional id against the binary op output. - assert_eq!( - reduction.expect_reduce().len(), - 1, - "grouped by the one `__name__` key" - ); - let NonASAPOp::BinaryOp { lhs, rhs, .. } = child.expect_non_asap() else { - panic!("expected a BinaryOp child, got {child:?}"); - }; - // Both independently-bound sides carry `__name__` at the same position, so - // the outer group key is consistent across the union. - let (ls, rs) = (lhs.schema.clone(), rhs.schema.clone()); - assert_eq!(ls.column_id("__name__"), rs.column_id("__name__")); - assert_eq!( - ls.column_id("__name__"), - Some(reduction.expect_reduce().keys()[0]) - ); - - // The general case (a plain label, not just `__name__`) also lowers. - assert!(matches!( - ok("sum by (job)(metric_a or metric_b)").expect_non_asap(), - NonASAPOp::Aggregate { .. } - )); -} - -#[test] -fn aggregate_over_binary_op_nests() { - // `sum(rate(a[5m]) + rate(b[5m]))` — an aggregate whose argument is a binary - // op over two range vectors. The old template only accepted a single inner - // selector/call; now the binary op lowers and the outer sum wraps it. - let qe = ok("sum(rate(a[5m]) + rate(b[5m]))"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!( - matches!(child.expect_non_asap(), NonASAPOp::BinaryOp { .. }), - "argument lowers as a BinaryOp, got {child:?}" - ); -} - -// ───────────────────────────────────────────────────────────────────────────── -// J. Subqueries (basics §Subqueries; subquery.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn subquery_wraps_inner_query() { - // SEMANTICS: `[range:res]` evaluates the inner query across a range. - let qe = ok("rate(demo_api_request_duration_seconds_count[5m])[1h:]"); - assert!(matches!( - qe.expect_non_asap(), - NonASAPOp::PromqlSubquery { .. } - )); - assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); -} - -#[test] -fn over_time_of_subquery_reduces_per_series() { - // SEMANTICS (PromQL): `max_over_time(rate(...)[1h:])` chains a sub-query into - // a range-vector function — the sub-query evaluates `rate` across a 1h range, - // then `max_over_time` takes the max of those samples *per series*. It lowers - // to a per-series `Max` reduction over a `PromqlSubquery` (issue #27). - let qe = ok("max_over_time(rate(demo_api_request_duration_seconds_count[5m])[1h:])"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected an Aggregate at the root, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "`*_over_time` has no grouping — reduces per series" - ); - assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - // The reduction rides directly on the sub-query (the structural range marker - // that keeps it label-preserving), which wraps the inner `rate`. - assert!( - matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), - "the `Max` reduces over a PromqlSubquery, got {child:?}" - ); - assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Rate))); -} - -#[test] -fn quantile_over_time_of_subquery_carries_phi() { - // The `quantile_over_time` φ parameter is read from arg 0; the sub-query is - // arg 1. It lowers to a per-series `Quantile(φ)` over the `PromqlSubquery`. - let qe = ok("quantile_over_time(0.9, rate(demo[5m])[1h:])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected an Aggregate, got {qe:?}"); - }; - assert!( - matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.9).abs() < 1e-9) - ); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::PromqlSubquery { .. } - )); -} - -#[test] -fn aggregation_over_over_time_of_subquery_keeps_labels() { - // `sum by (job) (max_over_time(rate(m[5m])[1h:]))` — the inner - // `max_over_time` is per-series (label-preserving), so the `job` label - // survives for the OUTER cross-series `sum by (job)` to group on. If the - // inner `Max` collapsed labels, `job` would not resolve here. - let qe = ok("sum by (job) (max_over_time(rate(demo{job=\"api\"}[5m])[1h:]))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - assert!( - matches!(reduction, Reduction::Reduce(by) if !by.is_empty()), - "outer `sum by (job)` groups on a label" - ); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - // Inner node is the per-series `max_over_time` reduction over the subquery. - let NonASAPOp::Aggregate { - reduction: inner_reduction, - measures: inner_measures, - child: inner_child, - .. - } = child.expect_non_asap() - else { - panic!("expected inner Aggregate, got {child:?}"); - }; - assert_eq!(inner_reduction, &Reduction::PerEntity); - assert!(matches!(inner_measures.as_slice(), [AggIntent::Max { .. }])); - assert!(matches!( - inner_child.expect_non_asap(), - NonASAPOp::PromqlSubquery { .. } - )); -} - -#[test] -fn nested_subquery_from_prometheus_docs() { - // SEMANTICS (PromQL): the *nested sub-query* example from the official docs - // (): - // - // max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:]) - // - // Two stacked sub-queries, each feeding a range-vector function; the outer - // `[10m:]` uses the **default resolution** (no explicit step). Each level - // lowers to its own node, so the whole spine pins as: - // - // Max ∘ PromqlSubquery{10m, res: None} ∘ Deriv ∘ PromqlSubquery{30s, res: 5s} - // ∘ Rate ∘ TimeRange{5s} ∘ Scan - // - // Every reduction is per-series (no grouping), so the output schema stays - // the label-preserving `[ts, value]`. - let qe = ok("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"); - - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected `max_over_time` Aggregate at the root, got {qe:?}"); - }; - assert_eq!(reduction, &Reduction::PerEntity); - assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - - let NonASAPOp::PromqlSubquery { - range, - resolution, - child, - } = child.expect_non_asap() - else { - panic!("expected the outer `[10m:]` PromqlSubquery, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(600)); - assert_eq!(*resolution, None, "`[10m:]` keeps the default resolution"); - - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = child.expect_non_asap() - else { - panic!("expected the `deriv` Aggregate, got {child:?}"); - }; - assert_eq!(reduction, &Reduction::PerEntity); - assert!(matches!(measures.as_slice(), [AggIntent::Deriv])); - - let NonASAPOp::PromqlSubquery { - range, - resolution, - child, - } = child.expect_non_asap() - else { - panic!("expected the inner `[30s:5s]` PromqlSubquery, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(30)); - assert_eq!(*resolution, Some(Duration::from_secs(5))); - - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected the `rate` Aggregate, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { - panic!("expected the `[5s]` TimeRange under rate, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(5)); - - // Per-series end to end: the schema keeps the (ts, value) floor and stays open. - let schema = qe.schema.clone(); - assert_eq!( - schema - .fields - .iter() - .map(|c| c.name.as_str()) - .collect::>(), - vec!["ts", "value"], - ); - assert!(!schema.closed, "per-series chain never freezes the schema"); -} - -// ───────────────────────────────────────────────────────────────────────────── -// K. Time-shift modifiers (basics §Offset/@; at_modifier.test) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn offset_modifier_lowers_to_a_time_shift() { - // SEMANTICS (PromQL, issue #40): `offset 5m` shifts the lookback 5m into the - // past — a `TimeShift` wrapper over the selector (signed ms; a negative - // offset shifts forward). Schema is unchanged (the shift only moves *when*). - let qe = ok("http_requests_total offset 5m"); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected an ingestion TimeRange, got {qe:?}"); - }; - let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { - panic!("expected a TimeShift, got {qe:?}"); - }; - assert_eq!(shift.offset_ms, 300_000); - assert!(shift.at.is_none()); - assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); - - // `offset -5m` shifts forward → negative ms. - let qe = ok("http_requests_total offset -5m"); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected an ingestion TimeRange"); - }; - let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { - panic!("expected a TimeShift"); - }; - assert_eq!(shift.offset_ms, -300_000); -} - -#[test] -fn at_modifier_lowers_to_a_time_shift() { - // SEMANTICS (PromQL, issue #40): `@ ` pins the evaluation to an absolute - // instant (PromQL seconds → IR milliseconds); `@ start()` / `@ end()` anchor - // to the query range bounds. - let qe = ok("http_requests_total @ 1609746000"); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected an ingestion TimeRange"); - }; - let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { - panic!("expected a TimeShift for `@ `"); - }; - assert_eq!(shift.at, Some(AtModifier::Timestamp(1_609_746_000_000))); - assert_eq!(shift.offset_ms, 0); - - let qe = ok("http_requests_total @ start()"); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected an ingestion TimeRange"); - }; - let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { - panic!("expected a TimeShift for `@ start()`"); - }; - assert_eq!(shift.at, Some(AtModifier::Start)); - - // Offset and `@` compose: `@ end() offset 5m` carries both. - let qe = ok("http_requests_total @ end() offset 5m"); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected an ingestion TimeRange, got {qe:?}"); - }; - let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { - panic!("expected a TimeShift, got {qe:?}"); - }; - assert_eq!(shift.at, Some(AtModifier::End)); - assert_eq!(shift.offset_ms, 300_000); -} - -#[test] -fn offset_on_a_ranged_selector_wraps_inside_the_time_range() { - // `rate(m[5m] offset 1h)` — the offset is on the ranged selector, so the - // `TimeShift` sits *under* the `TimeRange` (the 5m window is taken at the - // shifted time), and the whole thing under the per-series `Rate` (#40). - let qe = ok("rate(http_requests_total[5m] offset 1h)"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected the rate Aggregate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let NonASAPOp::TimeRange { child, .. } = child.expect_non_asap() else { - panic!("expected a TimeRange under rate, got {child:?}"); - }; - let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { - panic!("expected a TimeShift under the TimeRange, got {child:?}"); - }; - assert_eq!(shift.offset_ms, 3_600_000); - assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); -} - -// ───────────────────────────────────────────────────────────────────────────── -// L. Unsupported functions (functions.test) — clean rejection -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn unsupported_functions_are_rejected() { - // These parse fine but have no intent-algebra lowering yet. Each must return - // a clean LoweringError rather than mislower. - for q in [ - "step()", - "range()", - r#"histogram_quantiles("le", 0.5, 0.9, x)"#, - // NOTE: counter-derivatives (#44), math/trig (#45, §O), presence (#47, - // §P), time/calendar (#46, §Q), vector/scalar (#48, §R), - // label_replace/label_join (#50, §T) and the extra range reducers + - // sort family (#51, §U) now lower — see those sections. `info` (#84), - // `min_of`/`max_of` (#89) are pinned in §R / §U. - ] { - let _ = rejected(q); - } -} - -// ───────────────────────────────────────────────────────────────────────────── -// M. Counter-derivative range functions (functions.test; issue #44) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn count_over_time_value_column_is_float64() { - // #69: a per-series range reduction produces a PromQL sample value, which is - // always float64. `count_over_time`'s `Count` intent types `Int64`, but the - // derived `value` column must be `Float64` like every other range reducer. - let schema = ok("count_over_time(m[5m])").schema.clone(); - let value = schema - .fields - .iter() - .find(|c| c.name == "value") - .expect("value column"); - assert_eq!(value.dtype, DataType::Float64); -} - -#[test] -fn counter_derivative_functions_lower_to_distinct_intents() { - // Each range function reduces one series' window to one value per series - // (label-preserving), riding on a `TimeRange`, and carries its OWN intent — - // deliberately not aliased to rate/increase/count. - for (q, want) in [ - ("changes(m[15m])", AggIntent::Changes), - ("delta(m[5m])", AggIntent::Delta), - ("idelta(m[5m])", AggIntent::IDelta), - ("deriv(m[1h])", AggIntent::Deriv), - ("resets(m[1h])", AggIntent::Resets), - ] { - let qe = ok(q); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected an Aggregate for {q:?}, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); - assert_eq!( - measures.as_slice(), - std::slice::from_ref(&want), - "{q}: wrong intent" - ); - assert!( - matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. }), - "{q}: reduction rides on a TimeRange, got {child:?}" - ); - } -} - -#[test] -fn predict_linear_carries_horizon_seconds() { - // `predict_linear(v[w], t)` — the 2nd (scalar) arg is the prediction horizon - // in seconds; it must be carried in the intent (it changes the result). - let qe = ok("predict_linear(node_filesystem_avail_bytes[3h], 86400)"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - measures.as_slice(), - &[AggIntent::PredictLinear { seconds: 86400.0 }] - ); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -#[test] -fn double_exponential_smoothing_carries_factors() { - let want = AggIntent::DoubleExpSmoothing { - smoothing: 0.5, - trend: 0.3, - }; - let a = ok("double_exponential_smoothing(m[10m], 0.5, 0.3)"); - assert_eq!(intents(&a).as_slice(), std::slice::from_ref(&want)); -} - -#[test] -fn aggregation_over_counter_derivative_keeps_labels() { - // A counter-derivative is per-series (label-preserving), so an outer - // `sum by (job)` can group on a label the inner `changes` preserved. - let qe = ok(r#"sum by (job) (changes(m{job="api"}[15m]))"#); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - assert!( - matches!(reduction, Reduction::Reduce(by) if !by.is_empty()), - "outer sum groups on job" - ); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Changes))); - let _ = child; -} - -#[test] -fn outer_stat_over_counter_derivative_nests_two_levels() { - // A cross-series stat over a counter-derivative is a genuine two-level - // reduction: the derivative runs per series (inner), the stat aggregates - // across series (outer). They must not collapse into one node — and a - // grouped outer (`avg by (dc)`) must resolve its key against the labels the - // inner reduction preserved, threading any scalar param (predict horizon). - let qe = ok("avg by (dc) (predict_linear(m[3h], 3600))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - assert!( - matches!(reduction, Reduction::Reduce(by) if !by.is_empty()), - "outer `avg by (dc)` groups on a label" - ); - assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let NonASAPOp::Aggregate { - reduction: inner_reduction, - measures: inner_measures, - .. - } = child.expect_non_asap() - else { - panic!("expected inner per-series Aggregate, got {child:?}"); - }; - assert_eq!( - inner_reduction, - &Reduction::PerEntity, - "inner derivative stays per-series" - ); - assert_eq!( - inner_measures.as_slice(), - std::slice::from_ref(&AggIntent::PredictLinear { seconds: 3600.0 }) - ); -} - -#[test] -fn topk_over_counter_derivative_is_generic_sort_limit() { - // `topk(k, deriv(...))` ranks the per-series derivative values — a generic - // `Sort + Limit`, NOT a heavy-hitter `TopK` (that's only `count_over_time`). - let qe = ok("topk(3, deriv(m[5m]))"); - let NonASAPOp::Limit { - n: Some(n), child, .. - } = qe.expect_non_asap() - else { - panic!("expected Limit, got {qe:?}"); - }; - assert_eq!(*n, 3); - assert!(matches!(child.expect_non_asap(), NonASAPOp::Sort { .. })); - assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Deriv))); - assert!( - !intents(&qe) - .iter() - .any(|i| matches!(i, AggIntent::TopK { .. })), - "counter-derivative topk is generic ranking, not a heavy-hitter sketch" - ); -} - -#[test] -fn counter_derivative_composes_in_binary_ops() { - // As a vector operand: `delta(a[5m]) / delta(b[5m])` is a BinaryOp of two - // per-series Delta reductions. - let ratio = ok("delta(a[5m]) / delta(b[5m])"); - let NonASAPOp::BinaryOp { - operator: BinaryOperator { kind: op, .. }, - lhs, - rhs, - .. - } = ratio.expect_non_asap() - else { - panic!("expected BinaryOp, got {ratio:?}"); - }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); - assert!( - matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) - ); - assert!( - matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) - ); - - // Under an aggregate over a binary op mixing a counter-derivative with - // another per-series function: `sum(rate(m[5m]) + changes(m[5m]))`. - let mixed = ok("sum(rate(m[5m]) + changes(m[5m]))"); - let NonASAPOp::Aggregate { - measures, child, .. - } = mixed.expect_non_asap() - else { - panic!("expected Aggregate, got {mixed:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::BinaryOp { .. } - )); - assert!(intents(&mixed).iter().any(|i| matches!(i, AggIntent::Rate))); - assert!(intents(&mixed) - .iter() - .any(|i| matches!(i, AggIntent::Changes))); -} - -#[test] -fn range_functions_over_a_subquery_reduce_per_series() { - // Issue #55 — the whole range-vector family accepts a sub-query argument - // (generalizing `*_over_time`, #42): `rate`/`increase`/`irate` and the - // counter-derivatives. Each lowers to a per-series `Aggregate{[f]}` directly - // over the `PromqlSubquery` — the sub-query is the range context, so there is NO - // separate `TimeRange` (that would double the range). - for (q, want) in [ - ("rate(sum(m)[5m:])", AggIntent::Rate), - ("increase(sum(m)[5m:])", AggIntent::Increase), - ("irate(sum(m)[5m:])", AggIntent::IRate), - ("changes(rate(m[5m])[1h:])", AggIntent::Changes), - ("delta(sum(m)[5m:])", AggIntent::Delta), - ("deriv(sum(m)[10m:])", AggIntent::Deriv), - ("resets(sum(m)[5m:])", AggIntent::Resets), - ] { - let qe = ok(q); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("{q}: expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); - assert_eq!( - measures.as_slice(), - std::slice::from_ref(&want), - "{q}: wrong intent" - ); - assert!( - matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), - "{q}: reduces directly over the PromqlSubquery (no TimeRange), got {child:?}" - ); - } -} - -#[test] -fn predict_linear_and_double_exp_over_a_subquery_carry_params() { - // The scalar params survive the sub-query path. - let pl = ok("predict_linear(sum(m)[1h:], 3600)"); - assert!(intents(&pl).iter().any( - |i| matches!(i, AggIntent::PredictLinear { seconds } if (*seconds - 3600.0).abs() < 1e-9) - )); - let de = ok("double_exponential_smoothing(sum(m)[10m:], 0.5, 0.3)"); - assert!(intents(&de).iter().any(|i| matches!( - i, - AggIntent::DoubleExpSmoothing { smoothing, trend } - if (*smoothing - 0.5).abs() < 1e-9 && (*trend - 0.3).abs() < 1e-9 - ))); -} - -// ───────────────────────────────────────────────────────────────────────────── -// N. Native-histogram accessors (functions.test; issue #43) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn histogram_quantile_classic_bucket_vs_native() { - // Two lowerings of `histogram_quantile(φ, …)`: the classic cumulative-bucket - // form → exact `HistogramQuantile`; native samples require a new type. - // The classic form is recognised by - // `by (le)`, a `_bucket` metric, or an `le` matcher (issue #43). - for classic in [ - "histogram_quantile(0.9, sum by (le) (rate(x_bucket[5m])))", - "histogram_quantile(0.9, rate(x_bucket[5m]))", // bare _bucket metric - r#"histogram_quantile(0.9, rate(x{le="0.5"}[5m]))"#, // le matcher - ] { - let qe = ok(classic); - assert!( - has( - &qe, - |i| matches!(i, AggIntent::HistogramQuantile { q, .. } if (*q - 0.9).abs() < 1e-9) - ), - "classic bucket form → HistogramQuantile: {classic}" - ); - assert!( - !has(&qe, |i| matches!(i, AggIntent::Quantile { .. })), - "{classic}" - ); - } - for native in [ - "histogram_quantile(0.9, my_native_histogram)", - "histogram_quantile(0.9, request_duration_seconds)", // raw samples (your extension) - ] { - rejected(native); - } -} - -#[test] -fn native_histogram_accessors_are_explicit_gaps() { - // Native histogram samples have no typed representation yet. - for q in [ - "histogram_count(v)", - "histogram_sum(v)", - "histogram_avg(v)", - "histogram_stddev(v)", - "histogram_stdvar(v)", - ] { - rejected(q); - } -} - -#[test] -fn histogram_fraction_is_an_explicit_gap() { - rejected("histogram_fraction(0, 0.2, v)"); -} - -// ───────────────────────────────────────────────────────────────────────────── -// O. Math / trig scalar-transform functions (functions.test; issue #45) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn math_functions_lower_to_typed_scalar_projections() { - for name in [ - "abs", "ceil", "floor", "sqrt", "ln", "log2", "sgn", "sin", "atanh", "deg", "rad", - ] { - let query = ok(&format!("{name}(v)")); - assert!( - matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) - ); - query.validate_structure().unwrap(); - } -} - -#[test] -fn clamp_and_round_carry_their_params() { - for (query, params) in [ - ("clamp(v,0,100)", vec![0.0, 100.0]), - ("clamp_min(v,1)", vec![1.0]), - ("clamp_max(v,5)", vec![5.0]), - ("round(v)", vec![1.0]), - ("round(v,5)", vec![5.0]), - ] { - let node = ok(query); - let ScalarExpr::FunctionCall { args, .. } = support::sample_expression(&node) else { - panic!() - }; - assert_eq!( - args.iter().skip(1).map(promql_scalar).collect::>(), - params.into_iter().map(Some).collect::>() - ); - } -} - -#[test] -fn pi_lowers_to_a_scalar_constant() { - // `pi()` is the constant π — a `ScalarExpr` leaf, not a `Math` intent. - assert!(promql_scalar(&support::scalar_root("pi()")) - .is_some_and(|v| (v - std::f64::consts::PI).abs() < 1e-12)); -} - -// ───────────────────────────────────────────────────────────────────────────── -// P. Presence functions (functions.test; issue #47) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn presence_functions_lower_to_presence_intents() { - for (q, want) in [ - (r#"absent(up{job="x"})"#, AggIntent::Absent), - ("absent_over_time(m[1h])", AggIntent::AbsentOverTime), - ("present_over_time(m[5m])", AggIntent::PresentOverTime), - ] { - let qe = ok(q); - assert!(intents(&qe).contains(&want), "{q}: got {:?}", intents(&qe)); - } -} - -#[test] -fn absent_keeps_matcher_labels_for_the_synthesized_output() { - // `absent(v)` synthesizes its output labels from `v`'s equality matchers, so - // those labels must survive into the schema — here `job` from `{job="x"}`. - let qe = ok(r#"absent(up{job="x"})"#); - let cols = qe.schema.clone(); - assert!( - cols.fields.iter().any(|c| c.name == "job"), - "matcher label `job` kept, got {:?}", - cols.fields.iter().map(|c| &c.name).collect::>() - ); -} - -// ───────────────────────────────────────────────────────────────────────────── -// Q. Time / calendar functions (functions.test; issue #46) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn time_lowers_to_the_eval_time_scalar() { - assert!(matches!( - support::scalar_root("time()"), - ScalarExpr::EvalTimestamp - )); -} - -#[test] -fn time_minus_vector_is_the_uptime_pattern() { - let qe = ok("time() - process_start_time_seconds"); - assert!( - matches!(support::sample_expression(&qe), ScalarExpr::Arithmetic { op: ArithmeticOpKind::Sub, left, .. } if matches!(left.as_ref(), ScalarExpr::EvalTimestamp)) - ); - assert!(qe.schema.time_index.is_some()); -} - -#[test] -fn calendar_functions_lower_to_time_fn_intents() { - assert!(has(&ok("timestamp(up)"), |i| *i - == AggIntent::TimeFn(TimeFunc::Timestamp))); - for name in [ - "minute", - "hour", - "day_of_week", - "day_of_month", - "day_of_year", - "month", - "year", - "days_in_month", - ] { - let query = ok(&format!("{name}(v)")); - assert!( - matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) - ); - } -} - -#[test] -fn no_arg_calendar_function_reads_the_eval_time() { - let query = ok("day_of_week()"); - let NonASAPOp::Project { child, .. } = query.expect_non_asap() else { - panic!() - }; - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::PromqlVectorFromScalar(ScalarExpr::EvalTimestamp) - )); - assert!( - matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name,.. } if name=="promql_day_of_week") - ); -} - -#[test] -fn timestamp_composes_under_an_outer_aggregation() { - // `sum by (job) (timestamp(up))` — the per-series `timestamp` transform sits - // below an ordinary grouped sum. Both intents must appear in the tree. - let qe = ok("sum by (job) (timestamp(up))"); - assert!(has(&qe, |i| *i == AggIntent::TimeFn(TimeFunc::Timestamp))); - assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); -} - -// ───────────────────────────────────────────────────────────────────────────── -// R. Type-conversion functions: vector() / scalar() (functions.test; issue #48) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn vector_promotes_a_scalar_to_a_vector() { - // SEMANTICS: `vector(s)` is the scalar→instant-vector bridge — a label-less - // single series carrying the scalar's value. - let qe = ok("vector(1)"); - let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { - panic!("expected PromqlVectorFromScalar, got {qe:?}"); - }; - assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); - // Vector-typed: schema has a time index (a scalar leaf has none). - let sch = qe.schema.clone(); - assert!(sch.time_index.is_some()); - assert!(sch.fields.iter().any(|c| c.name == "value")); -} - -#[test] -fn scalar_collapses_a_vector_to_a_scalar() { - let qe = support::scalar_root("scalar(node_load1)"); - let ScalarExpr::PromqlScalarFromVector(inner) = &qe else { - panic!() - }; - assert_eq!(first_scan(inner).0, "node_load1"); -} - -#[test] -fn vector_zero_is_a_vector_operand_of_a_set_op() { - // `up or vector(0)` — the dead-man's-switch. `or` is a set op between two - // vectors, so `vector(0)` must be a vector (a `PromqlVectorFromScalar`), never a - // folded scalar operand. - let qe = ok("up or vector(0)"); - let NonASAPOp::BinaryOp { - operator: BinaryOperator { kind: op, .. }, - rhs, - .. - } = qe.expect_non_asap() - else { - panic!("expected a BinaryOp, got {qe:?}"); - }; - assert_eq!(*op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)); - assert!(matches!( - rhs.expect_non_asap(), - NonASAPOp::PromqlVectorFromScalar(_) - )); -} - -#[test] -fn scalar_of_a_vector_feeds_a_threshold_comparison() { - let qe = ok("node_load1 > scalar(node_cpu_count)"); - let ScalarExpr::Compare { right, .. } = support::sample_expression(&qe) else { - panic!() - }; - assert!(matches!( - right.as_ref(), - ScalarExpr::PromqlScalarFromVector(_) - )); - assert!(qe.schema.time_index.is_some()); -} - -#[test] -fn info_lowers_to_a_label_enrichment_join() { - // `info(v, [selector])` is a label-enrichment *join* against the info - // metric(s) — it lowers to an `PromqlInfoEnrich` over the (unchanged) input vector - // (issue #84). The value/time axis pass through; the enriched labels are - // runtime, so the schema stays the child's. - let qe = ok("info(rate(http_requests_total[5m]))"); - let NonASAPOp::PromqlInfoEnrich { selector, child } = qe.expect_non_asap() else { - panic!("expected an PromqlInfoEnrich, got {qe:?}"); - }; - assert!(selector.is_empty(), "no selector → default target_info"); - // The child is the untouched input (a per-series rate reduction here). - assert!(has(child, |i| *i == AggIntent::Rate)); - assert!(qe.schema.clone().time_index.is_some()); -} - -#[test] -fn info_selector_carries_the_info_side_matchers() { - // `info(v, {__name__=~".+_info", data=~".+"})` — the selector picks the info - // metric(s) via `__name__` and constrains the data labels. Regex / `__name__` - // matchers are kept symbolically (not run through the single-metric selector - // path). - let qe = ok(r#"info(build_info, {__name__=~".+_info", another_data=~".+"})"#); - let NonASAPOp::PromqlInfoEnrich { selector, .. } = qe.expect_non_asap() else { - panic!("expected an PromqlInfoEnrich, got {qe:?}"); - }; - assert_eq!( - selector.len(), - 2, - "both selector matchers kept: {selector:?}" - ); - assert!(selector - .iter() - .any(|m| m.label == "__name__" && m.op == CompareOpKind::Regex)); - assert!(selector.iter().any(|m| m.label == "another_data")); -} - -#[test] -fn info_composes_under_an_aggregation_and_over_a_time_shift() { - // `sum(info(m))` — enrichment first, then a cross-series sum over it. - assert!(has(&ok("sum(info(node_uname_info))"), |i| matches!( - i, - AggIntent::Sum { .. } - ))); - // `offset` / `@` on the input now lower to a `TimeShift` under the info-join - // (issue #40) — the enrichment composes over the shifted selector. - assert!(matches!( - ok("info(metric @ 60)").expect_non_asap(), - NonASAPOp::PromqlInfoEnrich { .. } - )); - assert!(matches!( - ok("info(metric offset 1m)").expect_non_asap(), - NonASAPOp::PromqlInfoEnrich { .. } - )); -} - -// ───────────────────────────────────────────────────────────────────────────── -// S. Extended aggregation operators: group / count_values (aggregators.test; #49) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn group_lowers_to_a_constant_group_intent() { - // SEMANTICS: `group(v)` yields a constant 1 per group — a distinct intent, - // NOT folded onto `sum` (which would return the value sum instead of 1). - let qe = ok("group(up)"); - let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { - panic!("expected an Aggregate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Group])); - // Output column is the constant-1 `group` value. - let sch = qe.schema.clone(); - assert!(sch.fields.iter().any(|c| c.name == "group")); -} - -#[test] -fn group_by_keeps_the_grouping_keys() { - // `group by (job) (up)` — the grouping keys ride on `Aggregate.by`. - let qe = ok("group by (job) (up)"); - let sch = qe.schema.clone(); - assert!(sch.fields.iter().any(|c| c.name == "job")); - assert!(has(&qe, |i| *i == AggIntent::Group)); -} - -#[test] -fn count_values_groups_by_value_and_synthesizes_a_label() { - // SEMANTICS: `count_values("l", v)` groups the input series by their sample - // value, counts each distinct value, and emits that value as a new label - // `l`. The intent carries the label; schema gains a `Utf8` `l` column. - let qe = ok(r#"count_values("version", build_version)"#); - let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { - panic!("expected an Aggregate, got {qe:?}"); - }; - assert!( - matches!(measures.as_slice(), [AggIntent::CountValues { label }] if label == "version") - ); - let sch = qe.schema.clone(); - let version = sch - .fields - .iter() - .find(|c| c.name == "version") - .expect("synthesized `version` label column"); - assert_eq!( - version.dtype, - DataType::Utf8, - "the value becomes a string label" - ); - assert!( - sch.fields.iter().any(|c| c.name == "count"), - "and a count column" - ); -} - -#[test] -fn count_values_accepts_a_parenthesised_label_and_by_grouping() { - // `count_values by (job) ((("v")), m)` — nested parens around the string - // param, plus `by` grouping. Both survive. - let qe = ok(r#"count_values by (job) ((("v")), m)"#); - assert!(has( - &qe, - |i| matches!(i, AggIntent::CountValues { label } if label == "v") - )); - let sch = qe.schema.clone(); - assert!(sch.fields.iter().any(|c| c.name == "job")); - assert!(sch.fields.iter().any(|c| c.name == "v")); -} - -#[test] -fn count_values_label_colliding_with_a_group_key_is_not_duplicated() { - // `count_values by (job)("job", v)` — the synthesized label name collides - // with a group-by key. PromQL's synthesized label takes precedence; the - // output must carry a single `job` column, never two. - let qe = ok(r#"count_values by (job) ("job", version)"#); - let sch = qe.schema.clone(); - let jobs = sch.fields.iter().filter(|c| c.name == "job").count(); - assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.fields); - assert!(sch.fields.iter().any(|c| c.name == "count")); -} - -#[test] -fn limitk_and_limit_ratio_lower_to_series_sampling() { - // `limitk`/`limit_ratio` are series-*sampling* selection — a subset of whole - // series kept unchanged (NOT a ranking), so they lower to the dedicated - // `PromqlSeriesSample` node, never `topk`'s `Sort → Limit` (issue #86). - assert!(matches!( - ok("limitk(2, http_requests)").expect_non_asap(), - NonASAPOp::PromqlSeriesSample { - kind: SampleKind::LimitK(2), - .. - } - )); - assert!(matches!( - ok("limit_ratio(0.1, http_requests)").expect_non_asap(), - NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 - )); - // Series-preserving: the output schema equals the input's (ts, value). - let sch = ok("limitk(2, http_requests)").schema.clone(); - assert!(sch.fields.iter().any(|c| c.name == "value")); - assert!(sch.time_index.is_some()); -} - -#[test] -fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { - // A negative ratio selects the complementary fraction — it must survive, not - // be normalised away. Out-of-range magnitudes clamp to [-1, 1] (Prometheus). - assert!(matches!( - ok("limit_ratio(-0.5, http_requests)").expect_non_asap(), - NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 - )); - assert!(matches!( - ok("limit_ratio(1.1, http_requests)").expect_non_asap(), - NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 - )); -} - -#[test] -fn limitk_by_carries_the_grouping_and_composes_in_a_set_op() { - // `limitk by (group)` samples per group; the grouping label is seeded. - let qe = ok("limitk by (group) (2, http_requests)"); - let NonASAPOp::PromqlSeriesSample { by, .. } = qe.expect_non_asap() else { - panic!("expected a PromqlSeriesSample, got {qe:?}"); - }; - assert!(!by.is_empty(), "grouped sampling keeps its `by` keys"); - // `count(limitk(2, v) and v)` — the surviving series' identity matters, so - // the PromqlSeriesSample must be preserved under the set op (it must lower, not reject). - assert!(has( - &ok("count(limitk(2, http_requests) and http_requests)"), - |i| matches!(i, AggIntent::Count { .. }) - )); -} - -#[test] -fn dynamic_and_non_finite_sample_params_are_rejected() { - // A dynamic k/ratio (not a compile-time constant) or a NaN can't be a static - // `PromqlSeriesSample` param — rejected rather than mislowered. - let _ = rejected("limitk(NaN, http_requests)"); - let _ = rejected("limitk(scalar(foo), http_requests)"); - let _ = rejected("limit_ratio(time() % 17 / 17, http_requests)"); -} - -// ───────────────────────────────────────────────────────────────────────────── -// T. Label-rewrite functions: label_replace / label_join (functions.test; #50) -// ───────────────────────────────────────────────────────────────────────────── - -/// Descend single-child nodes to the first `PromqlRelabel`. -fn first_relabel(e: &OperatorNode) -> &OperatorNode { - match e.expect_non_asap() { - NonASAPOp::PromqlRelabel { .. } => e, - NonASAPOp::Aggregate { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::TimeRange { child, .. } - | NonASAPOp::TimeShift { child, .. } => first_relabel(child), - other => panic!("no PromqlRelabel reachable from {other:?}"), - } -} - -/// True when `value` is a `FunctionCall` with the given name. -fn is_fn_named(value: &ScalarExpr, name: &str) -> bool { - matches!(value, ScalarExpr::FunctionCall { name: n, .. } if n == name) -} - -#[test] -fn label_replace_is_a_relabel_over_the_vector() { - // SEMANTICS: `label_replace(v, dst, repl, src, regex)` rewrites the `dst` - // label per series from a regex over `src`; the sample value is untouched. - let qe = ok(r#"label_replace(up, "host", "$1", "instance", "(.+):.*")"#); - let NonASAPOp::PromqlRelabel { dst, value, child } = qe.expect_non_asap() else { - panic!("expected a PromqlRelabel, got {qe:?}"); - }; - assert_eq!(dst, "host"); - // The child is the untouched vector. - let (metric, _) = first_scan(child); - assert_eq!(metric, "up"); - // The value expression is a `label_replace` fn reading the `src` label. - assert!(is_fn_named(value, "label_replace")); - // Output: the child's columns + the synthesized `host` label; value & ts kept. - let sch = qe.schema.clone(); - assert!(sch.fields.iter().any(|c| c.name == "host")); - assert!(sch.fields.iter().any(|c| c.name == "value")); - assert!(sch.time_index.is_some(), "the vector's time axis survives"); -} - -#[test] -fn label_join_concatenates_source_labels() { - // SEMANTICS: `label_join(v, dst, sep, src…)` joins the source labels with - // `sep` into `dst`. - let qe = ok(r#"label_join(up, "combined", "-", "job", "instance")"#); - let NonASAPOp::PromqlRelabel { dst, value, .. } = qe.expect_non_asap() else { - panic!("expected a PromqlRelabel, got {qe:?}"); - }; - assert_eq!(dst, "combined"); - assert!(is_fn_named(value, "label_join")); - let sch = qe.schema.clone(); - assert!(sch.fields.iter().any(|c| c.name == "combined")); -} - -#[test] -fn label_replace_composes_under_an_aggregation() { - // `sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))` — - // relabel first, then group by the synthesized label. - let qe = ok(r#"sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))"#); - // A PromqlRelabel sits below the outer Sum. - let relabel = first_relabel(&qe); - assert!( - matches!(relabel.expect_non_asap(), NonASAPOp::PromqlRelabel { dst, .. } if dst == "host") - ); - assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); - let sch = qe.schema.clone(); - assert!(sch.fields.iter().any(|c| c.name == "host")); -} - -// ───────────────────────────────────────────────────────────────────────────── -// U. Long-tail: extra range reducers + the sort family (functions.test; #51) -// ───────────────────────────────────────────────────────────────────────────── - -#[test] -fn extra_over_time_reducers_lower_to_per_series_intents() { - // SEMANTICS: each is a per-series reduction of one series' range window to a - // single value — a `TimeRange`-wrapped `Aggregate` with the matching intent. - for (q, want) in [ - ("last_over_time(m[5m])", AggIntent::LastOverTime), - ("first_over_time(m[5m])", AggIntent::FirstOverTime), - ("mad_over_time(m[5m])", AggIntent::MadOverTime), - ("ts_of_min_over_time(m[5m])", AggIntent::TsOfMinOverTime), - ("ts_of_max_over_time(m[5m])", AggIntent::TsOfMaxOverTime), - ("ts_of_first_over_time(m[5m])", AggIntent::TsOfFirstOverTime), - ("ts_of_last_over_time(m[5m])", AggIntent::TsOfLastOverTime), - ] { - let qe = ok(q); - assert!(has(&qe, |i| *i == want), "{q}: {:?}", intents(&qe)); - // Per-series: the range window survives as a `TimeRange`. - assert!( - matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. })), - "{q} keeps its range as a TimeRange" - ); - } -} - -#[test] -fn last_over_time_composes_under_an_outer_aggregation() { - // `sum by (job) (last_over_time(m[5m]))` — per-series last, THEN cross-series - // sum. Both intents survive (issue #27's arbitrary nesting). - let qe = ok("sum by (job) (last_over_time(m[5m]))"); - assert!(has(&qe, |i| *i == AggIntent::LastOverTime)); - assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); -} - -#[test] -fn sort_and_sort_desc_reorder_by_value_without_a_limit() { - // SEMANTICS: `sort`/`sort_desc` reorder an instant vector by sample value. - // Row-preserving → a bare `Sort` (no `Limit`), ascending / descending. - for (q, ascending) in [ - ("sort(http_requests)", true), - ("sort_desc(http_requests)", false), - ] { - let qe = ok(q); - let NonASAPOp::Sort { keys, child, .. } = qe.expect_non_asap() else { - panic!("{q}: expected a Sort, got {qe:?}"); - }; - assert_eq!(keys.len(), 1); - assert_eq!(keys[0].ascending, ascending, "{q}"); - // No Limit above the Sort — every series is preserved. - assert!(!matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); - // The value column is what it ranks on: descend to the scan. - let (metric, _) = first_scan(child); - assert_eq!(metric, "http_requests"); - } -} - -#[test] -fn sort_by_label_orders_on_each_label_in_turn() { - // `sort_by_label(v, "group", "instance", "job")` — one ascending sort key per - // label, in argument order; the labels are seeded into the schema. - let qe = ok(r#"sort_by_label(http_requests, "group", "instance", "job")"#); - let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { - panic!("expected a Sort, got {qe:?}"); - }; - assert_eq!(keys.len(), 3, "one key per label"); - assert!(keys.iter().all(|k| k.ascending)); - let sch = qe.schema.clone(); - for label in ["group", "instance", "job"] { - assert!(sch.fields.iter().any(|c| c.name == label), "{label} seeded"); - } -} - -#[test] -fn sort_by_label_desc_is_descending() { - let qe = ok(r#"sort_by_label_desc(http_requests, "instance")"#); - let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { - panic!("expected a Sort, got {qe:?}"); - }; - assert!(keys.iter().all(|k| !k.ascending)); -} - -#[test] -fn min_of_max_of_fold_constant_scalars() { - // `min_of`/`max_of` are n-ary scalar reducers. When every argument is a - // constant they constant-fold to a `ScalarExpr` leaf, just like scalar - // arithmetic (#35) — the only form the intent algebra can hold (#89). - assert_eq!( - promql_scalar(&support::scalar_root("min_of(3, 5)")), - Some(3.0) - ); - assert_eq!( - promql_scalar(&support::scalar_root("max_of(3, 5)")), - Some(5.0) - ); - assert_eq!( - promql_scalar(&support::scalar_root("min_of(-2, -5)")), - Some(-5.0) - ); - // Nested folds and use as a threshold operand. - assert_eq!( - promql_scalar(&support::scalar_root("max_of(min_of(2, 3), 10)")), - Some(10.0) - ); - let qe = ok("up > max_of(1, 2)"); - let ScalarExpr::Compare { right: rhs, .. } = support::sample_expression(&qe) else { - panic!("{qe:?}") - }; - assert_eq!(promql_scalar(rhs), Some(2.0)); -} - -#[test] -fn min_of_max_of_ignore_nan_like_the_min_max_aggregators() { - // A NaN argument is skipped (Prometheus `min`/`max` NaN semantics). - assert_eq!( - promql_scalar(&support::scalar_root("max_of(3, NaN)")), - Some(3.0) - ); - assert_eq!( - promql_scalar(&support::scalar_root("min_of(NaN, 3)")), - Some(3.0) - ); -} - -#[test] -fn non_constant_min_of_max_of_is_rejected__GAP() { - // A dynamic argument (`step()` — itself unsupported, #89) can't be folded to - // a constant and there is no scalar min/max node, so it stays rejected - // rather than mislowered. These forms also only appear inside unsupported - // dynamic range / offset positions in the corpus. - let _ = rejected("min_of(step(), 1s)"); - let _ = rejected("max_of(min_of(step() + 1, 1h), 1ms)"); -} diff --git a/crates/frontend-promql/tests/unified_promql_lowering.rs b/crates/frontend-promql/tests/unified_promql_lowering.rs deleted file mode 100644 index de769361e..000000000 --- a/crates/frontend-promql/tests/unified_promql_lowering.rs +++ /dev/null @@ -1,1615 +0,0 @@ -//! End-to-end tests for PromQL → unresolved → canonical tree lowering. - -use std::rc::Rc; -use std::time::Duration; - -use asap_types::ir::{ - BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, -}; -use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, Reduction, ScalarValue, Source, -}; -use asap_types::types::AccuracyTarget; -use asap_types::workload::{ - AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, - Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, -}; - -use asap_frontend_promql::unified::{lower_promql_workload, PromqlError as LoweringError}; -#[path = "unified_support.rs"] -mod support; -use support::lower_promql; - -fn lower(q: &str) -> Rc { - lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) -} - -#[test] -fn frequency_extensions_lower_to_explicit_frequency_statistics() { - // ProjectASAP extensions reduce a frequency vector; numeric sample norms - // have different semantics and must never be silently aliased here. - for (query, expected) in [ - ( - "entropy_over_time(cpu_usage[5m])", - AggIntent::FrequencyEntropy { - col: None, - accuracy: AccuracyTarget::Exact, - }, - ), - ( - "l2_over_time(cpu_usage[5m])", - AggIntent::FrequencyL2 { - col: None, - accuracy: AccuracyTarget::Exact, - }, - ), - ] { - assert!(all_intents(&lower(query)) - .iter() - .any(|intent| std::mem::discriminant(intent) == std::mem::discriminant(&expected))); - } -} - -#[test] -fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { - // Distinct counts sample values, not samples or series; all lowering routes - // retain the caller's accuracy requirement, including subquery arguments. - for query in [ - "distinct_over_time(cpu_usage{job=\"worker\"}[5m] offset 1h)", - "distinct_over_time((cpu_usage + 1)[5m:1m])", - "sum by(job)(distinct_over_time(cpu_usage[5m]))", - ] { - for accuracy in [AccuracyTarget::Exact, AccuracyTarget::Epsilon(0.02)] { - let tree = lower_promql(query, accuracy.clone()).unwrap(); - let mut intents = Vec::new(); - collect_intents(&tree, &mut intents); - assert!( - intents.iter().any(|intent| matches!( - intent, AggIntent::Cardinality { accuracy: actual, .. } if actual == &accuracy - )), - "{query}: {tree:?}" - ); - assert!(!intents - .iter() - .any(|intent| matches!(intent, AggIntent::Count { .. }))); - } - } -} - -// ── Bare selectors & label matchers (folded onto Scan.predicates) ─────────────── - -#[test] -fn bare_selector_is_scan_with_predicates() { - let qe = lower(r#"http_requests_total{env="prod",status!="500"}"#); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected TimeRange, got {qe:?}"); - }; - let NonASAPOp::Scan { - source, predicates, .. - } = child.expect_non_asap() - else { - panic!("expected Scan, got {qe:?}"); - }; - assert!(matches!(source, Source::TimeSeries { metric } if metric == "http_requests_total")); - // The converter splits the matcher conjunction into one predicate per - // conjunct on the Scan. - assert_eq!(predicates.len(), 2); - assert!(predicates - .iter() - .all(|p| matches!(&p.0, ScalarExpr::Compare { .. }))); -} - -#[test] -fn regex_matcher_lowers_to_regex_compareop() { - let qe = lower(r#"http_requests_total{path=~"/api/.*"}"#); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected TimeRange, got {qe:?}"); - }; - let NonASAPOp::Scan { - predicates, schema, .. - } = child.expect_non_asap() - else { - panic!("expected Scan, got {qe:?}"); - }; - let ScalarExpr::Compare { - left, op, right, .. - } = &predicates[0].0 - else { - panic!("expected Compare, got {:?}", predicates[0].0); - }; - assert_eq!(*op, CompareOpKind::Regex); - // The label matcher's column is resolved positionally against the scan schema. - let path_id = schema.column_id("path").expect("path in scan schema"); - assert!(matches!(left.as_ref(), ScalarExpr::Column(id) if *id == path_id)); - assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); -} - -// ── *_over_time → Aggregate over TimeRange ────────────────────────────────────── - -#[test] -fn quantile_over_time_is_time_range_aggregate() { - let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate, got {qe:?}"); - }; - assert_eq!(reduction, &Reduction::PerEntity); - assert!( - matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.99).abs() < 1e-9) - ); - let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { - panic!("expected TimeRange child, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(300)); - // The label matcher folded onto the Scan. - assert!( - matches!(child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) - ); -} - -#[test] -fn outer_sum_by_over_quantile_over_time_groups_positionally() { - // `sum by (host) (quantile_over_time(...))`: inner per-series - // quantile-over-time (label-preserving), then an outer cross-series sum - // grouped on a positional `Aggregate.by` — the same shape SQL produces, not - // a name-based Partition. Leaf = [ts, value, host, service] (referenced - // names appended sorted) → host = col 2. - let qe = lower(r#"sum by (host) (quantile_over_time(0.99, latency{service="web"}[5m]))"#); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate grouped by host, got {qe:?}"); - }; - assert_eq!(reduction, &Reduction::by(vec![2])); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - // Inner: Aggregate{Quantile} over TimeRange (per-series over_time reduction). - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected Aggregate (quantile_over_time) under the outer Sum, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Quantile { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -#[test] -fn avg_over_time_maps_to_avg_intent() { - let qe = lower("avg_over_time(cpu_seconds_total[10m])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { - panic!("expected TimeRange child, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(600)); -} - -#[test] -fn stddev_and_stdvar_over_time() { - let qe = lower("stddev_over_time(m[5m])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::StdDev { - population: true, - .. - }] - )); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); - - let qe = lower("stdvar_over_time(m[5m])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::Variance { - population: true, - .. - }] - )); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -#[test] -fn histogram_quantile_wraps_inner_in_quantile() { - // The argument's structure (here `rate`) is preserved *under* the quantile, - // not squashed away. The `_bucket` metric + `le` matcher mark the classic - // form → `HistogramQuantile` over `Aggregate{Rate}` over Scan. - let qe = lower(r#"histogram_quantile(0.95, rate(http_duration_seconds_bucket{le="0.5"}[5m]))"#); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); - }; - assert!( - matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.95).abs() < 1e-9) - ); - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected inner Aggregate{{Rate}}, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let NonASAPOp::TimeRange { - range, - child: tr_child, - .. - } = child.expect_non_asap() - else { - panic!("expected TimeRange under Rate, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(300)); - assert!( - matches!(tr_child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) - ); -} - -#[test] -fn histogram_quantile_over_sum_by_le_preserves_grouping() { - // The canonical Prometheus histogram pattern. Previously returned - // UnsupportedFeature because `extract_matrix` couldn't see through the - // `sum by (le)` aggregate; now the `le` grouping survives into the - // canonical tree. - let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); - }; - // The `by (le)` grouping marks the classic cumulative-bucket form. - assert!( - matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.99).abs() < 1e-9) - ); - // `sum by (le)` survives as a positional Aggregate (by = [2], `le`) over the - // inner Rate — no name-based Partition. - let NonASAPOp::Aggregate { - reduction, - measures, - .. - } = child.expect_non_asap() - else { - panic!("expected `sum by (le)` as a positional Aggregate, got {child:?}"); - }; - assert_eq!(reduction, &Reduction::by(vec![2])); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); -} - -/// The classic `histogram_quantile` aggregate: its `without` keys, `le` -/// column, and output column names. -fn classic_histogram(qe: &OperatorNode) -> (Vec, usize, Vec) { - let NonASAPOp::Aggregate { - reduction: Reduction::Reduce(by), - measures, - .. - } = qe.expect_non_asap() - else { - panic!("expected a reducing Aggregate, got {qe:?}"); - }; - let [AggIntent::HistogramQuantile { le, .. }] = measures.as_slice() else { - panic!("expected HistogramQuantile, got {measures:?}"); - }; - assert!(by.is_without(), "histogram_quantile groups without (le)"); - let names = qe.schema.fields.iter().map(|c| c.name.clone()).collect(); - (by.keys().to_vec(), *le, names) -} - -// A classic histogram_quantile groups `without (le)` and names the child's -// `le` column, even when no matcher or grouping mentions `le`. -#[test] -fn classic_histogram_quantile_groups_without_le() { - let qe = lower("histogram_quantile(0.9, rate(http_duration_seconds_bucket[5m]))"); - let (keys, le, names) = classic_histogram(&qe); - let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { - unreachable!() - }; - let child = &child.schema; - assert_eq!(child.fields[le].name, "le"); - assert_eq!(keys, vec![le]); - assert_eq!(names, vec!["histogram_quantile"]); -} - -// An explicit `sum by (le, job)` argument keeps `job` and drops `le` and the -// renamed sample value from the output labels. -#[test] -fn classic_histogram_quantile_over_sum_by_keeps_other_labels() { - let qe = - lower("histogram_quantile(0.9, sum by (le, job) (rate(http_duration_seconds_bucket[5m])))"); - let (keys, le, names) = classic_histogram(&qe); - // `sum by (le, job)` outputs `[job, le, sum]`. - assert_eq!((keys, le), (vec![1], 1)); - assert_eq!(names, vec!["job", "histogram_quantile"]); -} - -// Out-of-range and NaN quantiles lower unchanged; execution returns -Inf/+Inf/NaN. -#[test] -fn classic_histogram_quantile_keeps_out_of_range_quantiles() { - for (query, expected) in [ - ("histogram_quantile(-1, x_bucket)", -1.), - ("histogram_quantile(2, x_bucket)", 2.), - ] { - let root = lower(query); - let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { - panic!("{query}"); - }; - assert!( - matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if *q == expected) - ); - } - let root = lower("histogram_quantile(NaN, x_bucket)"); - let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { - panic!("NaN"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if q.is_nan())); -} - -// An argument whose closed output lacks `le` has no buckets. Prometheus -// returns an empty vector; lowering rejects it rather than guess a column. -#[test] -fn classic_histogram_quantile_rejects_an_argument_without_le() { - let error = lower_promql( - "histogram_quantile(0.9, sum by (job) (rate(x_bucket[5m])))", - AccuracyTarget::Exact, - ) - .unwrap_err(); - assert!(error.to_string().contains("le"), "{error}"); -} - -// ── rate / increase carry their own window (no Window node) ───────────────────── - -#[test] -fn rate_has_time_range_child_not_window() { - let qe = lower("rate(http_requests_total[5m])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate for rate, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { - panic!("expected TimeRange child (not Window), got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(300)); -} - -#[test] -fn increase_maps_to_increase_intent() { - let qe = lower("increase(errors_total[1h])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate for increase, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { - panic!("expected TimeRange child, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(3600)); -} - -// ── outer aggregation over an inner range-vector func is two levels ───────────── - -#[test] -fn sum_over_rate_keeps_both_levels() { - // Regression: `sum(rate(m[w]))` — the most common PromQL shape — must keep - // the cross-series Sum, not collapse to a bare per-series Rate. - let qe = lower("sum(rate(http_requests_total[5m]))"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected inner Aggregate{{Rate}}, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -#[test] -fn sum_by_over_rate_groups_the_outer_sum() { - // `sum by (job) (rate(...))`: the grouping belongs to the OUTER sum and lands - // on a positional `Aggregate.by` (the same shape SQL produces) over the - // label-preserving inner Rate. Leaf = [ts, value, job] → by = [2]. - let qe = lower("sum by (job) (rate(http_requests_total[5m]))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate grouped by job, got {qe:?}"); - }; - assert_eq!(reduction, &Reduction::by(vec![2])); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) - )); -} - -#[test] -fn count_over_rate_keeps_both_levels() { - // The `Outer::Count` sibling of the `sum(rate(...))` bug. - let qe = lower("count(rate(http_requests_total[5m]))"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate{{Count}}, got {qe:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) - )); -} - -#[test] -fn count_over_distinct_over_time_preserves_both_aggregates() { - // One series with window samples [1, 2] produces one distinct-count - // result (value 2). The outer count counts that one series, yielding 1. - for (query, reduction) in [ - ( - "count(distinct_over_time(unique_users[5m]))", - Reduction::by(vec![]), - ), - ( - "count by (job) (distinct_over_time(unique_users[5m]))", - Reduction::by(vec![2]), - ), - ] { - let tree = lower(query); - let NonASAPOp::Aggregate { - measures, - reduction: actual, - child, - .. - } = tree.expect_non_asap() - else { - panic!("expected outer Count: {tree:?}"); - }; - assert!( - matches!(measures.as_slice(), [AggIntent::Count { .. }]), - "{query}: {tree:?}" - ); - assert_eq!(actual, &reduction, "{query}"); - let NonASAPOp::Aggregate { - measures, - reduction, - child, - .. - } = child.expect_non_asap() - else { - panic!("expected inner per-series Cardinality: {tree:?}"); - }; - assert!( - matches!(measures.as_slice(), [AggIntent::Cardinality { .. }]), - "{query}: {tree:?}" - ); - assert_eq!(reduction, &Reduction::PerEntity, "{query}"); - assert!( - matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, .. } if range.as_secs() == 300) - ); - } -} - -// ── count / cardinality ─────────────────────────────────────────────────────── - -// Both selector fast paths and recursive vector expressions count rows, not values. -#[test] -fn count_never_lowers_to_distinct_sample_values() { - for query in [ - "count(up)", - "count by (job) (up)", - "count without (instance) (up)", - "count(up + 1)", - "count(count_over_time(up[5m]))", - "count_over_time(up[5m])", - ] { - let tree = lower(query); - let intents = all_intents(&tree); - assert!( - intents.iter().any(|i| matches!(i, AggIntent::Count { .. })), - "{query}: {tree:?}" - ); - assert!( - !intents - .iter() - .any(|i| matches!(i, AggIntent::Cardinality { .. })), - "{query}: {tree:?}" - ); - } -} - -#[test] -fn count_over_time_is_count_intent() { - let qe = lower("count_over_time(m[5m])"); - let NonASAPOp::Aggregate { - measures, child, .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -#[test] -fn outer_count_counts_series() { - // `count by (symbol) (count_over_time(...))`: inner per-series sample count - // over the window (label-preserving), outer cross-series row count grouped - // on a positional `Aggregate.by`. Leaf = [ts, value, symbol] → symbol = col 2. - let qe = lower("count by (symbol) (count_over_time(financial_last_trade_price[5m]))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected outer Aggregate grouped by symbol, got {qe:?}"); - }; - assert_eq!(reduction, &Reduction::by(vec![2])); - assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - // Inner: Aggregate{Count} over TimeRange (per-series count_over_time). - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected Aggregate (count_over_time) under the outer count, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -// ── topk / bottomk ──────────────────────────────────────────────────────────── - -#[test] -fn topk_over_count_is_heavy_hitter_topk() { - let qe = lower(r#"topk by (service) (10, count_over_time(requests{env="prod"}[1m]))"#); - // Heavy-hitter: Aggregate{TopK} with grouping resolved to positional ids. - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate with TopK, got {qe:?}"); - }; - // `service` is the only group key → resolved to a positional ColumnId. - assert_eq!(reduction.expect_reduce().len(), 1); - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 10, .. }] - )); - // The count_over_time under the TopK is a TimeRange-backed aggregate. - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected Aggregate (count_over_time) under TopK, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { - panic!("expected TimeRange under Count aggregate, got {child:?}"); - }; - assert_eq!(*range, Duration::from_secs(60)); - assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); -} - -#[test] -fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { - let qe = lower(r#"topk by (service) (5, sum_over_time(requests{env="prod"}[1m]))"#); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate with TopK, got {qe:?}"); - }; - assert_eq!(reduction.expect_reduce().len(), 1); - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 5, .. }] - )); - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected Aggregate (sum_over_time) under TopK, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -#[test] -fn topk_over_avg_is_generic_sort_limit() { - let qe = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let NonASAPOp::Limit { - n: Some(n), - offset, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected Limit, got {qe:?}"); - }; - assert_eq!(*n, 5); - assert_eq!(*offset, 0); - let NonASAPOp::Sort { - keys, - partition_by, - child, - } = child.expect_non_asap() - else { - panic!("expected Sort under Limit, got {child:?}"); - }; - assert_eq!(keys.len(), 1); - assert!(!keys[0].ascending, "topk ranks descending"); - // `by (host)` is per-group ranking → it rides on `Sort.partition_by` - // (positional), not a `Partition` node (issue #12). `host` is col 2 in - // the per-series avg schema [ts, value, host]. - assert_eq!(partition_by, &vec![2]); - // Underneath: the label-preserving windowed avg aggregate (by: []), no - // intervening Partition. - assert!( - matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, measures, .. } - if reduction == &Reduction::PerEntity && matches!(measures.as_slice(), [AggIntent::Avg { .. }])), - "expected bare per-series Avg aggregate under Sort, got {child:?}" - ); -} - -#[test] -fn ungrouped_topk_over_sum_is_heavy_hitter() { - let qe = lower("topk(5, sum_over_time(m[5m]))"); - assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); - assert!(has_intent(&qe, |i| matches!(i, AggIntent::Sum { .. }))); - assert!(has_intent(&qe, |i| matches!( - i, - AggIntent::TopK { k: 5, .. } - ))); -} - -#[test] -fn bottomk_over_count_is_generic_sort_ascending() { - // `bottomk` is never a heavy-hitter (descending=false), even over count. - let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let NonASAPOp::Limit { - n: Some(n), child, .. - } = qe.expect_non_asap() - else { - panic!("expected Limit, got {qe:?}"); - }; - assert_eq!(*n, 3); - let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { - panic!("expected Sort"); - }; - assert!(keys[0].ascending, "bottomk ranks ascending"); - // Count intent is still present (as the inner aggregate), no TopK. - assert!(has_intent(&qe, |i| matches!(i, AggIntent::Count { .. }))); - assert!(!has_intent(&qe, |i| matches!(i, AggIntent::TopK { .. }))); -} - -#[test] -fn bottomk_is_always_generic_sort_ascending() { - let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let NonASAPOp::Limit { - n: Some(n), child, .. - } = qe.expect_non_asap() - else { - panic!("expected Limit, got {qe:?}"); - }; - assert_eq!(*n, 3); - let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { - panic!("expected Sort"); - }; - assert!(keys[0].ascending, "bottomk ranks ascending"); -} - -#[test] -fn topk_count_output_schema_carries_group_key() { - // The inner Count is per-series (label-preserving), so the group-by key - // (`service`) flows through to the outer TopK's `by` column. Leaf schema = - // [ts, value, service] → TopK groups on service (col 2). - let qe = lower("topk by (service) (5, count_over_time(m[1m]))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected Aggregate{{TopK}}, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::by(vec![2]), - "service is col 2 in [ts, value, service]" - ); - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 5, .. }] - )); - // Inner Count aggregate is visible with its TimeRange child. - let NonASAPOp::Aggregate { - measures, child, .. - } = child.expect_non_asap() - else { - panic!("expected inner Aggregate{{Count}}, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!( - child.expect_non_asap(), - NonASAPOp::TimeRange { .. } - )); -} - -// ── binary ops ──────────────────────────────────────────────────────────────── - -#[test] -fn binary_op_division() { - let qe = lower("rate(a[5m]) / rate(b[5m])"); - let NonASAPOp::BinaryOp { - operator: BinaryOperator { kind: op, .. }, - lhs, - rhs, - .. - } = qe.expect_non_asap() - else { - panic!("expected BinaryOp, got {qe:?}"); - }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); - assert!( - matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) - ); - assert!( - matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) - ); -} - -#[test] -fn binary_op_with_on_grouping() { - let qe = lower("a / on(host) b"); - let NonASAPOp::BinaryOp { - operator: BinaryOperator { vector_match, .. }, - .. - } = qe.expect_non_asap() - else { - panic!("expected BinaryOp, got {qe:?}"); - }; - let vm = vector_match.as_ref().expect("vector_match present"); - use asap_types::pre_asap::VectorMatchKind; - assert_eq!(vm.kind, VectorMatchKind::On); - assert_eq!(vm.labels, vec!["host".to_string()]); -} - -// `bool` changes a comparison from a filter to a 0/1 result, so the IR must -// carry it. -#[test] -fn bool_comparisons_are_distinct() { - let op = |q: &str| match lower(q).expect_non_asap() { - NonASAPOp::BinaryOp { - operator, - return_bool, - .. - } => (operator.kind.clone(), *return_bool), - NonASAPOp::Filter { - pred: asap_types::ir::Predicate(ScalarExpr::Compare { op, .. }), - .. - } => (BinaryOpKind::Compare(op.clone()), false), - NonASAPOp::Project { cols, .. } => { - let ScalarExpr::Case { branches, .. } = &cols[1].expr else { - panic!() - }; - let ScalarExpr::Compare { op, .. } = &branches[0].0 else { - panic!() - }; - (BinaryOpKind::Compare(op.clone()), true) - } - other => panic!("expected BinaryOp, got {other:?}"), - }; - assert_eq!( - op("a > 1"), - (BinaryOpKind::Compare(CompareOpKind::Gt), false) - ); - assert_eq!( - op("a > bool 1"), - (BinaryOpKind::Compare(CompareOpKind::Gt), true) - ); - assert_eq!( - op("a == bool on(job) b"), - (BinaryOpKind::Compare(CompareOpKind::Eq), true) - ); -} - -#[test] -fn binary_op_binds_each_branch_against_its_own_schema() { - // Each side scans a different metric and groups by a different label. With a - // single root schema threaded to both branches, the left scan would leak the - // right's group key (and vice-versa). Per-branch binding keeps them separate. - let qe = lower("count by (job) (a) / count by (region) (b)"); - let NonASAPOp::BinaryOp { lhs, rhs, .. } = qe.expect_non_asap() else { - panic!("expected BinaryOp, got {qe:?}"); - }; - let lcols = scan_columns(lhs); - let rcols = scan_columns(rhs); - assert!( - lcols.iter().any(|c| c == "job") && !lcols.iter().any(|c| c == "region"), - "lhs scan schema leaked the rhs key: {lcols:?}" - ); - assert!( - rcols.iter().any(|c| c == "region") && !rcols.iter().any(|c| c == "job"), - "rhs scan schema leaked the lhs key: {rcols:?}" - ); -} - -/// Collect every `AggIntent` in the tree, root-to-leaf. -fn all_intents(e: &OperatorNode) -> Vec { - let mut out = Vec::new(); - collect_intents(e, &mut out); - out -} - -fn collect_intents(e: &OperatorNode, out: &mut Vec) { - match e.expect_non_asap() { - NonASAPOp::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - collect_intents(child, out); - } - NonASAPOp::TimeRange { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } => collect_intents(child, out), - NonASAPOp::BinaryOp { lhs, rhs, .. } => { - collect_intents(lhs, out); - collect_intents(rhs, out); - } - _ => {} - } -} - -/// True if any `AggIntent` anywhere in the tree satisfies `pred`. -fn has_intent bool>(e: &OperatorNode, pred: F) -> bool { - all_intents(e).iter().any(pred) -} - -/// Field names on the first `Scan` reachable by descending single-child nodes. -fn scan_columns(e: &OperatorNode) -> Vec { - match e.expect_non_asap() { - NonASAPOp::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), - NonASAPOp::Aggregate { child, .. } - | NonASAPOp::TimeRange { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } => scan_columns(child), - _ => vec![], - } -} - -// ── without(...) grouping (issue #39) ─────────────────────────────────────────── - -#[test] -fn without_grouping_lowers_to_the_exclusion_form() { - // `sum without (instance) (rate(m[5m]))` — a cross-series reduction over the - // per-series rate, grouped by every label except `instance`. The excluded - // label is stored positionally (the SchemaResolver seeds it), the grouping is the - // `without` form, and the output schema stays open. - let qe = lower("sum without (instance) (rate(m[5m]))"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = qe.expect_non_asap() - else { - panic!("expected an Aggregate, got {qe:?}"); - }; - let by = reduction.expect_reduce(); - assert!(by.is_without()); - assert_eq!(by.keys().len(), 1, "excluded `instance`"); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - // The inner per-series rate is preserved (label-preserving) under the outer - // cross-series `without` reduction. - assert!( - matches!(child.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Rate])) - ); - assert!(!qe.schema.clone().closed); -} - -// ── parameter validation (reject rather than silently truncate/garble) ────────── - -#[test] -fn fractional_or_negative_topk_k_is_rejected() { - // `as u64` would silently truncate 2.7→2 / saturate -1→0. - assert!(lower_promql("topk(2.7, count_over_time(m[1m]))", AccuracyTarget::Exact).is_err()); - assert!(lower_promql("bottomk(2.5, sum_over_time(m[1m]))", AccuracyTarget::Exact).is_err()); -} - -#[test] -fn out_of_range_quantile_phi_is_accepted() { - // Prometheus defines out-of-range phi results; lowering must preserve it. - for query in [ - "quantile(1.5, up)", - "quantile_over_time(1.5, m[5m])", - "histogram_quantile(2.0, rate(b_bucket[5m]))", - ] { - assert!( - lower_promql(query, AccuracyTarget::Exact).is_ok(), - "{query}" - ); - } -} - -#[test] -fn function_wrapped_range_vector_is_rejected_not_stripped() { - // `rate(abs(m[5m]))` must NOT silently lower as `rate(m[5m])` — the wrapper - // is rejected (here, at parse or in extract_matrix), never stripped. - assert!( - lower_promql("rate(abs(http_requests_total[5m]))", AccuracyTarget::Exact).is_err(), - "function-wrapped range vector should be rejected" - ); -} - -#[test] -fn pathologically_nested_query_is_rejected_not_stack_overflow() { - // 300 nested parens parse fine but exceed the walker's depth limit (256); - // it must return an error, not overflow the stack. - let q = format!("{}m{}", "(".repeat(300), ")".repeat(300)); - let err = lower_promql(&q, AccuracyTarget::Exact).unwrap_err(); - assert!(format!("{err}").contains("nesting"), "got {err}"); -} - -// Behavior: every parser-accepted `fill` modifier form is rejected with a -// fill-specific lowering error rather than silently dropped. -#[test] -fn fill_modifiers_are_rejected_not_ignored() { - for q in [ - "a + fill(0) b", - "a + fill_left(1) b", - "a + fill_right(2) b", - "a + fill_left(1) fill_right(2) b", - "a + fill_right(2) fill_left(1) b", - "a + on(job) fill(0) b", - "a * ignoring(instance) group_left(env) fill_right(0) b", - "a > bool on(job) fill(0) b", - "sum(a - on(job) group_right fill_left(0) b)", - ] { - match lower_promql(q, AccuracyTarget::Exact) { - Err(LoweringError::UnsupportedFeature(m)) if m.contains("`fill`") => {} - other => panic!("expected fill rejection for {q:?}, got {other:?}"), - } - } -} - -// ── accuracy propagation ────────────────────────────────────────────────────── - -#[test] -fn accuracy_target_flows_into_quantile_intent() { - let qe = lower_promql( - "quantile_over_time(0.9, m[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(); - let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { - panic!("expected Aggregate"); - }; - assert!(matches!( - &measures[0], - AggIntent::Quantile { accuracy: AccuracyTarget::Epsilon(e), .. } if (*e - 0.01).abs() < 1e-12 - )); -} - -// ── schema flow (positional, carried on Scan; derived on demand) ───────────────── - -#[test] -fn aggregate_output_schema_preserves_time_axis_and_labels() { - let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); - // Per-series reduction: the root is Aggregate { TimeRange { Scan } }. - // The SchemaResolver adds all referenced label names (group keys AND filter - // predicate columns) to the scan schema, so `env` appears as a column - // even though it is only used as a filter. - // per_series_reduction_schema preserves the time axis and all label columns. - let NonASAPOp::Aggregate { .. } = qe.expect_non_asap() else { - panic!("expected Aggregate, got {qe:?}"); - }; - let schema = &qe.schema; - let names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); - assert_eq!(names, vec!["ts", "value", "env"]); - assert_eq!( - schema.time_index, - Some(0), - "per-series over_time preserves the time axis" - ); -} - -#[test] -fn scan_schema_carries_ts_value_and_group_keys() { - // `service` is a group key → the SchemaResolver lands it in the self-contained - // Scan schema (positional). `env` is only a filter, so it is not a column. - let qe = lower("count by (service) (count_over_time(requests[1m]))"); - fn find_scan(n: &OperatorNode) -> &OperatorNode { - match n.expect_non_asap() { - NonASAPOp::Scan { .. } => n, - NonASAPOp::TimeRange { child, .. } - | NonASAPOp::Aggregate { child, .. } - | NonASAPOp::Filter { child, .. } => find_scan(child), - other => panic!("unexpected node {other:?}"), - } - } - let NonASAPOp::Scan { schema, .. } = find_scan(&qe).expect_non_asap() else { - unreachable!() - }; - let mut names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); - names.sort(); - assert_eq!(names, vec!["service", "ts", "value"]); - assert_eq!(schema.time_index, Some(0)); // ts -} - -// ── batch entry point ───────────────────────────────────────────────────────── - -#[test] -fn batch_lowers_each_entry_and_reads_per_query_accuracy() { - let workload = PlanningWorkload { - query_workload: QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: Some(vec![ - BatchEntry { - query: Query("rate(a[5m])".into()), - requirements: QueryRequirements::default(), - predictability: Predictability::Unknown, - invocations: 1, - execute_at: None, - time_selection: TimeSelection::default(), - }, - BatchEntry { - query: Query("quantile_over_time(0.9, b[5m])".into()), - requirements: QueryRequirements { - accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Epsilon(0.02)), - ..Default::default() - }, - predictability: Predictability::Unknown, - invocations: 1, - execute_at: None, - time_selection: TimeSelection::default(), - }, - ]), - repeating_queries: None, - }, - data_workload: Some(DataWorkload { - data_ingestion_interval: Evidence { - value: Some(DurationMs(1_000)), - ..Default::default() - }, - ..Default::default() - }), - }; - let results = lower_promql_workload(&workload, 0).expect("valid workload"); - assert_eq!(results.len(), 2); -} - -#[test] -fn batch_rejects_non_promql_language() { - use asap_types::workload::SqlDialect; - let workload = PlanningWorkload { - query_workload: QueryWorkload { - language: QueryLanguage::SQL(SqlDialect::DataFusionSQL), - query_batch: Some(vec![BatchEntry { - query: Query("SELECT 1".into()), - requirements: QueryRequirements::default(), - predictability: Predictability::Unknown, - invocations: 1, - execute_at: None, - time_selection: TimeSelection::default(), - }]), - repeating_queries: None, - }, - data_workload: None, - }; - assert!(matches!( - lower_promql_workload(&workload, 0), - Err(LoweringError::WrongLanguage(_)) - )); -} - -// ── #12: one home per grouping concept (the canonical `Partition` node is removed) ── -// -// `Partition` and `Aggregate.by` were two ways to express grouping. #12 collapses -// them: a reducing GROUP BY → `Aggregate.by`; per-group *ranking* (split without -// reduce) → `Sort.partition_by`; parallel sharding → a deployment's own -// physical stage. There is no longer a canonical `Partition` node. These -// tests pin both surviving canonical homes. - -#[test] -fn reducing_group_by_lowers_to_aggregate_by() { - // Cross-series reduce, no keys → bare `Aggregate { reduction: Reduce([]) }`. - let q = lower("sum(http_requests_total)"); - assert!( - matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::by(vec![])) - ); - - // Cross-series reduce grouped by a label → `Aggregate.reduction`. - let q = lower("sum by (job) (http_requests_total)"); - assert!( - matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } - if reduction.expect_reduce().len() == 1) - ); - - // Reduce over a label-preserving `rate` grouped by a label → still - // `Aggregate.reduction` (the keys resolve against rate's preserved schema). - let q = lower("sum by (job) (rate(http_requests_total[5m]))"); - assert!( - matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } - if reduction.expect_reduce().len() == 1) - ); -} - -#[test] -fn generic_topk_grouping_lowers_to_sort_partition_by() { - // Per-group ranking (`topk by (host)`, non-heavy-hitter) groups *without* - // reducing → the grouping rides on `Sort.partition_by`, and the windowed - // reduction beneath stays label-preserving (`by: []`). No `Partition` node. - let q = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { - panic!("expected Limit, got {q:?}"); - }; - let NonASAPOp::Sort { - partition_by, - child, - .. - } = child.expect_non_asap() - else { - panic!("expected Sort, got {child:?}"); - }; - assert_eq!(partition_by, &vec![2], "host is col 2 in [ts, value, host]"); - assert!( - matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) - ); -} - -#[test] -fn topk_over_bare_selector_by_label_ranks_per_group() { - // `topk(3, http_requests_total) by (job)` — top-3 series per `job`. A bare - // instant selector ranks its OWN samples; it must not be wrapped in an - // implicit cross-series `Sum`, which would collapse the `job` partition - // label before `Sort.partition_by` resolves it (issue #30 — follow-up to the - // Partition→Sort.partition_by reframe in #12). Expected: - // Limit{3} → Sort{value desc, partition_by:[job]} → Scan - let q = lower("topk(3, http_requests_total) by (job)"); - let NonASAPOp::Limit { - n: Some(n), child, .. - } = q.expect_non_asap() - else { - panic!("expected Limit, got {q:?}"); - }; - assert_eq!(*n, 3); - let NonASAPOp::Sort { - keys, - partition_by, - child, - } = child.expect_non_asap() - else { - panic!("expected Sort, got {child:?}"); - }; - assert!(!keys[0].ascending, "topk ranks descending"); - assert_eq!(partition_by, &vec![2], "job is col 2 in [ts, value, job]"); - // No implicit reducing aggregate — the selector is label-preserving, so the - // sort is directly over the selector horizon (the `job` label survives to partition by). - assert!( - matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })), - "ranking is over the bare selector horizon, not a reducing Aggregate, got {child:?}" - ); - assert!( - !has_intent(&q, |i| matches!(i, AggIntent::Sum { .. })), - "no implicit Sum is introduced over a bare selector" - ); -} - -#[test] -fn topk_over_bare_selector_ranks_raw_samples() { - // Even without `by`, `topk(3, m)` ranks the raw instant-vector samples — it - // does not sum them. The sort sits directly over the Scan, partition empty. - let q = lower("topk(3, http_requests_total)"); - let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { - panic!("expected Limit, got {q:?}"); - }; - let NonASAPOp::Sort { - partition_by, - child, - .. - } = child.expect_non_asap() - else { - panic!("expected Sort, got {child:?}"); - }; - assert!(partition_by.is_empty(), "no `by` → global ranking"); - assert!( - matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) - ); - assert!(!has_intent(&q, |i| matches!(i, AggIntent::Sum { .. }))); -} - -// ── Issue #109: histogram_quantiles fans out into one branch per φ ────────── - -/// The `(label value, intent)` of each `histogram_quantiles` branch. -fn quantile_branches(q: &OperatorNode) -> Vec<(String, AggIntent)> { - let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { - panic!("expected a Concat at the root, got {q:?}"); - }; - children - .iter() - .map(|c| { - let NonASAPOp::PromqlRelabel { value, child, .. } = c.expect_non_asap() else { - panic!("expected PromqlRelabel per branch, got {c:?}"); - }; - let ScalarExpr::Literal(ScalarValue::Utf8(v)) = value else { - panic!("expected a literal label value, got {value:?}"); - }; - let NonASAPOp::Aggregate { measures, .. } = child.expect_non_asap() else { - panic!("expected an Aggregate under the PromqlRelabel, got {child:?}"); - }; - (v.clone(), measures[0].clone()) - }) - .collect() -} - -#[test] -fn histogram_quantiles_rejects_unrepresented_native_histograms() { - assert!(lower_promql( - r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#, - AccuracyTarget::Exact - ) - .is_err()); -} - -#[test] -fn histogram_quantiles_over_classic_buckets_interpolates() { - // `_bucket` argument → exact cumulative-bucket interpolation, never a sketch. - let q = lower(r#"histogram_quantiles(request_duration_seconds_bucket, "q", 0.5, 0.9)"#); - for (_, intent) in quantile_branches(&q) { - assert!( - matches!(intent, AggIntent::HistogramQuantile { .. }), - "classic buckets → HistogramQuantile, got {intent:?}" - ); - } -} - -#[test] -fn histogram_quantiles_branches_are_union_compatible() { - // `Concat` derives its schema from the first child, so every branch must - // agree on column names — the φ lives in the label, not the column name. - let q = lower(r#"histogram_quantiles(testhistogram3_bucket, "q", 0.5, 0.9)"#); - let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { - panic!("expected Concat"); - }; - let shapes: Vec> = children - .iter() - .map(|c| { - c.schema - .clone() - .fields - .iter() - .map(|c| c.name.clone()) - .collect() - }) - .collect(); - assert_eq!(shapes[0], shapes[1], "branches must be union-compatible"); - assert_eq!(shapes[0], vec!["value".to_string(), "q".to_string()]); - assert_eq!( - q.schema.fields.len(), - 2, - "the merged schema describes every branch" - ); -} - -#[test] -fn histogram_quantiles_uses_the_given_label_name() { - let q = lower(r#"histogram_quantiles(h_bucket, "phi", 0.5)"#); - let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { - panic!("expected Concat"); - }; - let NonASAPOp::PromqlRelabel { dst, .. } = children[0].expect_non_asap() else { - panic!("expected PromqlRelabel"); - }; - assert_eq!(dst, "phi"); -} - -#[test] -fn histogram_quantiles_formats_small_quantiles_like_prometheus() { - // `labels.FormatOpenMetricsFloat`: Go's %g, so exponent form below 1e-4. - let q = lower(r#"histogram_quantiles(h_bucket, "q", 0.00001)"#); - assert_eq!(quantile_branches(&q)[0].0, "1e-05"); -} - -#[test] -fn histogram_quantiles_rejects_an_out_of_range_quantile() { - // Same rule as `histogram_quantile(φ, …)` — one bad φ fails the whole call. - for q in [ - r#"histogram_quantiles(h_bucket, "q", -0.1)"#, - r#"histogram_quantiles(h_bucket, "q", 1.01)"#, - r#"histogram_quantiles(h_bucket, "q", 0.5, NaN)"#, - ] { - assert!( - lower_promql(q, AccuracyTarget::Exact).is_err(), - "{q} should be rejected" - ); - } -} - -// ── TimeRange.kind: instant vs range selectors ────────────────────────────────── - -#[test] -fn bare_instant_selector_is_an_instant_time_range() { - // `up` reads the latest sample per series within the workload's ingestion - // interval (1s in `support::workload`): an `Instant` lookback of that length. - let qe = lower("up"); - let NonASAPOp::TimeRange { range, kind, child } = qe.expect_non_asap() else { - panic!("expected TimeRange, got {qe:?}"); - }; - assert_eq!(*kind, TimeRangeKind::Instant); - assert_eq!(*range, Duration::from_secs(1)); - assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); -} - -#[test] -fn explicit_range_selector_is_a_range_time_range() { - // `m[5m]` keeps its own window and is a `Range` selection — both under a - // range function and as a bare matrix selector. - let qe = lower("rate(m[5m])"); - let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { - panic!("expected Aggregate, got {qe:?}"); - }; - let NonASAPOp::TimeRange { range, kind, .. } = child.expect_non_asap() else { - panic!("expected TimeRange, got {child:?}"); - }; - assert_eq!(*kind, TimeRangeKind::Range); - assert_eq!(*range, Duration::from_secs(300)); - - let qe = lower("m[5m]"); - assert!(matches!( - qe.expect_non_asap(), - NonASAPOp::TimeRange { - kind: TimeRangeKind::Range, - .. - } - )); -} - -#[test] -fn instant_and_range_selectors_of_equal_length_stay_distinct() { - // The kind is part of the shape: a 1s range selector is not the same tree as - // the 1s instant lookback injected around a bare selector. - assert_ne!(lower("up"), lower("up[1s]")); -} - -// ── the `bool` modifier → `return_bool` ───────────────────────────────────────── - -#[test] -fn vector_scalar_comparison_without_bool_filters() { - let qe = lower("up > 0"); - assert!(matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. })); - assert!(matches!( - support::sample_expression(&qe), - ScalarExpr::Compare { - op: CompareOpKind::Gt, - .. - } - )); -} - -#[test] -fn vector_scalar_comparison_with_bool_sets_return_bool() { - let qe = lower("up > bool 0"); - assert!(matches!( - support::sample_expression(&qe), - ScalarExpr::Case { .. } - )); - assert_ne!(qe, lower("up > 0")); -} - -#[test] -fn vector_vector_comparison_with_bool_sets_return_bool() { - // `a > bool b` — the modifier lands on the vector/vector op itself, with - // the default (ignoring nothing) match. - let qe = lower("a > bool b"); - let NonASAPOp::BinaryOp { - operator, - return_bool, - lhs, - rhs, - } = qe.expect_non_asap() - else { - panic!("expected BinaryOp, got {qe:?}"); - }; - assert!(*return_bool); - assert_eq!(operator.kind, BinaryOpKind::Compare(CompareOpKind::Gt)); - assert!(matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); - assert!(matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); - assert!(!lower("a > b").expect_non_asap().children().is_empty()); - assert_ne!(qe, lower("a > b")); -} - -#[test] -fn bool_modifier_composes_with_vector_matching() { - let qe = lower("a > bool on(job) b"); - let NonASAPOp::BinaryOp { - operator, - return_bool, - .. - } = qe.expect_non_asap() - else { - panic!("expected BinaryOp, got {qe:?}"); - }; - assert!(*return_bool); - let vm = operator.vector_match.as_ref().expect("on(job) present"); - assert_eq!(vm.labels, vec!["job".to_string()]); -} - -// ── scalar expressions: negation, arithmetic, comparison ──────────────────────── - -#[test] -fn scalar_negation_of_time_is_a_negative_expression() { - // `-time()` is a scalar expression; its negation stays structural (the - // operand is not a constant to fold) and follows PromQL numeric rules. - let qe = support::scalar_root("-time()"); - let ScalarExpr::Negative { expr, semantics } = &qe else { - panic!("expected ScalarExpr(Negative), got {qe:?}"); - }; - assert_eq!(*semantics, ExprSemantics::Promql); - assert!(matches!(expr.as_ref(), ScalarExpr::EvalTimestamp)); - // Scalar-shaped: no time index. -} - -#[test] -fn scalar_negation_of_a_constant_still_folds() { - // `-(2)` is constant: it folds to one literal rather than a `Negative`. - assert_eq!( - support::promql_scalar(&support::scalar_root("-(2)")), - Some(-2.0) - ); -} - -#[test] -fn scalar_arithmetic_carries_promql_semantics() { - let qe = support::scalar_root("time() - 1"); - let ScalarExpr::Arithmetic { - op, - left, - right, - semantics, - } = &qe - else { - panic!("expected scalar(Arithmetic), got {qe:?}"); - }; - assert_eq!(*op, ArithmeticOpKind::Sub); - assert_eq!(*semantics, ExprSemantics::Promql); - assert!(matches!(left.as_ref(), ScalarExpr::EvalTimestamp)); - assert!(matches!( - right.as_ref(), - ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0 - )); -} - -#[test] -fn scalar_bool_comparison_is_a_zero_one_case_with_promql_semantics() { - // `1 < bool 2` → `Case(Compare(1 < 2) → 1.0, else 0.0)`: PromQL yields 0/1. - let qe = support::scalar_root("1 < bool 2"); - let ScalarExpr::Case { - operand, - branches, - else_expr, - } = &qe - else { - panic!("expected scalar(Case), got {qe:?}"); - }; - assert!(operand.is_none()); - let [(when, then)] = branches.as_slice() else { - panic!("expected one branch, got {branches:?}"); - }; - let ScalarExpr::Compare { - left, - op, - right, - semantics, - } = when - else { - panic!("expected a Compare condition, got {when:?}"); - }; - assert_eq!(*op, CompareOpKind::Lt); - assert_eq!(*semantics, ExprSemantics::Promql); - assert!(matches!(left.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); - assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 2.0)); - assert!(matches!(then, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); - assert!(matches!( - else_expr.as_deref(), - Some(ScalarExpr::Literal(ScalarValue::Float64(v))) if *v == 0.0 - )); -} - -#[test] -fn scalar_comparison_without_bool_is_rejected() { - // PromQL has no scalar filter: a scalar/scalar comparison needs `bool`. - for q in ["1 < 2", "time() > 0", "(1 + 1) == 2"] { - assert!( - lower_promql(q, AccuracyTarget::Exact).is_err(), - "{q} must be rejected without `bool`" - ); - } -} - -#[test] -fn label_matcher_predicates_carry_promql_semantics() { - let qe = lower(r#"up{job="api"}"#); - let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { - panic!("expected TimeRange, got {qe:?}"); - }; - let NonASAPOp::Scan { predicates, .. } = child.expect_non_asap() else { - panic!("expected Scan, got {child:?}"); - }; - assert!(matches!( - &predicates[0].0, - ScalarExpr::Compare { - semantics: ExprSemantics::Promql, - .. - } - )); -} diff --git a/crates/frontend-promql/tests/unified_support.rs b/crates/frontend-promql/tests/unified_support.rs deleted file mode 100644 index dee692ca1..000000000 --- a/crates/frontend-promql/tests/unified_support.rs +++ /dev/null @@ -1,95 +0,0 @@ -use std::rc::Rc; - -use asap_frontend_promql::unified::{ - lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, -}; -use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; -use asap_types::pre_asap::ScalarValue; -use asap_types::types::AccuracyTarget; -use asap_types::workload::{ - AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, - Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, -}; - -pub fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { - PlanningWorkload { - query_workload: QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: Some(vec![BatchEntry { - query: Query(query.into()), - requirements: QueryRequirements { - accuracy: AccuracyRequirement::Explicit(accuracy), - ..Default::default() - }, - predictability: Predictability::Unknown, - invocations: 1, - execute_at: None, - time_selection: TimeSelection::default(), - }]), - repeating_queries: None, - }, - data_workload: Some(DataWorkload { - data_ingestion_interval: Evidence { - value: Some(DurationMs(1_000)), - ..Default::default() - }, - ..Default::default() - }), - } -} - -#[allow(dead_code)] -pub fn lower_promql( - query: &str, - accuracy: AccuracyTarget, -) -> Result, PromqlError> { - let mut lowered = lower_promql_workload(&workload(query, accuracy), 0)?; - Ok(lowered.remove(0)) -} - -#[allow(dead_code)] -pub fn lower_promql_with_histograms( - query: &str, - accuracy: AccuracyTarget, - histograms: HistogramCatalog, -) -> Result, PromqlError> { - let mut lowered = - lower_promql_workload_with_histograms(&workload(query, accuracy), histograms, 0)?; - Ok(lowered.remove(0)) -} - -/// The value of a bare PromQL numeric literal / folded constant at an -/// scalar position (`Literal(Float64(v))`); `None` for any -/// other shape. -#[allow(dead_code)] -pub fn promql_scalar(node: &ScalarExpr) -> Option { - match node { - ScalarExpr::Literal(ScalarValue::Float64(v)) => Some(*v), - _ => None, - } -} - -#[allow(dead_code)] -pub fn scalar_root(query: &str) -> ScalarExpr { - match asap_frontend_promql::unified::lower_promql_query_workload( - &workload(query, AccuracyTarget::Exact), - 0, - ) - .unwrap() - .remove(0) - { - asap_types::ir::QueryRoot::Scalar(expr) => expr, - _ => panic!("expected scalar root: {query}"), - } -} - -#[allow(dead_code)] -pub fn sample_expression(node: &OperatorNode) -> &ScalarExpr { - match node.expect_non_asap() { - NonASAPOp::Project { cols, .. } => { - &cols[node.schema.column_id("value").unwrap_or(cols.len() - 1)].expr - } - NonASAPOp::Filter { pred, .. } => &pred.0, - other => panic!("expected sample expression, got {other:?}"), - } -} diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index 2b4891a08..12a0a53b5 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -5,15 +5,16 @@ use asap_aware_mapping::accuracy::{ }; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_aware_mapping::replacement::{default_strategies, search_workload_with_targets}; -use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; +use asap_aware_mapping::{ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG}; mod support; +use asap_types::ir::cse::share_common_sub_dags; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; use asap_types::post_asap::{ - compile_post_asap_dag, cse::share_common_summary_sub_dags, AccuracyError, BoundExpr, - CompositionOperator, ErrorMetric, FieldDataType, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SketchStatistic, SummaryExpr, SummaryInputExpr, SummaryNode, + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, FieldDataType, ProbabilityExpr, + ResultGuarantee, SketchAlgorithm, SketchStatistic, SummaryInputExpr, }; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, post_asap_dag}; // Synthetic evidence exercises structural sharing, never runtime accuracy. struct TestEvidence; @@ -49,22 +50,22 @@ impl AccuracyModel for TestEvidence { } } -fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { +fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { let root = lower_promql(query, accuracy).unwrap(); - SketchAlgorithmStrategy::new_with_planning_inputs(&DefaultCostModel, &TestEvidence, &EqualSplitAllocator) - .replacements(&TargetSubDAG::new(&Rc::new(root))) + ASAPStrategies::new_with_planning_inputs(&DefaultCostModel, &TestEvidence, &EqualSplitAllocator) + .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| { - let Replacement::Summary(node) = candidate.replacement else { return None }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { return None }; - matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + let Replacement::SubDAG(node) = candidate.replacement else { return None }; + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator else { return None }; + matches!(&summary_input.operator, Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::UnivMon).then_some(node) }).expect("UnivMon candidate") } #[test] -fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { - // Equal data, grouping and window produce one state independently of readout. +fn four_evaluations_share_one_value_frequency_state_and_keep_honest_guarantees() { + // Equal data, grouping and window produce one state independently of evaluation. let accuracy = AccuracyTarget::Epsilon(0.02); let roots: Vec<_> = [ ("distinct_over_time(m[5m])", accuracy.clone()), @@ -76,26 +77,25 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { .enumerate() .map(|(id, (query, accuracy))| (id, candidate(query, accuracy))) .collect(); - let roots = share_common_summary_sub_dags(roots); + let roots = share_common_sub_dags(roots); let mut first_state = None; for (index, root) in &roots { - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - .. - } = &root.expr + }) = &root.operator else { panic!() }; if let Some(first) = &first_state { assert!( Rc::ptr_eq(first, summary_input), - "state must be shared across readouts" + "state must be shared across evaluations" ); } else { first_state = Some(Rc::clone(summary_input)); } - let SummaryExpr::SummaryAgg { input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &summary_input.operator else { panic!() }; assert!(matches!(input.item, Some(SummaryInputExpr::Column(_)))); @@ -108,7 +108,7 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { assert!(root.guarantee.as_ref().is_some_and(|g| g.is_exact())); } else { assert!(!root.guarantee.as_ref().unwrap().is_exact()); - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { panic!() }; assert!( @@ -118,12 +118,12 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { "production has no calibrated error bound" ); } - compile_post_asap_dag(root).unwrap(); + post_asap_dag(root); } } #[test] -fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { +fn uncalibrated_frequency_evaluations_do_not_bypass_accuracy_targets() { // An unmeasured heuristic remains inspectable but is never certified or // automatically selected for a caller-visible bounded-error result. for query in ["entropy_over_time(m[5m])", "l2_over_time(m[5m])"] { @@ -135,16 +135,16 @@ fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { delta: 0.01, }, ] { - let root = Rc::new(lower_promql(query, target.clone()).unwrap()); - let candidates = SketchAlgorithmStrategy::default_cost_model() - .replacements(&TargetSubDAG::new(&root)); + let root = lower_promql(query, target.clone()).unwrap(); + let candidates = + ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let unknown = candidates .iter() .filter(|candidate| { matches!( &candidate.replacement, - Replacement::Summary(node) - if matches!(&node.expr, SummaryExpr::SummaryEstimate { .. }) + Replacement::SubDAG(node) + if matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) && node.guarantee.is_none() && candidate.has_missing_accuracy_evidence() ) diff --git a/crates/frontend-sql/src/error.rs b/crates/frontend-sql/src/error.rs index 11d5acff4..059404ac1 100644 --- a/crates/frontend-sql/src/error.rs +++ b/crates/frontend-sql/src/error.rs @@ -1,11 +1,11 @@ use std::fmt; -use asap_types::pre_asap::ResolveDAGError; +use asap_frontend_common::ResolveDAGError; /// Errors from lowering a SQL query (parse + plan via DataFusion → the -/// canonical, unresolved DAG, built directly → -/// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved DAG, issue #179). +/// name-based [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree → +/// [`resolve_root`](asap_frontend_common::resolve_root) binds it into the +/// unified IR). /// /// Carries no PromQL type — the SQL front end never depends on the PromQL /// parser. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` @@ -28,8 +28,8 @@ pub enum SqlError { UnsupportedFeature(String), /// The workload's query language is not SQL. WrongLanguage(String), - /// Resolving the canonical unresolved DAG failed (name resolution - /// against the bound schema). + /// Resolving the name-based tree failed (name resolution against the + /// bound schema, or schema derivation). Convert(ResolveDAGError), } diff --git a/crates/frontend-sql/src/lib.rs b/crates/frontend-sql/src/lib.rs index aa1449e5e..1747b09c8 100644 --- a/crates/frontend-sql/src/lib.rs +++ b/crates/frontend-sql/src/lib.rs @@ -1,25 +1,27 @@ -//! SQL front end: parse + plan (via DataFusion) → the canonical, unresolved -//! shape, built directly (issue #179) → [`resolve_root`]. +//! SQL front end: parse + plan (via DataFusion) → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree, built directly +//! (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved -//! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational DAG; `resolve_root` runs the -//! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. +//! Emits the shared front-end tree (`UnresolvedOp` / `UnresolvedScalar`, +//! name-based [`ColumnRef`](asap_types::pre_asap::ColumnRef)s) directly, rather +//! than a separate per-language relational tree; `resolve_root` binds it into +//! the unified [`OperatorNode`] IR, deriving every schema on the way. //! Depends on DataFusion only — never on the PromQL parser. pub mod error; pub mod sql; -use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use std::rc::Rc; + +use asap_frontend_common::resolve_root; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{QueryLanguage, QueryWorkload, SqlDialect}; pub use error::SqlError; pub use sql::{SqlCatalog, SqlLowerer}; -/// Lower a single SQL query string to the canonical, resolved `QueryExpr`, +/// Lower a single SQL query string to the resolved, canonical operator DAG, /// parsed as `SqlDialect::DataFusionSQL`. /// /// The `catalog` supplies table schemas (used both to plan the SQL with @@ -29,7 +31,7 @@ pub async fn lower_sql( query: &str, catalog: &SqlCatalog, accuracy: AccuracyTarget, -) -> Result { +) -> Result, SqlError> { lower_sql_dialect(query, catalog, SqlDialect::DataFusionSQL, accuracy).await } @@ -46,20 +48,17 @@ pub async fn lower_sql_dialect( catalog: &SqlCatalog, dialect: SqlDialect, accuracy: AccuracyTarget, -) -> Result { +) -> Result, SqlError> { let unresolved = SqlLowerer::with_dialect(catalog, dialect) .lower(query, &accuracy) .await?; - let resolved = resolve_root(&unresolved)?; - // Binding resolves names; schema inference also checks result types such - // as temporal subtraction, whose duration unit the IR cannot represent. - resolved - .output_schema() - .map_err(|error| SqlError::InvalidExpression(error.to_string()))?; - Ok(resolved) + // Binding resolves names and derives every node's schema; result-type + // checks (such as temporal subtraction, whose duration unit the IR cannot + // represent) surface here as `ResolveDAGError::Schema`. + Ok(resolve_root(&unresolved)?) } -/// Lower every SQL batch entry in `workload` to a `QueryExpr`. +/// Lower every SQL batch entry in `workload` to an operator DAG. /// /// One `Result` per entry — errors are per-query, not fatal for the batch. /// Returns `WrongLanguage` for every entry if the workload is not SQL, and @@ -67,7 +66,7 @@ pub async fn lower_sql_dialect( pub async fn lower_sql_batch( workload: &QueryWorkload, catalog: &SqlCatalog, -) -> Vec> { +) -> Vec, SqlError>> { let entries = match &workload.query_batch { Some(e) if !e.is_empty() => e, _ => return vec![], @@ -102,6 +101,3 @@ pub async fn lower_sql_batch( } results } - -/// Unified SQL lowering; promoted to the root API at the planner cutover. -pub mod unified; diff --git a/crates/frontend-sql/src/sql/collection_planning.rs b/crates/frontend-sql/src/sql/collection_planning.rs index 9d28cd1df..62bbbbb6d 100644 --- a/crates/frontend-sql/src/sql/collection_planning.rs +++ b/crates/frontend-sql/src/sql/collection_planning.rs @@ -1,10 +1,10 @@ //! DataFusion planning adapters. Types come from the canonical signature rules; //! physical evaluation deliberately remains the query engine's responsibility. use super::types::{arrow_to_dtype, dtype_to_arrow, scalar_value_to_asap}; -use asap_types::pre_asap::scalar_type_rules::{ - element_access_type, struct_field_type, MapScalarFunction, -}; -use asap_types::pre_asap::{Field, QueryExpr, Schema}; +use asap_types::ir::scalar::{element_access_type, struct_field_type}; +use asap_types::ir::ScalarExpr; +use asap_types::pre_asap::scalar_type_rules::MapScalarFunction; +use asap_types::pre_asap::{Field, Schema}; use datafusion::arrow::datatypes::{DataType, Field as ArrowField, FieldRef}; use datafusion::common::{DataFusionError, Result, ScalarValue as DfScalarValue}; use datafusion::logical_expr::{ @@ -108,10 +108,10 @@ impl CollectionPlanningFunction { .map(|index| { if let Some(Some(value)) = literals.and_then(|args| args.get(index)) { scalar_value_to_asap(value) - .map(QueryExpr::Literal) + .map(ScalarExpr::Literal) .map_err(|error| DataFusionError::Plan(error.to_string())) } else { - Ok(QueryExpr::Column(index)) + Ok(ScalarExpr::Column(index)) } }) .collect::>>()?; diff --git a/crates/frontend-sql/src/sql/expr.rs b/crates/frontend-sql/src/sql/expr.rs index 32ba893df..3faf9c306 100644 --- a/crates/frontend-sql/src/sql/expr.rs +++ b/crates/frontend-sql/src/sql/expr.rs @@ -2,12 +2,14 @@ use std::rc::Rc; use datafusion::logical_expr::{BinaryExpr, Expr, Operator}; +use asap_frontend_common::UnresolvedScalar as Unresolved; +use asap_types::ir::ExprSemantics; use asap_types::pre_asap::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; use crate::error::SqlError as LoweringError; use super::types::{arrow_to_dtype, scalar_value_to_asap}; -use super::Unresolved; +use super::SqlLowerer; pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { match expr { @@ -24,258 +26,276 @@ pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { } } -/// Translate a DataFusion `Expr` to the canonical, unresolved DAG. -/// Returns `UnsupportedFeature` for anything not needed in v1. -pub(super) fn df_expr_to_unresolved(expr: &Expr) -> Result { - match expr { - // Preserve DataFusion's relation qualifier so a column name shared - // across a join (`a.k` vs `b.k`) resolves to the correct side. - Expr::Column(col) => Ok(Unresolved::Column(match &col.relation { - Some(rel) => ColumnRef::Qualified { - table: rel.to_string(), - name: col.name.clone(), - }, - None => ColumnRef::Named(col.name.clone()), - })), - - // Keep Arrow date literals equivalent to SQL CAST('YYYY-MM-DD' AS DATE), - // including typed nulls, without adding another canonical scalar variant. - Expr::Literal( - sv @ (datafusion::common::ScalarValue::Date32(_) - | datafusion::common::ScalarValue::Date64(_)), - _, - ) => { - let text = sv.cast_to(&datafusion::arrow::datatypes::DataType::Utf8)?; - // Arrow formats Date64 with a time suffix; the canonical Date has - // no time-of-day, just like Date64 catalog registration as Date32. - let text = match text { - datafusion::common::ScalarValue::Utf8(Some(value)) => { - ScalarValue::Utf8(value.split('T').next().unwrap().to_owned()) - } - other => scalar_value_to_asap(&other)?, - }; - Ok(Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(text)), - to: asap_types::pre_asap::schema::DataType::Date, - try_cast: false, - }) - } - Expr::Literal(sv, _) => scalar_value_to_asap(sv).map(Unresolved::Literal), - - Expr::Alias(a) => df_expr_to_unresolved(&a.expr), +impl SqlLowerer<'_> { + /// Translate a DataFusion `Expr` to the name-based scalar tree. Every + /// `Compare` / `Arithmetic` / `Negative` carries `ExprSemantics::Sql`. + /// Subquery-valued expressions lower their plan as a root of its own + /// (which is why this is a method: the plan walk needs the catalog). + /// Returns `UnsupportedFeature` for anything not needed in v1. + pub(super) fn lower_expr(&self, expr: &Expr) -> Result { + let bx = |e: &Expr| self.lower_expr(e).map(Box::new); + match expr { + // Preserve DataFusion's relation qualifier so a column name shared + // across a join (`a.k` vs `b.k`) resolves to the correct side. + Expr::Column(col) => Ok(Unresolved::Column(match &col.relation { + Some(rel) => ColumnRef::Qualified { + table: rel.to_string(), + name: col.name.clone(), + }, + None => ColumnRef::Named(col.name.clone()), + })), - Expr::BinaryExpr(BinaryExpr { left, op, right }) => match op { - Operator::And => { - let parts = split_conjuncts(expr); - let lowered: Result, _> = - parts.iter().map(|e| df_expr_to_unresolved(e)).collect(); - Ok(Unresolved::BoolAnd(lowered?)) - } - Operator::Or => { - let parts = split_disjuncts(expr); - let lowered: Result, _> = - parts.iter().map(|e| df_expr_to_unresolved(e)).collect(); - Ok(Unresolved::BoolOr(lowered?)) + // Keep Arrow date literals equivalent to SQL CAST('YYYY-MM-DD' AS DATE), + // including typed nulls, without adding another canonical scalar variant. + Expr::Literal( + sv @ (datafusion::common::ScalarValue::Date32(_) + | datafusion::common::ScalarValue::Date64(_)), + _, + ) => { + let text = sv.cast_to(&datafusion::arrow::datatypes::DataType::Utf8)?; + // Arrow formats Date64 with a time suffix; the canonical Date has + // no time-of-day, just like Date64 catalog registration as Date32. + let text = match text { + datafusion::common::ScalarValue::Utf8(Some(value)) => { + ScalarValue::Utf8(value.split('T').next().unwrap().to_owned()) + } + other => scalar_value_to_asap(&other)?, + }; + Ok(Unresolved::Cast { + expr: Box::new(Unresolved::Literal(text)), + to: asap_types::pre_asap::schema::DataType::Date, + try_cast: false, + }) } - Operator::Eq => compare(left, CompareOpKind::Eq, right), - Operator::NotEq => compare(left, CompareOpKind::Ne, right), - Operator::Lt => compare(left, CompareOpKind::Lt, right), - Operator::LtEq => compare(left, CompareOpKind::Le, right), - Operator::Gt => compare(left, CompareOpKind::Gt, right), - Operator::GtEq => compare(left, CompareOpKind::Ge, right), - // BinaryExpr LIKE/ILIKE operators (from optimizer rewrites) - Operator::LikeMatch => compare(left, CompareOpKind::Like, right), - Operator::ILikeMatch => compare(left, CompareOpKind::ILike, right), - Operator::NotLikeMatch => compare(left, CompareOpKind::NotLike, right), - Operator::NotILikeMatch => compare(left, CompareOpKind::NotILike, right), - // Arithmetic - Operator::Plus => arith(left, ArithmeticOpKind::Add, right), - Operator::Minus => arith(left, ArithmeticOpKind::Sub, right), - Operator::Multiply => arith(left, ArithmeticOpKind::Mul, right), - Operator::Divide => arith(left, ArithmeticOpKind::Div, right), - Operator::Modulo => arith(left, ArithmeticOpKind::Mod, right), - other => Err(LoweringError::UnsupportedFeature(format!( - "operator: {other:?}" - ))), - }, + Expr::Literal(sv, _) => scalar_value_to_asap(sv).map(Unresolved::Literal), - // SQL LIKE / ILIKE (dedicated expr node from the SQL parser) - Expr::Like(like) => { - let op = match (like.negated, like.case_insensitive) { - (false, false) => CompareOpKind::Like, - (true, false) => CompareOpKind::NotLike, - (false, true) => CompareOpKind::ILike, - (true, true) => CompareOpKind::NotILike, - }; - compare(&like.expr, op, &like.pattern) - } + Expr::Alias(a) => self.lower_expr(&a.expr), - // Unary minus: negate literals directly; wrap others in -1 * x. - Expr::Negative(inner) => { - let inner = df_expr_to_unresolved(inner)?; - match inner { - Unresolved::Literal(ScalarValue::Int64(v)) => { - Ok(Unresolved::Literal(ScalarValue::Int64(-v))) + Expr::BinaryExpr(BinaryExpr { left, op, right }) => match op { + Operator::And => { + let parts = split_conjuncts(expr); + let lowered: Result, _> = + parts.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::BoolAnd(lowered?)) } - Unresolved::Literal(ScalarValue::Float64(v)) => { - Ok(Unresolved::Literal(ScalarValue::Float64(-v))) + Operator::Or => { + let parts = split_disjuncts(expr); + let lowered: Result, _> = + parts.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::BoolOr(lowered?)) } - other => Ok(Unresolved::Arithmetic { - op: ArithmeticOpKind::Mul, - left: Rc::new(Unresolved::Literal(ScalarValue::Int64(-1))), - right: Rc::new(other), - }), + Operator::Eq => self.compare(left, CompareOpKind::Eq, right), + Operator::NotEq => self.compare(left, CompareOpKind::Ne, right), + Operator::Lt => self.compare(left, CompareOpKind::Lt, right), + Operator::LtEq => self.compare(left, CompareOpKind::Le, right), + Operator::Gt => self.compare(left, CompareOpKind::Gt, right), + Operator::GtEq => self.compare(left, CompareOpKind::Ge, right), + // BinaryExpr LIKE/ILIKE operators (from optimizer rewrites) + Operator::LikeMatch => self.compare(left, CompareOpKind::Like, right), + Operator::ILikeMatch => self.compare(left, CompareOpKind::ILike, right), + Operator::NotLikeMatch => self.compare(left, CompareOpKind::NotLike, right), + Operator::NotILikeMatch => self.compare(left, CompareOpKind::NotILike, right), + // Arithmetic + Operator::Plus => self.arith(left, ArithmeticOpKind::Add, right), + Operator::Minus => self.arith(left, ArithmeticOpKind::Sub, right), + Operator::Multiply => self.arith(left, ArithmeticOpKind::Mul, right), + Operator::Divide => self.arith(left, ArithmeticOpKind::Div, right), + Operator::Modulo => self.arith(left, ArithmeticOpKind::Mod, right), + other => Err(LoweringError::UnsupportedFeature(format!( + "operator: {other:?}" + ))), + }, + + // SQL LIKE / ILIKE (dedicated expr node from the SQL parser) + Expr::Like(like) => { + let op = match (like.negated, like.case_insensitive) { + (false, false) => CompareOpKind::Like, + (true, false) => CompareOpKind::NotLike, + (false, true) => CompareOpKind::ILike, + (true, true) => CompareOpKind::NotILike, + }; + self.compare(&like.expr, op, &like.pattern) } - } - // SQL CASE expression - Expr::Case(c) => { - let operand = c - .expr - .as_ref() - .map(|e| df_expr_to_unresolved(e).map(Rc::new)) - .transpose()?; - let branches = c - .when_then_expr - .iter() - .map(|(when, then)| { - Ok((df_expr_to_unresolved(when)?, df_expr_to_unresolved(then)?)) + // Unary minus. (DataFusion's planner already folds `-` + // into a negative literal, so this is a non-literal operand.) + Expr::Negative(inner) => Ok(Unresolved::Negative { + expr: bx(inner)?, + semantics: ExprSemantics::Sql, + }), + + // SQL CASE expression + Expr::Case(c) => { + let operand = c.expr.as_deref().map(bx).transpose()?; + let branches = c + .when_then_expr + .iter() + .map(|(when, then)| Ok((self.lower_expr(when)?, self.lower_expr(then)?))) + .collect::, LoweringError>>()?; + let else_expr = c.else_expr.as_deref().map(bx).transpose()?; + Ok(Unresolved::Case { + operand, + branches, + else_expr, }) - .collect::, LoweringError>>()?; - let else_expr = c - .else_expr - .as_ref() - .map(|e| df_expr_to_unresolved(e).map(Rc::new)) - .transpose()?; - Ok(Unresolved::Case { - operand, - branches, - else_expr, - }) - } + } - Expr::Not(inner) => Ok(Unresolved::Not(Rc::new(df_expr_to_unresolved(inner)?))), + Expr::Not(inner) => Ok(Unresolved::Not(bx(inner)?)), - Expr::IsNull(inner) => Ok(Unresolved::IsNull(Rc::new(df_expr_to_unresolved(inner)?))), + Expr::IsNull(inner) => Ok(Unresolved::IsNull(bx(inner)?)), - Expr::IsNotNull(inner) => Ok(Unresolved::IsNotNull(Rc::new(df_expr_to_unresolved( - inner, - )?))), + Expr::IsNotNull(inner) => Ok(Unresolved::IsNotNull(bx(inner)?)), - Expr::Cast(c) => { - let inner = df_expr_to_unresolved(&c.expr)?; - let to = arrow_to_dtype(c.field.data_type())?; - Ok(Unresolved::Cast { - expr: Rc::new(inner), - to, + // DataFusion 54 coerces a mixed signed/unsigned integer comparison + // (e.g. `approx_distinct(x) >= 1000`) through `Decimal128(20, 0)`. + // Canonical integers are all Int64, so that widening is a no-op. + Expr::Cast(c) + if matches!( + c.field.data_type(), + datafusion::arrow::datatypes::DataType::Decimal128(_, 0) + ) => + { + self.lower_expr(&c.expr) + } + Expr::Cast(c) => Ok(Unresolved::Cast { + expr: bx(&c.expr)?, + to: arrow_to_dtype(c.field.data_type())?, try_cast: false, - }) - } + }), - // TRY_CAST returns NULL on conversion failure; preserve that semantic. - Expr::TryCast(c) => { - let inner = df_expr_to_unresolved(&c.expr)?; - let to = arrow_to_dtype(c.field.data_type())?; - Ok(Unresolved::Cast { - expr: Rc::new(inner), - to, + // TRY_CAST returns NULL on conversion failure; preserve that semantic. + Expr::TryCast(c) => Ok(Unresolved::Cast { + expr: bx(&c.expr)?, + to: arrow_to_dtype(c.field.data_type())?, try_cast: true, - }) - } + }), - Expr::InList(il) => { - let expr = df_expr_to_unresolved(&il.expr)?; - let list: Result, _> = il.list.iter().map(df_expr_to_unresolved).collect(); - Ok(Unresolved::InList { - expr: Rc::new(expr), - list: list?, - negated: il.negated, - }) - } + Expr::InList(il) => { + let list: Result, _> = il.list.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::InList { + expr: bx(&il.expr)?, + list: list?, + negated: il.negated, + }) + } + + Expr::Between(b) => { + // Normalize: `x BETWEEN low AND high` → `x >= low AND x <= high`. + // `x NOT BETWEEN low AND high` → `x < low OR x > high`. + if b.negated { + let lt = self.compare(&b.expr, CompareOpKind::Lt, &b.low)?; + let gt = self.compare(&b.expr, CompareOpKind::Gt, &b.high)?; + Ok(Unresolved::BoolOr(vec![lt, gt])) + } else { + let x_low = self.compare(&b.expr, CompareOpKind::Ge, &b.low)?; + let x_high = self.compare(&b.expr, CompareOpKind::Le, &b.high)?; + Ok(Unresolved::BoolAnd(vec![x_low, x_high])) + } + } - Expr::Between(b) => { - // Normalize: `x BETWEEN low AND high` → `x >= low AND x <= high`. - // `x NOT BETWEEN low AND high` → `x < low OR x > high`. - let x_low = compare(&b.expr, CompareOpKind::Ge, &b.low)?; - let x_high = compare(&b.expr, CompareOpKind::Le, &b.high)?; - if b.negated { - // NOT BETWEEN: invert each side - let lt = compare(&b.expr, CompareOpKind::Lt, &b.low)?; - let gt = compare(&b.expr, CompareOpKind::Gt, &b.high)?; - Ok(Unresolved::BoolOr(vec![lt, gt])) - } else { - Ok(Unresolved::BoolAnd(vec![x_low, x_high])) + // `NOW()` / `CURRENT_TIMESTAMP` read the SQL statement evaluation + // time. Keep this timestamp-typed leaf distinct from PromQL's + // Float64 Unix-seconds `EvalTimestamp`. Issue #184. + Expr::ScalarFunction(sf) + if sf.args.is_empty() + && matches!( + sf.func.name().to_ascii_lowercase().as_str(), + "now" | "current_timestamp" + ) => + { + Ok(Unresolved::CurrentTimestamp) } - } - // `NOW()` / `CURRENT_TIMESTAMP` read the SQL statement evaluation - // time. Keep this timestamp-typed leaf distinct from PromQL's - // Float64 Unix-seconds `EvalTimestamp`. Issue #184. - Expr::ScalarFunction(sf) - if sf.args.is_empty() - && matches!( - sf.func.name().to_ascii_lowercase().as_str(), - "now" | "current_timestamp" - ) => - { - Ok(Unresolved::CurrentTimestamp) - } + Expr::ScalarFunction(sf) => { + let args: Result, _> = sf.args.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::FunctionCall { + name: if sf.func.name().eq_ignore_ascii_case("arrayelement") { + "asap_element_access".into() + } else if sf.func.name().eq_ignore_ascii_case("tupleelement") { + "asap_struct_field".into() + } else if sf.func.name() == super::collection_planning::MAP_PLANNING_NAME { + "map".into() + } else { + sf.func.name().to_string() + }, + args: args?, + }) + } - Expr::ScalarFunction(sf) => { - let args: Result, _> = sf.args.iter().map(df_expr_to_unresolved).collect(); - Ok(Unresolved::FunctionCall { - name: if sf.func.name().eq_ignore_ascii_case("arrayelement") { - "asap_element_access".into() - } else if sf.func.name().eq_ignore_ascii_case("tupleelement") { - "asap_struct_field".into() - } else if sf.func.name() == super::collection_planning::MAP_PLANNING_NAME { - "map".into() - } else { - sf.func.name().to_string() - }, - args: args?, - }) - } + // Subquery-valued expressions. Each subquery plan is lowered as a + // root of its own; `resolve_root` binds it in its own scope, so an + // outer reference inside it has nothing to resolve against — a + // correlated subquery is rejected rather than mislowered. + Expr::ScalarSubquery(sq) => Ok(Unresolved::ScalarSubquery(Rc::new( + self.lower_uncorrelated_subquery(sq, "scalar subquery")?, + ))), + Expr::Exists(ex) => Ok(Unresolved::Exists { + subquery: Rc::new(self.lower_uncorrelated_subquery(&ex.subquery, "EXISTS")?), + negated: ex.negated, + }), + Expr::InSubquery(is) => { + let fields = is.subquery.subquery.schema().fields().len(); + if fields != 1 { + return Err(LoweringError::InvalidExpression(format!( + "IN (subquery) must select exactly one column, got {fields}" + ))); + } + Ok(Unresolved::InSubquery { + expr: bx(&is.expr)?, + subquery: Rc::new( + self.lower_uncorrelated_subquery(&is.subquery, "IN (subquery)")?, + ), + negated: is.negated, + }) + } - // Subquery-valued expressions in a predicate/projection — `x > (SELECT - // …)`, `x IN (SELECT …)`, `EXISTS (SELECT …)`. These need a subquery - // node in the unresolved expression IR (and a correlated-vs-uncorrelated - // decision); rejected cleanly until that lands rather than mislowered. - // Derived tables in `FROM` (the common nesting shape) ARE supported — - // see `lower_plan`'s `SubqueryAlias` arm. - Expr::ScalarSubquery(_) | Expr::InSubquery(_) | Expr::Exists(_) => Err( - LoweringError::UnsupportedFeature("subquery-valued expression in predicate".into()), - ), + other => Err(LoweringError::UnsupportedFeature(format!( + "expression: {}", + other + ))), + } + } - other => Err(LoweringError::UnsupportedFeature(format!( - "expression: {}", - other - ))), + fn lower_uncorrelated_subquery( + &self, + sq: &datafusion::logical_expr::Subquery, + what: &str, + ) -> Result { + if !sq.outer_ref_columns.is_empty() { + return Err(LoweringError::UnsupportedFeature(format!( + "correlated {what}" + ))); + } + self.lower_plan(&sq.subquery) } -} -pub(super) fn compare( - left: &Expr, - op: CompareOpKind, - right: &Expr, -) -> Result { - Ok(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(left)?), - op, - right: Rc::new(df_expr_to_unresolved(right)?), - }) -} + pub(super) fn compare( + &self, + left: &Expr, + op: CompareOpKind, + right: &Expr, + ) -> Result { + Ok(Unresolved::Compare { + left: Box::new(self.lower_expr(left)?), + op, + right: Box::new(self.lower_expr(right)?), + semantics: ExprSemantics::Sql, + }) + } -pub(super) fn arith( - left: &Expr, - op: ArithmeticOpKind, - right: &Expr, -) -> Result { - Ok(Unresolved::Arithmetic { - op, - left: Rc::new(df_expr_to_unresolved(left)?), - right: Rc::new(df_expr_to_unresolved(right)?), - }) + fn arith( + &self, + left: &Expr, + op: ArithmeticOpKind, + right: &Expr, + ) -> Result { + Ok(Unresolved::Arithmetic { + op, + left: Box::new(self.lower_expr(left)?), + right: Box::new(self.lower_expr(right)?), + semantics: ExprSemantics::Sql, + }) + } } pub(super) fn split_disjuncts(expr: &Expr) -> Vec<&Expr> { @@ -296,12 +316,15 @@ pub(super) fn split_disjuncts(expr: &Expr) -> Vec<&Expr> { #[cfg(test)] mod tests { use super::*; + use crate::sql::SqlCatalog; use asap_types::pre_asap::schema::DataType; use datafusion::common::ScalarValue as DfScalarValue; // Typed Arrow dates normalize to the same typed form as SQL date casts. #[test] fn arrow_date_literals_preserve_value_and_type() { + let catalog = SqlCatalog::new(); + let lowerer = SqlLowerer::new(&catalog); for (value, expected) in [ ( DfScalarValue::Date32(Some(0)), @@ -314,15 +337,32 @@ mod tests { (DfScalarValue::Date32(None), ScalarValue::Null), (DfScalarValue::Date64(None), ScalarValue::Null), ] { - let actual = df_expr_to_unresolved(&Expr::Literal(value, None)).unwrap(); + let actual = lowerer.lower_expr(&Expr::Literal(value, None)).unwrap(); assert_eq!( actual, Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(expected)), + expr: Box::new(Unresolved::Literal(expected)), to: DataType::Date, try_cast: false, } ); } } + + // Unary minus over a non-literal is the `Negative` scalar, SQL-flavoured. + #[test] + fn unary_minus_lowers_to_negative_with_sql_semantics() { + let catalog = SqlCatalog::new(); + let lowerer = SqlLowerer::new(&catalog); + let expr = Expr::Negative(Box::new(Expr::Column( + datafusion::common::Column::new_unqualified("x"), + ))); + assert_eq!( + lowerer.lower_expr(&expr).unwrap(), + Unresolved::Negative { + expr: Box::new(Unresolved::Column(ColumnRef::Named("x".into()))), + semantics: ExprSemantics::Sql, + } + ); + } } diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index 3c708a2ed..78a067f53 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -1,12 +1,12 @@ -//! SQL → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! SQL → the name-based front-end tree +//! ([`UnresolvedOp`](asap_frontend_common::UnresolvedOp) / +//! [`UnresolvedScalar`](asap_frontend_common::UnresolvedScalar)). //! //! Parses SQL via DataFusion (over the catalog's registered tables), then -//! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with -//! unresolved `ColumnRef`s directly (issue #179) — the same DAG shape -//! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, -//! positional `QueryExpr`. Unlike PromQL's front end, SQL's +//! walks the unoptimized `LogicalPlan` and emits `UnresolvedOp` nodes with +//! unresolved `ColumnRef`s directly (issue #179) — the same tree shape +//! [`resolve_root`](asap_frontend_common::resolve_root) binds into the +//! positional, unified `OperatorNode` IR. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit //! `asap_rate`/`asap_increase` bridge is the narrow exception: it //! spells a time-series range reducer with an explicit value, time-index, and @@ -52,17 +52,22 @@ use datafusion::prelude::{SessionConfig, SessionContext}; use datafusion::sql::parser::DFParser; use datafusion::sql::sqlparser::dialect::GenericDialect; +use asap_frontend_common::{ + resolve_root, UnresolvedOp as Unresolved, UnresolvedPredicate as Predicate, + UnresolvedProjectItem as ProjectItem, UnresolvedScalar as Scalar, UnresolvedSortKey as SortKey, +}; use asap_sql_function_catalog::{AggSemantic, Arity, RewriteKind}; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{ - GroupKeys, Predicate, ProjectItem, Reduction, SortKey, Source, - UnresolvedQueryExpr as Unresolved, WindowFrame, WindowFrameBound, WindowFrameOffset, +use asap_types::ir::operator_properties::{ + GroupKeys, Reduction, Source, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, }; +use asap_types::ir::TimeRangeKind; +use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::pre_asap::schema::{DataType, FieldDataType, Schema}; + use asap_types::pre_asap::{ - resolve_column_ref, resolve_root, ColumnRef, CompareOpKind, JoinKind, RelationalSetOpKind, - ScalarValue, WindowFuncKind, + resolve_column_ref, ColumnRef, CompareOpKind, JoinKind, RelationalSetOpKind, ScalarValue, + WindowFuncKind, }; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -76,7 +81,6 @@ mod types; pub use types::SqlCatalog; -use self::expr::df_expr_to_unresolved; use self::types::{arrow_to_dtype, scalar_value_to_asap, schema_to_arrow}; std::thread_local! { @@ -110,10 +114,10 @@ fn current_accuracy() -> AccuracyTarget { ACCURACY.with(|a| a.borrow().clone()) } -/// Lowers SQL strings to the canonical [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) -/// over a table [`SqlCatalog`]. Call -/// [`resolve_root`](asap_types::pre_asap::resolve_root) on the result for -/// the canonical, resolved DAG. +/// Lowers SQL strings to the name-based [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) +/// tree over a table [`SqlCatalog`]. Call +/// [`resolve_root`](asap_frontend_common::resolve_root) on the result for +/// the resolved operator DAG. pub struct SqlLowerer<'a> { catalog: &'a SqlCatalog, dialect: SqlDialect, @@ -139,7 +143,7 @@ impl<'a> SqlLowerer<'a> { Self { catalog, dialect } } - /// Parse + lower a SQL query to the canonical, unresolved shape, threading + /// Parse + lower a SQL query to the name-based tree, threading /// `accuracy` onto every approximate intent (`Count`, `Quantile`, /// `Cardinality`) as it is built. /// @@ -166,8 +170,8 @@ impl<'a> SqlLowerer<'a> { /// a rule) that isn't wanted here — e.g. it independently rejects a /// multi-column `IN (subquery)` before `lower_in_subquery`'s own arity /// check would. Going straight to `ApplyFunctionRewrites` avoids that - /// entirely: zero behavior change for every query that doesn't call a - /// catalog-listed ClickHouse builtin. + /// entirely. TypeCoercion then records implicit conversions explicitly, + /// including timestamp literals in predicates, before IR validation. pub async fn lower( &self, sql: &str, @@ -226,6 +230,8 @@ impl<'a> SqlLowerer<'a> { }) }) })?; + let plan = datafusion::optimizer::analyzer::type_coercion::TypeCoercion::new() + .analyze(plan, &ctx.state().options())?; let _guard = AccuracyGuard::install(accuracy.clone()); self.lower_plan(&plan) } @@ -275,8 +281,8 @@ impl<'a> SqlLowerer<'a> { // *scalar* builtin — same reason as the `AggregateUDF` loop above // (DataFusion otherwise rejects the call as an unknown function // during `SqlToRel` conversion), but with no rewrite step to follow: - // `df_expr_to_unresolved`'s `Expr::ScalarFunction` arm already lowers - // any scalar call generically to `Unresolved::FunctionCall { name, + // `lower_expr`'s `Expr::ScalarFunction` arm already lowers any + // scalar call generically to `UnresolvedScalar::FunctionCall { name, // args }`, so registering the stub is the entire fix (issue #230). for builtin in asap_sql_function_catalog::CLICKHOUSE_SCALAR_BUILTINS { ctx.register_udf(clickhouse_scalar_builtin_stub_udf( @@ -308,9 +314,24 @@ impl<'a> SqlLowerer<'a> { Ok(ctx) } - fn lower_plan(&self, plan: &LogicalPlan) -> Result { + pub(super) fn lower_plan(&self, plan: &LogicalPlan) -> Result { match plan { LogicalPlan::TableScan(scan) => self.lower_table_scan(scan), + // The one empty input row of a `SELECT` without `FROM`. + LogicalPlan::EmptyRelation(empty) => Ok(Unresolved::Values { + rows: if empty.produce_one_row { + vec![vec![]] + } else { + vec![] + }, + schema: Schema { + fields: vec![], + time_index: None, + unique_keys: vec![], + closed: true, + }, + }), + LogicalPlan::Values(values) => self.lower_values(values), LogicalPlan::Filter(filter) => self.lower_filter(filter), LogicalPlan::Projection(proj) => self.lower_projection(proj), LogicalPlan::Aggregate(agg) => self.lower_aggregate(agg), @@ -379,9 +400,7 @@ impl<'a> SqlLowerer<'a> { .iter() .map(|f| ProjectItem { alias: Some(f.name().clone()), - expr: Unresolved::Column(ColumnRef::Named( - f.name().clone(), - )), + expr: Scalar::Column(ColumnRef::Named(f.name().clone())), }) .collect(); Ok(Unresolved::Project { @@ -404,108 +423,48 @@ impl<'a> SqlLowerer<'a> { /// `WHERE` — a conjunction of ordinary predicates plus, possibly, subquery /// predicates (issue #111). /// - /// `c IN (SELECT …)` and `EXISTS (…)` are not expressions over rows; they are - /// *joins*. Each such conjunct peels off into a semi- / anti-join above the - /// filter's input, and the remaining conjuncts stay as an ordinary `Filter`. + /// The ordinary conjuncts stay one predicate, folded onto a bare `Scan` + /// (`filter_or_fold`). A subquery conjunct — `c IN (SELECT …)`, `EXISTS + /// (…)`, `x > (SELECT …)` — is a row filter whose predicate reads another + /// operator (`UnresolvedScalar::InSubquery` / `Exists` / + /// `ScalarSubquery`); each one becomes its own `Filter` **above** the + /// ordinary predicate, so the shared `canonicalize` pass can turn it into + /// the join it is without having to peel it out of a conjunction or off + /// a `Scan` (it only lifts subqueries out of `Filter` / `Project`). A + /// semi-join only ever drops left rows, so the two orders agree. /// - /// The residual filter is applied **below** the joins, which is where it sat - /// before: a semi-join only ever drops left rows, so the two orders agree — - /// and keeping the fold-onto-`Scan` (`filter_or_fold`) below the joins - /// matches where the old converter folded it too. + /// The one subquery shape still lowered to a join here is a *correlated* + /// `EXISTS`: its correlation references both sides, which only a join + /// predicate can bind (a subquery referenced from a scalar position is + /// resolved as a root in its own scope). fn lower_filter(&self, filter: &logical_expr::Filter) -> Result { let mut conjuncts = Vec::new(); split_conjunction(&filter.predicate, &mut conjuncts); - let (subqueries, residual): (Vec<_>, Vec<_>) = conjuncts - .into_iter() - .partition(|e| matches!(e, Expr::InSubquery(_) | Expr::Exists(_))); + let (subqueries, residual): (Vec<_>, Vec<_>) = + conjuncts.into_iter().partition(|e| reads_subquery(e)); let input = self.lower_plan(&filter.input)?; let mut node = match rebuild_conjunction(&residual) { - Some(pred) => filter_or_fold(df_expr_to_unresolved(&pred)?, input), + Some(pred) => filter_or_fold(self.lower_expr(&pred)?, input), None => input, }; for sq in subqueries { node = match sq { - Expr::InSubquery(is) => self.lower_in_subquery(is, node)?, - Expr::Exists(ex) => self.lower_exists(ex, node)?, - _ => unreachable!("partitioned above"), + Expr::Exists(ex) if !ex.subquery.outer_ref_columns.is_empty() => { + self.lower_correlated_exists(ex, node)? + } + other => Unresolved::Filter { + pred: Predicate(self.lower_expr(other)?), + child: Rc::new(node), + }, }; } Ok(node) } - /// `c IN (SELECT k FROM …)` → a semi-join on `c = k` (issue #111). - fn lower_in_subquery( - &self, - is: &logical_expr::expr::InSubquery, - left: Unresolved, - ) -> Result { - if is.negated { - // `NOT IN` is not an anti-join. Under three-valued logic a single - // NULL among the subquery's rows makes `c NOT IN (…)` UNKNOWN for - // every `c`, so the query returns nothing — while an anti-join - // returns every unmatched left row. Reject rather than mislower. - return Err(LoweringError::UnsupportedFeature( - "NOT IN (subquery): its NULL semantics are not an anti-join".into(), - )); - } - if !is.subquery.outer_ref_columns.is_empty() { - return Err(LoweringError::UnsupportedFeature( - "correlated IN (subquery)".into(), - )); - } - let inner = is.subquery.subquery.as_ref(); - let fields = inner.schema().fields(); - if fields.len() != 1 { - return Err(LoweringError::InvalidExpression(format!( - "IN (subquery) must select exactly one column, got {}", - fields.len() - ))); - } - let key = &fields[0]; - // Project the key under a name the outer relation cannot also carry. The - // join predicate resolves against the concatenated `left ++ right` - // schema, and a bare `hosts.service` over an unqualified subquery output - // falls back to a name lookup that finds the *left's* `service` first — - // silently making the predicate `service = service`, i.e. always true. - let right = match inner { - // Rebuild the subquery's projection with the synthetic alias, so a - // computed key (`SELECT bytes + 1 …`) is named rather than becoming - // the anonymous `col_0` that nothing can reference. - LogicalPlan::Projection(p) if p.expr.len() == 1 => Unresolved::Project { - cols: vec![ProjectItem { - alias: Some(IN_SUBQUERY_KEY.to_string()), - expr: df_expr_to_unresolved(unalias(&p.expr[0]))?, - }], - qualifier: None, - child: Rc::new(self.lower_plan(&p.input)?), - }, - other => Unresolved::Project { - cols: vec![ProjectItem { - alias: Some(IN_SUBQUERY_KEY.to_string()), - expr: Unresolved::Column(ColumnRef::Named(key.name().clone())), - }], - qualifier: None, - child: Rc::new(self.lower_plan(other)?), - }, - }; - Ok(Unresolved::Join { - kind: JoinKind::Semi, - pred: Predicate(Rc::new(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(&is.expr)?), - op: CompareOpKind::Eq, - right: Rc::new(Unresolved::Column(ColumnRef::Named( - IN_SUBQUERY_KEY.to_string(), - ))), - })), - left: Rc::new(left), - right: Rc::new(right), - }) - } - /// `[NOT] EXISTS (SELECT … WHERE inner.k = outer.k)` → a semi- / anti-join /// on the correlation predicate (issue #111). - fn lower_exists( + fn lower_correlated_exists( &self, ex: &logical_expr::expr::Exists, left: Unresolved, @@ -526,12 +485,9 @@ impl<'a> SqlLowerer<'a> { // the join predicate. Whatever is left stays an ordinary inner filter. let (inner, correlation) = split_correlation(inner)?; let right = self.lower_plan(&inner)?; - // No correlation conjunct (a genuinely uncorrelated `EXISTS`) means - // the join condition is unconditionally true — same convention as an - // unconditional `JOIN` (`lower_join`, below). let pred = match correlation { - Some(e) => Predicate(Rc::new(df_expr_to_unresolved(&e)?)), - None => Predicate(Rc::new(Unresolved::Literal(ScalarValue::Boolean(true)))), + Some(e) => Predicate(self.lower_expr(&e)?), + None => Predicate(Scalar::Literal(ScalarValue::Boolean(true))), }; Ok(Unresolved::Join { kind, @@ -541,6 +497,37 @@ impl<'a> SqlLowerer<'a> { }) } + /// `VALUES (…), (…)` — one row per values row, typed by DataFusion's + /// declared schema. Row expressions have no input-column scope. + fn lower_values(&self, values: &logical_expr::Values) -> Result { + let rows = values + .values + .iter() + .map(|row| row.iter().map(|e| self.lower_expr(e)).collect()) + .collect::>, LoweringError>>()?; + let fields = values + .schema + .fields() + .iter() + .map(|f| { + Ok(asap_types::pre_asap::Field::plain( + f.name().clone(), + arrow_to_dtype(f.data_type())?, + f.is_nullable(), + )) + }) + .collect::, LoweringError>>()?; + Ok(Unresolved::Values { + rows, + schema: Schema { + fields, + time_index: None, + unique_keys: vec![], + closed: true, + }, + }) + } + /// Table leaf — carries the catalog's resolved schema directly on `Scan` /// (`schema: Some(_)`), so `resolve_root`'s SchemaResolver doesn't need to /// usage-derive it (SQL is never schemaless). Projection pushdown is left @@ -604,23 +591,17 @@ impl<'a> SqlLowerer<'a> { let mut conjuncts = join .on .iter() - .map(|(l, r)| { - Ok(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(l)?), - op: CompareOpKind::Eq, - right: Rc::new(df_expr_to_unresolved(r)?), - }) - }) + .map(|(l, r)| self.compare(l, CompareOpKind::Eq, r)) .collect::, LoweringError>>()?; if let Some(filter) = &join.filter { - conjuncts.push(df_expr_to_unresolved(filter)?); + conjuncts.push(self.lower_expr(filter)?); } - let pred = Predicate(Rc::new(match conjuncts.len() { + let pred = Predicate(match conjuncts.len() { // No condition (a CROSS JOIN) is unconditionally true. - 0 => Unresolved::Literal(ScalarValue::Boolean(true)), + 0 => Scalar::Literal(ScalarValue::Boolean(true)), 1 => conjuncts.pop().unwrap(), - _ => Unresolved::BoolAnd(conjuncts), - })); + _ => Scalar::BoolAnd(conjuncts), + }); Ok(Unresolved::Join { kind, pred, @@ -643,6 +624,10 @@ impl<'a> SqlLowerer<'a> { .window_expr .first() .ok_or_else(|| LoweringError::InvalidExpression("empty window expression".into()))?; + let first = match first { + Expr::Alias(alias) => alias.expr.as_ref(), + other => other, + }; let Expr::WindowFunction(wf) = first else { return Err(LoweringError::InvalidExpression( "expected a window function in Window plan node".into(), @@ -653,12 +638,12 @@ impl<'a> SqlLowerer<'a> { .params .args .iter() - .map(df_expr_to_unresolved) + .map(|e| self.lower_expr(e)) .collect::, _>>()?; // Nth_value: lift N from the (literal) 2nd arg, keep only the column. let func = if matches!(func, WindowFuncKind::NthValue(None)) { let n = match args.get(1) { - Some(Unresolved::Literal(ScalarValue::Int64(n))) if *n > 0 => *n as u64, + Some(Scalar::Literal(ScalarValue::Int64(n))) if *n > 0 => *n as u64, other => { return Err(LoweringError::InvalidExpression(format!( "NTH_VALUE requires a positive integer literal 2nd arg, got {other:?}" @@ -681,7 +666,7 @@ impl<'a> SqlLowerer<'a> { .order_by .iter() .map(|s| { - df_expr_to_unresolved(&s.expr).map(|expr| SortKey { + self.lower_expr(&s.expr).map(|expr| SortKey { expr, ascending: s.asc, nulls_first: s.nulls_first, @@ -716,7 +701,7 @@ impl<'a> SqlLowerer<'a> { let input = self.lower_plan(&proj.input)?; return Ok(match bridge { PlanningBridge::PromqlSubquery { range, resolution } => { - let child = Rc::new(temporal_bridge_projection(proj, input)?); + let child = Rc::new(self.temporal_bridge_projection(proj, input)?); Unresolved::PromqlSubquery { range, resolution: Some(resolution), @@ -745,22 +730,22 @@ impl<'a> SqlLowerer<'a> { .map(|e| match e { Expr::Alias(a) => { let expr = if temporal_input && is_temporal_output_column(&a.expr) { - Unresolved::Column(ColumnRef::Named("value".into())) + Scalar::Column(ColumnRef::Named("value".into())) } else { - df_expr_to_unresolved(&a.expr)? + self.lower_expr(&a.expr)? }; - Ok::, LoweringError>(ProjectItem { + Ok::(ProjectItem { expr, alias: Some(a.name.clone()), }) } _ => { let expr = if temporal_input && is_temporal_output_column(e) { - Unresolved::Column(ColumnRef::Named("value".into())) + Scalar::Column(ColumnRef::Named("value".into())) } else { - df_expr_to_unresolved(e)? + self.lower_expr(e)? }; - Ok::, LoweringError>(ProjectItem { expr, alias: None }) + Ok::(ProjectItem { expr, alias: None }) } }) .collect::, _>>()?; @@ -807,7 +792,7 @@ impl<'a> SqlLowerer<'a> { // reducer expression (`GROUP BY date_trunc(…)`, `SUM(a * 8)`) has no // slot. Materialize each one as a derived column in a `Project` beneath // the aggregate, then group/reduce over that column (issue #110). - let mut derived = DerivedCols::default(); + let mut derived = DerivedCols::new(self); // DataFusion strips `AS m` from a grouping expression, so the aggregate // schema's field name is what the enclosing Projection references — @@ -832,7 +817,7 @@ impl<'a> SqlLowerer<'a> { .get(i) .cloned() .unwrap_or_else(|| other.to_string()); - derived.materialize(name.clone(), df_expr_to_unresolved(other)?)?; + derived.materialize(name.clone(), self.lower_expr(other)?)?; keys.push(ColumnRef::Named(name)); } } @@ -874,7 +859,7 @@ impl<'a> SqlLowerer<'a> { .iter() .map(|f| { f.as_ref() - .map(|f| Ok(Predicate(Rc::new(df_expr_to_unresolved(f)?)))) + .map(|f| Ok(Predicate(self.lower_expr(f)?))) .transpose() }) .collect::, LoweringError>>()? @@ -927,12 +912,7 @@ impl<'a> SqlLowerer<'a> { )) })?; - let resolved_input = resolve_root(&input)?; - let input_schema = resolved_input.output_schema().map_err(|error| { - LoweringError::InvalidExpression(format!( - "cannot derive temporal aggregate input schema: {error}" - )) - })?; + let input_schema = resolve_root(&input)?.schema.clone(); let timestamp_id = resolve_column_ref(×tamp_ref, &input_schema).map_err(|error| { LoweringError::InvalidExpression(format!("{name} timestamp argument: {error}")) })?; @@ -1005,18 +985,18 @@ impl<'a> SqlLowerer<'a> { let mut cols = vec![ ProjectItem { alias: Some("ts".into()), - expr: Unresolved::Column(timestamp_ref.clone()), + expr: Scalar::Column(timestamp_ref.clone()), }, ProjectItem { alias: Some("value".into()), - expr: Unresolved::Column(value_ref.clone()), + expr: Scalar::Column(value_ref.clone()), }, ]; for group_ref in group_refs { let group_name = named_ref(&group_ref).to_string(); cols.push(ProjectItem { alias: Some(group_name), - expr: Unresolved::Column(group_ref), + expr: Scalar::Column(group_ref), }); } let child = Unresolved::Project { @@ -1024,8 +1004,11 @@ impl<'a> SqlLowerer<'a> { qualifier: None, child: Rc::new(input), }; + // The explicit window is a range selector over the series, the same + // shape PromQL's `rate(m[5m])` lowers to. let child = Unresolved::TimeRange { range: Duration::from_millis(window_ms), + kind: TimeRangeKind::Range, child: Rc::new(child), }; let intent = match name.as_str() { @@ -1104,7 +1087,7 @@ impl<'a> SqlLowerer<'a> { // Reducer arguments still materialize as derived columns (#110); the // grouping keys are plain columns, so they only need carrying through. - let mut derived = DerivedCols::default(); + let mut derived = DerivedCols::new(self); for e in &distinct { derived.passthrough(e)?; } @@ -1142,10 +1125,10 @@ impl<'a> SqlLowerer<'a> { .map(|((name, dtype), e)| ProjectItem { alias: Some(name.clone()), expr: if level.contains(e) { - Unresolved::Column(ColumnRef::Named(name.clone())) + Scalar::Column(ColumnRef::Named(name.clone())) } else { - Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(ScalarValue::Null)), + Scalar::Cast { + expr: Box::new(Scalar::Literal(ScalarValue::Null)), to: dtype.clone(), try_cast: false, } @@ -1153,7 +1136,7 @@ impl<'a> SqlLowerer<'a> { }) .chain(output_names.iter().map(|n| ProjectItem { alias: Some(n.clone()), - expr: Unresolved::Column(ColumnRef::Named(n.clone())), + expr: Scalar::Column(ColumnRef::Named(n.clone())), })) .collect(); Ok(Unresolved::Project { @@ -1185,7 +1168,7 @@ impl<'a> SqlLowerer<'a> { .expr .iter() .map(|s| { - df_expr_to_unresolved(&s.expr).map(|expr| SortKey { + self.lower_expr(&s.expr).map(|expr| SortKey { expr, ascending: s.asc, nulls_first: s.nulls_first, @@ -1205,8 +1188,10 @@ impl<'a> SqlLowerer<'a> { // Count-ranked `LIMIT k` over a `Sort` is promoted to the heavy-hitter // `TopK` by the shared `canonicalize` pass (issue #34), not here. Ok(Unresolved::Limit { - n: eval_fetch(&limit.fetch).unwrap_or(usize::MAX), + // No (literal) fetch is offset-only. + n: eval_fetch(&limit.fetch), offset: eval_fetch(&limit.skip).unwrap_or(0), + partition_by: GroupKeys::none(), child: Rc::new(self.lower_plan(&limit.input)?), }) } @@ -1286,49 +1271,52 @@ fn planning_bridge( /// its output slot (`... asap_promql_subquery(...) AS value ...`). This makes /// the bridge schema-preserving without silently retaining columns that SQL /// projected away. -fn temporal_bridge_projection( - projection: &logical_expr::Projection, - child: Unresolved, -) -> Result { - let cols = projection - .expr - .iter() - .map(|expr| { - if let Expr::ScalarFunction(call) = unalias(expr) { - if call - .func - .name() - .eq_ignore_ascii_case("asap_promql_subquery") - { - let Expr::Alias(alias) = expr else { - return Err(LoweringError::InvalidExpression( - "asap_promql_subquery must have an alias naming its child value column" - .into(), - )); - }; - return Ok(ProjectItem { - expr: Unresolved::Column(ColumnRef::Named(alias.name.clone())), +impl SqlLowerer<'_> { + fn temporal_bridge_projection( + &self, + projection: &logical_expr::Projection, + child: Unresolved, + ) -> Result { + let cols = projection + .expr + .iter() + .map(|expr| { + if let Expr::ScalarFunction(call) = unalias(expr) { + if call + .func + .name() + .eq_ignore_ascii_case("asap_promql_subquery") + { + let Expr::Alias(alias) = expr else { + return Err(LoweringError::InvalidExpression( + "asap_promql_subquery must have an alias naming its child value column" + .into(), + )); + }; + return Ok(ProjectItem { + expr: Scalar::Column(ColumnRef::Named(alias.name.clone())), + alias: Some(alias.name.clone()), + }); + } + } + match expr { + Expr::Alias(alias) => Ok(ProjectItem { + expr: self.lower_expr(&alias.expr)?, alias: Some(alias.name.clone()), - }); + }), + other => Ok(ProjectItem { + expr: self.lower_expr(other)?, + alias: None, + }), } - } - match expr { - Expr::Alias(alias) => Ok(ProjectItem { - expr: df_expr_to_unresolved(&alias.expr)?, - alias: Some(alias.name.clone()), - }), - other => Ok(ProjectItem { - expr: df_expr_to_unresolved(other)?, - alias: None, - }), - } + }) + .collect::, LoweringError>>()?; + Ok(Unresolved::Project { + cols, + qualifier: None, + child: Rc::new(child), }) - .collect::, LoweringError>>()?; - Ok(Unresolved::Project { - cols, - qualifier: None, - child: Rc::new(child), - }) + } } fn positive_millis_literal(expr: &Expr, argument: &str) -> Result { @@ -1413,8 +1401,8 @@ fn arity_to_signature(arity: Arity) -> Signature { // (which must become a real `AggIntent`, hence the rewrite to a native // DataFusion aggregate shape `lower_agg_intent` can classify), a scalar // function call in this IR is already deliberately opaque — -// `expr::df_expr_to_unresolved`'s `Expr::ScalarFunction` arm lowers *any* -// scalar call generically to `Unresolved::FunctionCall { name, args }`, with +// `SqlLowerer::lower_expr`'s `Expr::ScalarFunction` arm lowers *any* +// scalar call generically to `UnresolvedScalar::FunctionCall { name, args }`, with // zero name-specific logic. So teaching DataFusion's planner to accept a // ClickHouse scalar builtin's name — a stub `ScalarUDF`, registered below — // is the entire fix; the existing generic lowering already does the rest. @@ -1933,23 +1921,19 @@ fn lower_arg_selector( })) } -/// The name an `IN (subquery)`'s key column is projected under, so the join -/// predicate cannot bind it to a same-named column of the outer relation. -const IN_SUBQUERY_KEY: &str = "__asap_in_key"; - /// Fold `pred` directly onto `child.predicates` when `child` is a bare `Scan` /// (a `WHERE` directly over a table), otherwise wrap it in an ordinary /// `Filter` — canonical's invariant that a `Filter` never sits directly over a /// `Scan`. A front end emitting the canonical shape directly is responsible /// for maintaining that invariant itself (issue #179). -fn filter_or_fold(pred: Unresolved, child: Unresolved) -> Unresolved { +fn filter_or_fold(pred: Scalar, child: Unresolved) -> Unresolved { match child { Unresolved::Scan { source, mut predicates, schema, } => { - predicates.push(Predicate(Rc::new(pred))); + predicates.push(Predicate(pred)); Unresolved::Scan { source, predicates, @@ -1957,7 +1941,7 @@ fn filter_or_fold(pred: Unresolved, child: Unresolved) -> Unresolved { } } other => Unresolved::Filter { - pred: Predicate(Rc::new(pred)), + pred: Predicate(pred), child: Rc::new(other), }, } @@ -1974,6 +1958,18 @@ fn split_conjunction<'a>(expr: &'a Expr, out: &mut Vec<&'a Expr>) { } } +/// Whether `expr` reads another operator anywhere inside it (`EXISTS`, +/// `IN (…)`, a scalar subquery). +fn reads_subquery(expr: &Expr) -> bool { + expr.exists(|e| { + Ok(matches!( + e, + Expr::ScalarSubquery(_) | Expr::InSubquery(_) | Expr::Exists(_) + )) + }) + .expect("the predicate never fails") +} + /// Re-`AND` the conjuncts, or `None` when there are none left. fn rebuild_conjunction(conjuncts: &[&Expr]) -> Option { conjuncts @@ -2122,9 +2118,9 @@ fn expand_grouping_set(gs: &logical_expr::GroupingSet) -> Vec> { /// The projection also has to carry through the plain columns the aggregate /// still references, since a `Project` replaces its child's schema rather than /// extending it. -#[derive(Default)] -struct DerivedCols { - cols: Vec>, +struct DerivedCols<'l> { + lowerer: &'l SqlLowerer<'l>, + cols: Vec, /// Whether any column is genuinely derived. Without one the aggregate keeps /// its original child, so DAGs that lower today keep their exact shape. any: bool, @@ -2133,12 +2129,21 @@ struct DerivedCols { collision: Option, } -impl DerivedCols { +impl<'l> DerivedCols<'l> { + fn new(lowerer: &'l SqlLowerer<'l>) -> Self { + Self { + lowerer, + cols: Vec::new(), + any: false, + collision: None, + } + } + /// Add `alias := expr`, or note a collision if `alias` already means /// something else. `Project` carries one relation qualifier for all its /// columns, so `a.k` and `b.k` cannot both survive it — but that only /// matters when a projection gets inserted at all. - fn push(&mut self, alias: String, expr: Unresolved) { + fn push(&mut self, alias: String, expr: Scalar) { let existing = self .cols .iter() @@ -2161,12 +2166,12 @@ impl DerivedCols { let Expr::Column(c) = unalias(expr) else { return Ok(()); }; - self.push(c.name.clone(), df_expr_to_unresolved(expr)?); + self.push(c.name.clone(), self.lowerer.lower_expr(expr)?); Ok(()) } /// A genuinely derived column: `alias` now names `expr`'s value. - fn materialize(&mut self, alias: String, expr: Unresolved) -> Result<(), LoweringError> { + fn materialize(&mut self, alias: String, expr: Scalar) -> Result<(), LoweringError> { self.any = true; self.push(alias, expr); Ok(()) @@ -2188,7 +2193,7 @@ impl DerivedCols { let mut rewritten = agg_fn.clone(); for arg in &mut rewritten.params.args { let alias = unalias(arg).to_string(); - self.materialize(alias.clone(), df_expr_to_unresolved(arg)?)?; + self.materialize(alias.clone(), self.lowerer.lower_expr(arg)?)?; *arg = Expr::Column(DfColumn::new_unqualified(alias)); } return Ok(Expr::AggregateFunction(rewritten)); @@ -2211,12 +2216,12 @@ impl DerivedCols { } match agg_col_name(&agg_fn.params.args) { Some(name) => { - self.push(name, df_expr_to_unresolved(arg)?); + self.push(name, self.lowerer.lower_expr(arg)?); Ok(expr.clone()) } None => { let alias = unalias(arg).to_string(); - self.materialize(alias.clone(), df_expr_to_unresolved(arg)?)?; + self.materialize(alias.clone(), self.lowerer.lower_expr(arg)?)?; let mut agg_fn = agg_fn.clone(); agg_fn.params.args[0] = Expr::Column(DfColumn::new_unqualified(alias)); Ok(Expr::AggregateFunction(agg_fn)) @@ -2287,7 +2292,7 @@ fn expr_to_group_ref(expr: &Expr) -> Result { match expr { // Preserve the relation qualifier so a GROUP BY / PARTITION BY key over a // join (`b.k` vs `a.k`) resolves to the correct side — the same rule the - // scalar predicate path uses (`df_expr_to_unresolved`). + // scalar predicate path uses (`lower_expr`). Expr::Column(col) => Ok(match &col.relation { Some(rel) => ColumnRef::Qualified { table: rel.to_string(), diff --git a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs index 23f757985..cc24329cd 100644 --- a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs +++ b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs @@ -35,9 +35,12 @@ //! `Err`, never panics. The pinned per-query outcomes document today's real //! coverage so a regression (or a future improvement) is visible, not silent. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use datafusion::error::DataFusionError; @@ -92,7 +95,7 @@ fn queries() -> Vec { .collect() } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result, LoweringError> { lower_sql_dialect( q, &catalog(), @@ -218,19 +221,19 @@ async fn corpus_lowering_matches_the_pinned_per_query_outcome() { ); } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match node.expect_non_asap() { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } @@ -260,7 +263,7 @@ async fn top_k_queries_are_count_grouped_by_prefix() { idx + 1 ); assert!( - matches!(qe, QueryExpr::Limit { .. }), + matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. }), "q{} ({label}) top-k shape keeps the LIMIT at the root: {qe:?}", idx + 1 ); diff --git a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs index 4c6f0e8ea..d7c750d7d 100644 --- a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs +++ b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs @@ -70,7 +70,7 @@ fn catalog() -> SqlCatalog { .with_table("bgp.bgp_updates", updates) } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result, SqlError> { lower_sql_dialect( q, &catalog(), @@ -91,6 +91,8 @@ async fn lower(q: &str) -> Result { /// is that signal, ratcheted so a category shifting size is visible. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] enum Category { + /// A planned expression lacks a faithful registered IR type contract. + InvalidRepresentation, Lowered, /// `DataFusionError::Plan` -- almost entirely "unknown function" for a /// ClickHouse-only builtin (`uniqExact`, `countIf`, `splitByChar`, ...). @@ -119,6 +121,7 @@ fn categorize(err: &SqlError) -> Category { SqlError::DataFusion(DataFusionError::SQL(_, _)) => Category::Parse, SqlError::DataFusion(DataFusionError::NotImplemented(_)) => Category::NotImplemented, SqlError::UnsupportedFeature(_) => Category::UnsupportedFeature, + SqlError::Convert(_) => Category::InvalidRepresentation, _ => Category::Other, } } @@ -194,12 +197,15 @@ async fn corpus_lowering_matches_the_pinned_aggregate_tally() { // 152 -> 154: `ScalarValue::Interval` (this branch) converts the // `INTERVAL x unit` literal the two `toStartOfInterval(...)` queries // carry. - // 154 -> 156 (DataFusion 54, issue #611): q128 calls `greatest`, which - // DataFusion now provides (was `Plan`), and q129's subquery - // `ORDER BY count(*)` no longer fails as `UnsupportedFeature("expression: - // count(*)")`. - expect(Category::Lowered, 156); - expect(Category::Plan, 39); + // Previously admitted ClickHouse stubs used placeholder Float64 types. + // Unregistered functions and incompatible operands now fail closed. + // DataFusion 54 (issue #611): q128's `greatest` is now provided by + // DataFusion (was `Plan`) and q129's subquery `ORDER BY count(*)` no + // longer fails as `UnsupportedFeature`; one lowers, the other now fails + // closed as `InvalidRepresentation`. + expect(Category::Lowered, 106); + expect(Category::InvalidRepresentation, 54); + expect(Category::Plan, 40); expect(Category::Schema, 0); expect(Category::Parse, 0); // One query that used to fail at `uniqExact` (`Plan`) now clears that @@ -211,7 +217,7 @@ async fn corpus_lowering_matches_the_pinned_aggregate_tally() { // Typed Map access lowers one prior gap; six array accesses now fail // during typed planning because the Map adapter rejects array inputs. expect(Category::NotImplemented, 0); - expect(Category::UnsupportedFeature, 5); + expect(Category::UnsupportedFeature, 0); // Was 2: the two `toStartOfInterval(...)` queries whose `INTERVAL`-literal // conversion gap the `toStartOfInterval` note above describes. Both now // lower end to end and are counted in `Lowered`. diff --git a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs index caed56594..cce394a08 100644 --- a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs +++ b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs @@ -19,9 +19,12 @@ //! Schema: `packets(srcip, dstip, srcport, dstport, proto, time, pkt_len)`; //! flow / 5-tuple = `(srcip, dstip, srcport, dstport, proto)`. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/synthetic_packet_trace_queries.sql"); @@ -67,110 +70,61 @@ fn queries() -> Vec { // ── DAG helpers ────────────────────────────────────────────────────────────── -/// Every `AggIntent` in the DAG, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { - let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - // Scalar expression variants (issue #205): `AggIntent` only ever - // lives in `Aggregate.measures`, never nested inside a scalar - // expression DAG, so there's nothing to recurse into here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} - } - } - go(e, &mut out); - out +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + +/// Every `AggIntent` in the DAG, root-to-leaf (every reachable node — +/// `AggIntent` only ever lives in `Aggregate.measures`). +fn intents(e: &Rc) -> Vec { + OperatorNode::reachable(e) + .iter() + .filter_map(|node| match op(node) { + NonASAPOp::Aggregate { measures, .. } => Some(measures.clone()), + _ => None, + }) + .flatten() + .collect() } /// The first `Aggregate`'s `(by, measures)` along the single-child spine. SQL /// never lowers to `Reduction::PerEntity` (it has no per-series concept), so /// `expect_reduce()` here is a safe, load-bearing assumption for these tests. -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::SQLWindowFunc { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } /// Whether a `SQLWindowFunc` (analytic `OVER (…)`) node appears anywhere. -fn has_window_func(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::SQLWindowFunc { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => has_window_func(child), +fn has_window_func(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::SQLWindowFunc { .. } => true, + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => has_window_func(child), _ => false, } } -async fn lower(q: &str) -> QueryExpr { +async fn lower(q: &str) -> Rc { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) diff --git a/crates/frontend-sql/tests/maintained_population.rs b/crates/frontend-sql/tests/maintained_population.rs index a6347c4aa..dc7d55c44 100644 --- a/crates/frontend-sql/tests/maintained_population.rs +++ b/crates/frontend-sql/tests/maintained_population.rs @@ -2,17 +2,18 @@ use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_types::{ - post_asap::{ - compile_post_asap_dag, - maintained_population::{MaintainedPopulation, PopulationInput}, - share_common_summary_sub_dags, SummaryExpr, ValueOperation, + ir::{ + apply_lifecycle_timings, cse::share_common_sub_dags, + physical_export::compile_physical_asap_dag, ASAPOp, LifecycleAssignment, NonASAPOp, + Operator, OperatorNode, TimingMemo, }, - pre_asap::{DataType, Field, QueryExpr, Schema}, + post_asap::maintained_population::{MaintainedPopulation, PopulationInput}, + pre_asap::{DataType, Field, Schema}, types::AccuracyTarget, }; use std::rc::Rc; -async fn aggregate(q: &str) -> Rc { +async fn aggregate(q: &str) -> Rc { let catalog = SqlCatalog::new().with_table( "samples", Schema::new(vec![ @@ -20,38 +21,38 @@ async fn aggregate(q: &str) -> Rc { Field::plain("job", DataType::Utf8, false), ]), ); - let root = lower_sql(q, &catalog, AccuracyTarget::Exact).await.unwrap(); - Rc::new(root) + lower_sql(q, &catalog, AccuracyTarget::Exact).await.unwrap() } -fn population( - mut node: &asap_types::post_asap::SummaryNode, -) -> ( - &Rc, - &MaintainedPopulation, -) { - while let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { .. }, - .. - } = &node.expr - { +/// The `MaintainPopulation` node a candidate's evaluation reads, and its spec. +fn population(mut node: &OperatorNode) -> (&Rc, &MaintainedPopulation) { + while let Operator::NonASAP(NonASAPOp::Project { child, .. }) = &node.operator { node = child; } - let SummaryExpr::ValueOperation { child, .. } = &node.expr else { - panic!("readout") + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &node.operator else { + panic!("evaluation") }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!("state") }; (child, population) } -// Quantile parameters are readout identity, while source, value column and grouping are state identity. +/// Export `plan` the way the planner does: assign the default lifecycle +/// timings, then compile the timed DAG. +fn compile(plan: &Rc) -> Result<(), String> { + let timed = apply_lifecycle_timings( + plan, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .map_err(|e| e.to_string())?; + compile_physical_asap_dag(&timed) + .map(|_| ()) + .map_err(|e| e.to_string()) +} + +// Quantile parameters are evaluation identity, while source, value column and grouping are state identity. #[tokio::test] async fn sql_quantiles_share_rows_without_promql_lookback() { let roots = vec![ @@ -59,7 +60,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { aggregate("SELECT approx_percentile_cont(latency, 0.99) FROM samples").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -67,7 +68,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); } let (a, spec) = population(&plans[0].1); let (b, _) = population(&plans[1].1); @@ -107,9 +108,9 @@ async fn sql_filters_separate_populations() { assert_ne!(population(&a).1.input, population(&b).1.input); } -// All four scalar readouts can share the same non-null numeric SQL population. +// All four scalar evaluations can share the same non-null numeric SQL population. #[tokio::test] -async fn sql_scalar_readouts_share_membership() { +async fn sql_scalar_evaluations_share_membership() { let mut roots = Vec::new(); for function in [ "median(latency)", @@ -120,7 +121,7 @@ async fn sql_scalar_readouts_share_membership() { roots.push(aggregate(&format!("SELECT {function} FROM samples")).await); } let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -128,27 +129,29 @@ async fn sql_scalar_readouts_share_membership() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } } -// A readout cannot reinterpret a label column as its numeric population. +// A evaluation cannot reinterpret a label column as its numeric population. #[tokio::test] async fn malformed_table_population_fails_validation() { let root = aggregate("SELECT median(latency) FROM samples").await; let rule = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)); let mut candidate = rule.candidate(&root).unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &mut Rc::make_mut(&mut candidate).expr else { + let Operator::NonASAP(NonASAPOp::Project { child, .. }) = + &mut Rc::make_mut(&mut candidate).operator + else { unreachable!() }; - let SummaryExpr::ValueOperation { child, .. } = &mut Rc::make_mut(child).expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = + &mut Rc::make_mut(child).operator + else { unreachable!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &mut Rc::make_mut(child).expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = + &mut Rc::make_mut(child).operator else { unreachable!() }; @@ -156,7 +159,7 @@ async fn malformed_table_population_fails_validation() { unreachable!() }; *value_column = 1; - assert!(compile_post_asap_dag(&candidate).is_err()); + assert!(compile(&candidate).is_err()); } // SQL ORDER BY value DESC LIMIT k uses the same maximum-k state contract. @@ -167,7 +170,7 @@ async fn sql_topk_limits_share_maximum_k() { aggregate("SELECT * FROM samples ORDER BY latency DESC LIMIT 5").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -175,7 +178,7 @@ async fn sql_topk_limits_share_maximum_k() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); assert_eq!(population(plan).1.max_k, 5); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } @@ -187,7 +190,7 @@ async fn sql_topk_over_an_identity_select_list_is_recognized() { let root = aggregate("SELECT latency, job FROM samples ORDER BY latency DESC LIMIT 5").await; let rule = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)); let plan = rule.candidate(&root).expect("SQL topk"); - compile_post_asap_dag(&plan).unwrap(); + compile(&plan).unwrap(); assert_eq!(population(&plan).1.max_k, 5); } diff --git a/crates/frontend-sql/tests/netflow/netflow.rs b/crates/frontend-sql/tests/netflow/netflow.rs index 22236b850..da680d664 100644 --- a/crates/frontend-sql/tests/netflow/netflow.rs +++ b/crates/frontend-sql/tests/netflow/netflow.rs @@ -4,9 +4,12 @@ //! aggregate over a netflow table, a time predicate, optional grouping, //! optional `ORDER BY`/`LIMIT`, plus the nested aggregate shape. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::pre_asap::{AggIntent, GroupKeys}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/netflow.sql"); @@ -109,11 +112,10 @@ async fn netflow_sql_corpus_lowers_to_expected_intents() { ); for (idx, (query, expected)) in queries.iter().zip(EXPECTED).enumerate() { + // A successful `lower_sql` already derived every node's schema. let qe = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|err| panic!("q{} failed to lower:\n{query}\n{err}", idx + 1)); - qe.output_schema() - .unwrap_or_else(|err| panic!("q{} schema derivation failed: {err}", idx + 1)); assert!( has_scan_predicate(&qe), "q{} should retain the netflow time predicate on the Scan: {qe:?}", @@ -123,7 +125,7 @@ async fn netflow_sql_corpus_lowers_to_expected_intents() { } } -fn assert_expected(qe: &QueryExpr, expected: Expected, case_no: usize) { +fn assert_expected(qe: &Rc, expected: Expected, case_no: usize) { match expected { Expected::Quantile { q, by } => { let (actual_by, measures) = first_aggregate(qe).expect("expected Aggregate"); @@ -190,53 +192,58 @@ impl AggKind { } } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } -fn has_scan_predicate(qe: &QueryExpr) -> bool { +fn has_scan_predicate(qe: &Rc) -> bool { any_node( qe, - |node| matches!(node, QueryExpr::Scan { predicates, .. } if !predicates.is_empty()), + |node| matches!(op(node), NonASAPOp::Scan { predicates, .. } if !predicates.is_empty()), ) } -fn has_topk(qe: &QueryExpr, k: usize) -> bool { +fn has_topk(qe: &Rc, k: usize) -> bool { any_node(qe, |node| { matches!( - node, - QueryExpr::Aggregate { measures, .. } + op(node), + NonASAPOp::Aggregate { measures, .. } if measures.iter().any(|agg| matches!(agg, AggIntent::TopK { k: actual, .. } if *actual == k)) ) }) } fn aggregate_by_with( - qe: &QueryExpr, + qe: &Rc, by: &'static [usize], pred: impl Fn(&AggIntent) -> bool, ) -> bool { let expected_by = GroupKeys::by(by.to_vec()); let mut found = false; visit(qe, &mut |node| { - if let QueryExpr::Aggregate { + if let NonASAPOp::Aggregate { reduction, measures, .. - } = node + } = op(node) { found |= *reduction.expect_reduce() == expected_by && measures.iter().any(&pred); } @@ -244,78 +251,26 @@ fn aggregate_by_with( found } -fn all_intents(qe: &QueryExpr) -> Vec { +fn all_intents(qe: &Rc) -> Vec { let mut intents = Vec::new(); visit(qe, &mut |node| { - if let QueryExpr::Aggregate { measures, .. } = node { + if let NonASAPOp::Aggregate { measures, .. } = op(node) { intents.extend(measures.iter().cloned()); } }); intents } -fn any_node(qe: &QueryExpr, pred: impl Fn(&QueryExpr) -> bool) -> bool { +fn any_node(qe: &Rc, pred: impl Fn(&OperatorNode) -> bool) -> bool { let mut found = false; visit(qe, &mut |node| found |= pred(node)); found } -fn visit(qe: &QueryExpr, f: &mut impl FnMut(&QueryExpr)) { - f(qe); - match qe { - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => visit(child, f), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - visit(lhs, f); - visit(rhs, f); - } - QueryExpr::Concat { children, .. } => { - for child in children { - visit(child, f); - } - } - QueryExpr::PromqlVectorFromScalar(child) | QueryExpr::PromqlScalarFromVector(child) => { - visit(child, f) - } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - // Scalar expression variants (issue #205) aren't relational nodes; - // this visitor only walks the relational DAG, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} +/// Every reachable operator node, parents before children — including the +/// operators referenced from scalar positions (subqueries). +fn visit(qe: &Rc, f: &mut impl FnMut(&OperatorNode)) { + for node in OperatorNode::reachable(qe) { + f(&node); } } diff --git a/crates/frontend-sql/tests/pearson_corr.rs b/crates/frontend-sql/tests/pearson_corr.rs index 618f6c38f..d4930c98d 100644 --- a/crates/frontend-sql/tests/pearson_corr.rs +++ b/crates/frontend-sql/tests/pearson_corr.rs @@ -2,7 +2,11 @@ use std::rc::Rc; use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::{AggIntent, DataType, Field, QueryExpr, Schema}; +use asap_types::ir::{ + apply_lifecycle_timings, physical_export::compile_physical_asap_dag, LifecycleAssignment, + NonASAPOp, OperatorNode, ScalarExpr, TimingMemo, +}; +use asap_types::pre_asap::{AggIntent, DataType, Field, Schema}; use asap_types::types::AccuracyTarget; fn catalog() -> SqlCatalog { @@ -16,20 +20,20 @@ fn catalog() -> SqlCatalog { .with_table("b", schema) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> Rc { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap() } -fn aggregate(query: &QueryExpr) -> (&[AggIntent], &QueryExpr) { - match query { - QueryExpr::Aggregate { +fn aggregate(query: &OperatorNode) -> (&[AggIntent], &OperatorNode) { + match query.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => (measures, child), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } => aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } => aggregate(child), other => panic!("expected aggregate, got {other:?}"), } } @@ -46,17 +50,14 @@ async fn corr_materializes_both_arguments() { let query = lower(sql).await; let (measures, child) = aggregate(&query); assert_eq!(measures, &[AggIntent::PearsonCorr { left: 0, right: 1 }]); - let QueryExpr::Project { cols, .. } = child else { + let NonASAPOp::Project { cols, .. } = child.expect_non_asap() else { panic!("derived inputs") }; assert_eq!(cols.len(), 2); assert!(cols .iter() - .any(|col| !matches!(col.expr, QueryExpr::Column(_)))); - assert_eq!( - query.output_schema().unwrap().fields[0].dtype, - DataType::Float64 - ); + .any(|col| !matches!(col.expr, ScalarExpr::Column(_)))); + assert_eq!(query.schema.fields[0].dtype, DataType::Float64); } } @@ -66,11 +67,11 @@ async fn corr_preserves_qualified_join_inputs() { let query = lower("SELECT corr(a.x, b.x) FROM a JOIN b ON a.g = b.g").await; let (measures, child) = aggregate(&query); assert_eq!(measures[0].input_cols(), vec![0, 1]); - let QueryExpr::Project { cols, .. } = child else { + let NonASAPOp::Project { cols, .. } = child.expect_non_asap() else { panic!("paired projection") }; - assert_eq!(cols[0].expr, QueryExpr::Column(0)); - assert_eq!(cols[1].expr, QueryExpr::Column(3)); + assert_eq!(cols[0].expr, ScalarExpr::Column(0)); + assert_eq!(cols[1].expr, ScalarExpr::Column(3)); } // Grouping and sibling reducers cannot drop either correlation argument. @@ -82,12 +83,12 @@ async fn corr_coexists_with_grouping_having_and_other_measures() { .iter() .find(|m| matches!(m, AggIntent::PearsonCorr { .. })) .unwrap(); - let schema = child.output_schema().unwrap(); + let schema = &child.schema; for id in pair.input_cols() { assert!(id < schema.fields.len()); } assert!(measures.iter().any(|m| matches!(m, AggIntent::Sum { .. }))); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[1].name, "r"); assert_eq!(output.fields[1].dtype, DataType::Float64); assert!(output.fields[1].nullable); @@ -99,8 +100,8 @@ async fn corr_repeated_input_and_serialization() { let query = lower("SELECT corr(x, x) FROM a").await; assert_eq!(aggregate(&query).0[0].input_cols(), vec![0, 0]); let encoded = serde_json::to_string(&query).unwrap(); - let decoded: QueryExpr = serde_json::from_str(&encoded).unwrap(); - assert_eq!(query, decoded); + let decoded: OperatorNode = serde_json::from_str(&encoded).unwrap(); + assert_eq!(*query, decoded); } // Unsupported modifiers and window calls fail instead of silently changing semantics. @@ -128,10 +129,10 @@ async fn corr_filter_is_a_measure_filter() { let query = lower("SELECT corr(x, y) FILTER (WHERE g > 0) FROM a").await; let (measures, _) = aggregate(&query); assert_eq!(measures[0].input_cols(), vec![0, 1]); - fn filters(query: &QueryExpr) -> &[Option] { - match query { - QueryExpr::Aggregate { filters, .. } => filters, - QueryExpr::Project { child, .. } | QueryExpr::Filter { child, .. } => filters(child), + fn filters(query: &OperatorNode) -> &[Option] { + match query.expect_non_asap() { + NonASAPOp::Aggregate { filters, .. } => filters, + NonASAPOp::Project { child, .. } | NonASAPOp::Filter { child, .. } => filters(child), other => panic!("expected aggregate, got {other:?}"), } } @@ -145,12 +146,17 @@ async fn corr_filter_is_a_measure_filter() { // Exact fallback retains the complete typed query and compiles to a post-ASAP DAG. #[tokio::test] async fn corr_survives_exact_plan_compilation() { - let query = Rc::new(lower("SELECT corr(x, y) AS r FROM a").await); - let plan = asap_aware_mapping::replacement::keep_pre_asap(&query).unwrap(); + let query = lower("SELECT corr(x, y) AS r FROM a").await; + let plan = asap_aware_mapping::replacement::retain_exact(&query).unwrap(); assert!(plan.guarantee.as_ref().unwrap().is_exact()); - let asap_types::post_asap::SummaryExpr::KeepPreAsap(retained) = &plan.expr else { - panic!("expected exact fallback"); - }; - assert_eq!(aggregate(retained).0, aggregate(&query).0); - asap_types::post_asap::compile_post_asap_dag(&plan).unwrap(); + // The exact fallback is the query's own operator DAG, no ASAP node added. + assert!(!plan.contains_asap(), "expected exact fallback"); + assert_eq!(aggregate(&plan).0, aggregate(&query).0); + let timed = apply_lifecycle_timings( + &plan, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .unwrap(); + compile_physical_asap_dag(&timed).unwrap(); } diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index bac54c54d..8e88d0742 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -1,15 +1,24 @@ -//! End-to-end SQL → unresolved → canonical DAG lowering tests (positional IR). +//! End-to-end SQL → unresolved → resolved operator DAG lowering tests. //! //! Validates the DataFusion front end: SQL parses + plans, lowers directly to -//! the canonical, unresolved shape (`QueryExpr`, issue #179), and -//! the shared `resolve_root` produces the positional, resolved canonical -//! DAG (the same resolver the PromQL path uses). - -use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; +//! the name-based `UnresolvedOp` tree (issue #179), and the shared +//! `resolve_root` produces the positional, canonical `OperatorNode` DAG (the +//! same resolver the PromQL path uses). Every node's schema is derived during +//! resolution, so a successful `lower` already proves schema derivation is +//! total over the tree. + +use asap_types::ir::Predicate; +use std::rc::Rc; + +use asap_frontend_common::{UnresolvedOp, UnresolvedScalar}; +use asap_frontend_sql::{ + lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError, SqlLowerer, +}; +use asap_types::ir::{ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr}; use asap_types::pre_asap::schema::{DataType, Field, FieldDataType, Schema}; use asap_types::pre_asap::{ - AggIntent, CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, Reduction, ScalarValue, - Source, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, + AggIntent, CompareOpKind, GroupKeys, JoinKind, Reduction, ScalarValue, Source, + WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -43,37 +52,21 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> Rc { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) } +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + #[tokio::test] -async fn planning_subquery_bridge_reuses_canonical_promql_subquery() { - let query = lower( - "SELECT max(value) FROM (\ - SELECT asap_promql_subquery(21600000, 60000) AS value FROM (\ - SELECT sum(bytes) AS value FROM metrics))", - ) - .await; - let QueryExpr::Project { child, .. } = query else { - panic!("expected outer SQL projection"); - }; - let QueryExpr::Aggregate { child, .. } = child.as_ref() else { - panic!("expected outer max aggregate, got {child:?}"); - }; - let QueryExpr::PromqlSubquery { - range, - resolution, - child, - } = child.as_ref() - else { - panic!("expected canonical subquery bridge, got {child:?}"); - }; - assert_eq!(*range, std::time::Duration::from_secs(6 * 60 * 60)); - assert_eq!(*resolution, Some(std::time::Duration::from_secs(60))); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); +async fn planning_subquery_bridge_rejects_a_relation_without_vector_conversion() { + let result = lower_sql("SELECT max(value) FROM (SELECT asap_promql_subquery(21600000, 60000) AS value FROM (SELECT sum(bytes) AS value FROM metrics))", &catalog(), AccuracyTarget::Exact).await; + assert!(result.is_err()); } #[tokio::test] @@ -83,12 +76,12 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { SELECT service AS le, sum(bytes) AS value FROM metrics GROUP BY service)", ) .await; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = query + } = op(&query) else { panic!("expected canonical histogram aggregate"); }; @@ -98,7 +91,7 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { measures.as_slice(), [AggIntent::HistogramQuantile { q, le: 0 }] if (*q - 0.95).abs() < 1e-12 )); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); + assert!(matches!(op(child), NonASAPOp::Project { .. })); } #[tokio::test] @@ -134,31 +127,31 @@ async fn planning_relation_bridges_reject_ambiguous_shapes() { } /// Find the first `Aggregate` node along the single-child spine. -fn find_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn find_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_aggregate(child), _ => None, } } /// The first `Aggregate` node itself, for tests that need its child. -fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Aggregate { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find_aggregate_node(child), +fn find_aggregate_node(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Aggregate { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => find_aggregate_node(child), _ => None, } } @@ -166,47 +159,47 @@ fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { /// The names of the columns the first `Aggregate`'s reducers read, resolved /// against its child's schema, plus whether that child is a materializing /// `Project` (issue #110). -fn reducer_input_names(qe: &QueryExpr) -> (Vec, bool) { - let QueryExpr::Aggregate { +fn reducer_input_names(node: &OperatorNode) -> (Vec, bool) { + let NonASAPOp::Aggregate { measures, child, .. - } = find_aggregate_node(qe).expect("expected an Aggregate") + } = op(find_aggregate_node(node).expect("expected an Aggregate")) else { unreachable!() }; - let schema = child.output_schema().expect("child schema"); + let schema = &child.schema; let names = measures .iter() .flat_map(|a| a.input_cols()) .map(|id| schema.fields[id].name.clone()) .collect(); - (names, matches!(**child, QueryExpr::Project { .. })) + (names, matches!(op(child), NonASAPOp::Project { .. })) } /// Find the first `Join` node along the single-child spine. -fn find_join(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Join { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_join(child), +fn find_join(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Join { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_join(child), _ => None, } } /// The first `Filter` node along the single-child spine. -fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Filter { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_filter(child), +fn find_filter(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Filter { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_filter(child), _ => None, } } @@ -215,14 +208,14 @@ fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { async fn where_folds_predicate_onto_scan() { // WHERE folds onto the Scan predicates, below the SELECT projection. let qe = lower("SELECT * FROM metrics WHERE service = 'api'").await; - let QueryExpr::Project { child, .. } = &qe else { + let NonASAPOp::Project { child, .. } = op(&qe) else { panic!("expected Project at root, got {qe:?}"); }; - let QueryExpr::Scan { + let NonASAPOp::Scan { source, predicates, schema, - } = child.as_ref() + } = op(child) else { panic!("expected Scan under the projection, got {child:?}"); }; @@ -257,9 +250,7 @@ async fn projection_over_aggregate_resolves_output_types_via_output_names() { // onto the canonical Aggregate so the Project resolves real types — not // the Utf8 fallback that an unresolved column would get. let qe = lower("SELECT SUM(bytes), AVG(latency) FROM metrics").await; - let schema = qe - .output_schema() - .expect("root projection schema derivation"); + let schema = &qe.schema; assert_eq!(schema.fields.len(), 2); assert_eq!( schema.fields[0].dtype, @@ -287,7 +278,7 @@ async fn single_agg_group_by_keeps_key_in_output_schema() { )); // Both the group key and the aggregate resolve in the root projection schema. - let schema = qe.output_schema().expect("root projection schema"); + let schema = &qe.schema; assert_eq!(schema.fields.len(), 2); assert_eq!( schema.fields[0].dtype, @@ -318,7 +309,7 @@ async fn count_ranked_topk_is_heavy_hitter() { "count-ranked topk → heavy-hitter TopK, got {measures:?}" ); // The inner child is the explicit Count, grouped by service (col 1). - let QueryExpr::Aggregate { child, .. } = &qe else { + let NonASAPOp::Aggregate { child, .. } = op(&qe) else { panic!("expected outer Aggregate, got {qe:?}"); }; let (inner_by, inner_measures) = find_aggregate(child).expect("expected inner Count aggregate"); @@ -429,7 +420,7 @@ async fn select_distinct_lowers_to_distinct_with_positional_cols() { // (not name-based ColumnRefs). DataFusion's `Distinct::All` dedups on every // column, so `cols` is empty here — but the field type is now `Vec`. let qe = lower("SELECT DISTINCT service FROM metrics").await; - let QueryExpr::Dedup { cols, .. } = &qe else { + let NonASAPOp::Dedup { cols, .. } = op(&qe) else { panic!("expected a Dedup at the root, got {qe:?}"); }; let _: &Vec = cols; // compile-time: positional ids, not ColumnRefs @@ -444,34 +435,35 @@ async fn inner_join_lowers_to_join_over_two_scans() { FROM metrics JOIN hosts ON metrics.service = hosts.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the DAG"); - let QueryExpr::Join { + let join = find_join(&qe).expect("expected a Join in the tree"); + let NonASAPOp::Join { kind, left, right, .. - } = join + } = op(join) else { unreachable!("find_join only returns Join"); }; assert_eq!(*kind, JoinKind::Inner); - assert!(matches!(left.as_ref(), QueryExpr::Scan { .. })); - assert!(matches!(right.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(op(left), NonASAPOp::Scan { .. })); + assert!(matches!(op(right), NonASAPOp::Scan { .. })); } /// The two `ColumnId`s an equijoin predicate `Column(l) = Column(r)` binds to, /// returned sorted so the assertion is independent of left/right ordering. -fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { - let QueryExpr::Join { pred, .. } = join else { +fn join_eq_columns(join: &OperatorNode) -> [usize; 2] { + let NonASAPOp::Join { pred, .. } = op(join) else { unreachable!("expected a Join"); }; - let QueryExpr::Compare { + let ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, - } = pred.0.as_ref() + .. + } = &pred.0 else { panic!("expected an equijoin Compare, got {:?}", pred.0); }; match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(l), QueryExpr::Column(r)) => { + (ScalarExpr::Column(l), ScalarExpr::Column(r)) => { let mut cols = [*l, *r]; cols.sort_unstable(); cols @@ -570,12 +562,12 @@ async fn qualified_where_over_join_resolves_to_right_side() { ) .await; let filter = find_filter(&qe).expect("expected a Filter over the join"); - let QueryExpr::Filter { pred, .. } = filter else { + let NonASAPOp::Filter { pred, .. } = op(filter) else { unreachable!("find_filter only returns Filter"); }; assert!( - matches!(pred.0.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Eq, .. } - if matches!(left.as_ref(), QueryExpr::Column(4))), + matches!(&pred.0, ScalarExpr::Compare { left, op: CompareOpKind::Eq, .. } + if matches!(left.as_ref(), ScalarExpr::Column(4))), "hosts.service must bind to concatenated position 4 (not the first `service`), got {:?}", pred.0 ); @@ -639,20 +631,24 @@ async fn aggregate_over_join_binds_against_concatenated_schema() { } // ── Issue #111: IN / EXISTS subquery predicates become semi / anti joins ──── +// +// The front end now leaves them as `UnresolvedScalar::{InSubquery, Exists}` +// filter conjuncts; the shared `canonicalize` pass (run by `resolve_root`) +// lowers each to the semi-/anti-join, so the resolved DAG a test sees is the +// same join shape the front end used to emit directly. /// The first `Join` node's `(kind, predicate, left column count)`. -fn join_parts(qe: &QueryExpr) -> (&JoinKind, &QueryExpr, usize) { - let QueryExpr::Join { +fn join_parts(node: &OperatorNode) -> (&JoinKind, &ScalarExpr, usize) { + let NonASAPOp::Join { kind, pred, left, right: _, - } = find_join(qe).expect("expected a Join") + } = op(find_join(node).expect("expected a Join")) else { unreachable!() }; - let left_len = left.output_schema().expect("left schema").fields.len(); - (kind, pred.0.as_ref(), left_len) + (kind, &pred.0, left.schema.fields.len()) } #[tokio::test] @@ -667,14 +663,15 @@ async fn in_subquery_lowers_to_a_semi_join() { // The predicate resolves against `left ++ right`. Both relations have a // `service` column, so a name-based lookup would bind *both* sides to the // left's — silently making this `service = service`, always true. The key is - // projected under a synthetic name to make that impossible. - let QueryExpr::Compare { left, right, .. } = pred else { + // bound positionally to the subquery's column (right after the left's), + // which makes that impossible. + let ScalarExpr::Compare { left, right, .. } = pred else { panic!("expected a comparison, got {pred:?}"); }; - assert_eq!(**left, QueryExpr::Column(1), "outer service"); + assert_eq!(**left, ScalarExpr::Column(1), "outer service"); assert_eq!( **right, - QueryExpr::Column(left_len), + ScalarExpr::Column(left_len), "the subquery key, not the outer column again" ); } @@ -685,20 +682,14 @@ async fn a_semi_join_outputs_only_the_left_schema() { let qe = lower("SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)").await; let join = find_join(&qe).expect("expected a Join"); - let names: Vec<_> = join - .output_schema() - .expect("join schema") - .fields - .iter() - .map(|c| c.name.clone()) - .collect(); + let names: Vec<_> = join.schema.fields.iter().map(|c| c.name.clone()).collect(); assert_eq!(names, ["ts", "service", "latency", "bytes"]); } #[tokio::test] async fn a_subquery_key_that_is_an_expression_still_binds() { - // `SELECT bytes + 1 …` has no column name of its own; it is projected under - // the synthetic key rather than becoming an unreferenceable `col_0`. + // `SELECT bytes + 1 …` has no column name of its own; the join key binds + // to it positionally rather than through an unreferenceable `col_0`. let qe = lower("SELECT service FROM metrics WHERE bytes IN (SELECT bytes + 1 FROM metrics)").await; assert_eq!(join_parts(&qe).0, &JoinKind::Semi); @@ -728,13 +719,13 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { AND service IN (SELECT service FROM hosts)", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } => scan_has_predicate(child), - QueryExpr::Join { left, right, .. } => { + fn scan_has_predicate(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } => scan_has_predicate(child), + NonASAPOp::Join { left, right, .. } => { scan_has_predicate(left) || scan_has_predicate(right) } _ => false, @@ -748,16 +739,16 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { } /// Find the first `SQLWindowFunc` node along the single-child spine. -fn find_windowfunc(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::SQLWindowFunc { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_windowfunc(child), +fn find_windowfunc(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::SQLWindowFunc { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_windowfunc(child), _ => None, } } @@ -771,12 +762,12 @@ async fn window_function_lowers_to_positional_windowfunc() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { + let NonASAPOp::SQLWindowFunc { func, partition_by, order_by, .. - } = win + } = op(win) else { unreachable!("find_windowfunc only returns SQLWindowFunc"); }; @@ -785,14 +776,14 @@ async fn window_function_lowers_to_positional_windowfunc() { assert_eq!(order_by.len(), 1); assert_eq!( order_by[0].expr, - QueryExpr::Column(3), + ScalarExpr::Column(3), "ORDER BY bytes → col 3" ); assert!(!order_by[0].ascending, "DESC"); // The window output column is appended to the schema (Int64 for ROW_NUMBER), // and the enclosing projection resolves it (output_name threading). - let schema = qe.output_schema().expect("root schema"); + let schema = &qe.schema; assert!( schema.fields.iter().any(|c| c.dtype == DataType::Int64), "row_number output column present, got {:?}", @@ -804,11 +795,11 @@ async fn window_function_lowers_to_positional_windowfunc() { async fn window_aggregate_lowers_to_windowfunc() { let qe = lower("SELECT service, SUM(bytes) OVER (PARTITION BY service) FROM metrics").await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::Sum); - assert_eq!(args, &vec![QueryExpr::Column(3)], "SUM(bytes) → arg col 3"); + assert_eq!(args, &vec![ScalarExpr::Column(3)], "SUM(bytes) → arg col 3"); } // ── Window frames (issue #268) ─────────────────────────────────────────────── @@ -832,8 +823,8 @@ async fn window_frame_is_captured_not_dropped() { ) .await; - let frame_of = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { frame, .. } = find_windowfunc(qe).unwrap() else { + let frame_of = |node: &OperatorNode| { + let NonASAPOp::SQLWindowFunc { frame, .. } = op(find_windowfunc(node).unwrap()) else { unreachable!(); }; frame @@ -873,9 +864,9 @@ async fn range_interval_frame_is_preserved() { RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW) FROM metrics", ) .await; - let QueryExpr::SQLWindowFunc { + let NonASAPOp::SQLWindowFunc { frame: Some(frame), .. - } = find_windowfunc(&qe).unwrap() + } = op(find_windowfunc(&qe).unwrap()) else { panic!("expected a window function with a concrete frame"); }; @@ -905,10 +896,10 @@ async fn range_numeric_frames_remain_scalar_offsets() { ) .await; - let start_bound = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { + let start_bound = |node: &OperatorNode| { + let NonASAPOp::SQLWindowFunc { frame: Some(frame), .. - } = find_windowfunc(qe).unwrap() + } = op(find_windowfunc(node).unwrap()) else { panic!("expected a window function with a concrete frame"); }; @@ -943,43 +934,17 @@ async fn groups_frame_is_rejected() { // ── Nested query functions: derived tables / inline views (issue #27) ─────────── -/// Collect every `AggIntent` in the DAG, root-to-leaf. -fn all_intents(qe: &QueryExpr) -> Vec { - let mut out = Vec::new(); - fn go(qe: &QueryExpr, out: &mut Vec) { - match qe { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - _ => {} - } - } - go(qe, &mut out); - out +/// Collect every `AggIntent` in the DAG, root-to-leaf (every reachable node, +/// including operators referenced from scalar positions). +fn all_intents(root: &Rc) -> Vec { + OperatorNode::reachable(root) + .iter() + .filter_map(|node| match op(node) { + NonASAPOp::Aggregate { measures, .. } => Some(measures.clone()), + _ => None, + }) + .flatten() + .collect() } #[tokio::test] @@ -1002,9 +967,9 @@ async fn derived_table_aggregate_over_aggregate_nests() { intents.iter().any(|i| matches!(i, AggIntent::Sum { .. })), "inner SUM survives, got {intents:?}" ); - // The whole nested DAG's output schema derives without error (positional - // resolution is total across the derived-table boundary). - assert_eq!(qe.output_schema().unwrap().fields.len(), 1); + // The whole nested tree's output schema derives (positional resolution + // is total across the derived-table boundary). + assert_eq!(qe.schema.fields.len(), 1); } #[tokio::test] @@ -1042,26 +1007,23 @@ async fn filter_over_derived_aggregate_resolves_alias_column() { assert!(all_intents(&qe) .iter() .any(|i| matches!(i, AggIntent::Sum { .. }))); - // Schema derivation is total across the boundary. - let _ = qe.output_schema().expect("nested schema derivation"); + // Schema derivation is total across the boundary: the root carries one. + assert_eq!(qe.schema.fields.len(), 2); } #[tokio::test] -async fn scalar_subquery_in_predicate_is_rejected() { - // A subquery-*valued* expression (`x > (SELECT …)`) needs a subquery node in - // the unresolved expression IR (and a correlated/uncorrelated decision); - // rejected cleanly until that lands. Derived tables in FROM (the common nesting - // shape) ARE supported — see the tests above. - let res = lower_sql( - "SELECT service FROM metrics WHERE bytes > (SELECT AVG(bytes) FROM metrics)", - &catalog(), - AccuracyTarget::Exact, - ) - .await; +async fn scalar_subquery_in_predicate_lowers_through_a_cross_join() { + let qe = + lower("SELECT service FROM metrics WHERE bytes > (SELECT AVG(bytes) FROM metrics)").await; + let filter = find_filter(&qe).unwrap(); + let NonASAPOp::Filter { pred, child } = op(filter) else { + panic!() + }; + assert!(matches!(op(child), NonASAPOp::Scan { .. })); assert!( - res.is_err(), - "scalar subquery in predicate should be rejected" + matches!(&pred.0,ScalarExpr::Compare { right,.. } if matches!(right.as_ref(),ScalarExpr::ScalarSubquery(_))) ); + qe.validate_structure().unwrap(); } #[tokio::test] @@ -1077,15 +1039,15 @@ async fn correlated_exists_lifts_its_correlation_into_the_join() { .await; let (kind, pred, left_len) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - let QueryExpr::Compare { left, right, .. } = pred else { + let ScalarExpr::Compare { left, right, .. } = pred else { panic!("expected the correlation as a comparison, got {pred:?}"); }; assert_eq!( **left, - QueryExpr::Column(left_len), + ScalarExpr::Column(left_len), "h.service (right side)" ); - assert_eq!(**right, QueryExpr::Column(1), "m.service (left side)"); + assert_eq!(**right, ScalarExpr::Column(1), "m.service (left side)"); } #[tokio::test] @@ -1104,24 +1066,62 @@ async fn an_uncorrelated_exists_is_an_unconditional_semi_join() { let qe = lower("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; let (kind, pred, _) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - assert_eq!(*pred, QueryExpr::Literal(ScalarValue::Boolean(true))); + assert_eq!(*pred, ScalarExpr::Literal(ScalarValue::Boolean(true))); } #[tokio::test] -async fn not_in_subquery_is_rejected_rather_than_mislowered_as_an_anti_join() { - // `NOT IN` is *not* an anti-join. Under three-valued logic a single NULL - // among the subquery's rows makes `c NOT IN (…)` UNKNOWN for every `c`, so - // the query returns nothing — while an anti-join returns every unmatched - // left row. Rejecting is the only correct option until the nullability is - // proven, and `NOT EXISTS` is the safe spelling. - let err = lower_sql( - "SELECT service FROM metrics WHERE service NOT IN (SELECT service FROM hosts)", - &catalog(), - AccuracyTarget::Exact, +async fn where_exists_resolves_to_a_semi_join_over_the_subquery() { + // The front end emits `Filter { Exists(s) }`; the resolved DAG is the + // `Semi` join with the subquery (a filtered `hosts` scan) on the right. + let qe = lower( + "SELECT service FROM metrics WHERE EXISTS (SELECT service FROM hosts WHERE region = 'eu')", ) - .await - .expect_err("NOT IN must not lower to an anti-join"); - assert!(format!("{err}").contains("NOT IN"), "got {err}"); + .await; + let NonASAPOp::Project { child, .. } = op(&qe) else { + panic!("expected the SELECT list as a Project, got {qe:?}"); + }; + let NonASAPOp::Join { + kind, + pred, + left, + right, + } = op(child) + else { + panic!("expected the Semi join directly under the Project, got {child:?}"); + }; + assert_eq!(*kind, JoinKind::Semi); + assert_eq!(pred.0, ScalarExpr::Literal(ScalarValue::Boolean(true))); + assert!( + matches!(op(left), NonASAPOp::Scan { .. }), + "left is metrics" + ); + let NonASAPOp::Project { child: scan, .. } = op(right) else { + panic!("expected the subquery's projection on the right, got {right:?}"); + }; + assert!( + matches!(op(scan), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1), + "the subquery's WHERE stays on its own Scan, got {scan:?}" + ); + assert_eq!( + child.schema.fields.len(), + 4, + "a semi join outputs the left's columns alone" + ); +} + +#[tokio::test] +async fn not_in_subquery_is_rejected_rather_than_mislowered_as_an_anti_join() { + let qe = + lower("SELECT service FROM metrics WHERE service NOT IN (SELECT service FROM hosts)").await; + let filter = find_filter(&qe).unwrap(); + let NonASAPOp::Filter { pred, .. } = op(filter) else { + panic!() + }; + assert!(matches!( + pred.0, + ScalarExpr::InSubquery { negated: true, .. } + )); + qe.validate_structure().unwrap(); } #[tokio::test] @@ -1137,6 +1137,177 @@ async fn a_correlated_in_subquery_is_rejected() { assert!(format!("{err}").contains("correlated IN"), "got {err}"); } +// ── Subquery-valued expressions at the `UnresolvedOp` level ───────────────── + +/// `SqlLowerer::lower` output, before `resolve_root`. +async fn lower_unresolved(sql: &str) -> UnresolvedOp { + let catalog = catalog(); + SqlLowerer::new(&catalog) + .lower(sql, &AccuracyTarget::Exact) + .await + .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) +} + +#[tokio::test] +async fn scalar_subquery_in_projection_lowers_to_a_scalar_subquery_item() { + // An uncorrelated `(SELECT max(v) FROM t2)` in the SELECT list is a + // `ScalarSubquery` projection item reading its own lowered plan; the + // cross-join rewrite is `canonicalize`'s job, not the front end's. + let tree = lower_unresolved("SELECT (SELECT max(latency) FROM metrics) FROM hosts").await; + let UnresolvedOp::Project { cols, child, .. } = &tree else { + panic!("expected the SELECT list as a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Scan { source: Source::Table { table_ref }, .. } + if table_ref == "hosts"), + "the outer relation stays the projection's child, got {child:?}" + ); + assert_eq!(cols.len(), 1); + let UnresolvedScalar::ScalarSubquery(sub) = &cols[0].expr else { + panic!("expected a ScalarSubquery item, got {:?}", cols[0].expr); + }; + let UnresolvedOp::Project { child: inner, .. } = sub.as_ref() else { + panic!("expected the subquery's own SELECT list, got {sub:?}"); + }; + assert!( + matches!(inner.as_ref(), UnresolvedOp::Aggregate { measures, .. } + if matches!(measures.as_slice(), [AggIntent::Max { .. }])), + "the subquery plan is lowered as a root of its own, got {inner:?}" + ); +} + +#[tokio::test] +async fn exists_and_in_subqueries_lower_to_scalar_filter_conjuncts() { + // The front end no longer builds the semi join itself: `EXISTS` / `IN + // (…)` are `Filter` predicates reading the subquery operator. + let tree = + lower_unresolved("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; + let UnresolvedOp::Project { child, .. } = &tree else { + panic!("expected a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } + if matches!(pred.0, UnresolvedScalar::Exists { negated: false, .. })), + "expected Filter {{ Exists }}, got {child:?}" + ); + + let tree = lower_unresolved( + "SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)", + ) + .await; + let UnresolvedOp::Project { child, .. } = &tree else { + panic!("expected a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } + if matches!(pred.0, UnresolvedScalar::InSubquery { negated: false, .. })), + "expected Filter {{ InSubquery }}, got {child:?}" + ); +} + +// ── `SELECT` without `FROM`, unary minus, SQL expression semantics ────────── + +#[tokio::test] +async fn select_without_from_projects_over_one_empty_row() { + // `SELECT 1` has no table: DataFusion's `EmptyRelation` is one empty + // input row, which the SELECT list projects a literal over. + let qe = lower("SELECT 1").await; + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert_eq!(cols.len(), 1); + assert_eq!(cols[0].expr, ScalarExpr::Literal(ScalarValue::Int64(1))); + let NonASAPOp::Values { rows, schema } = op(child) else { + panic!("expected Values under the Project, got {child:?}"); + }; + assert_eq!(rows, &vec![Vec::::new()], "one empty row"); + assert!(schema.fields.is_empty() && schema.closed); + assert_eq!(qe.schema.fields.len(), 1); + assert_eq!(qe.schema.fields[0].dtype, DataType::Int64); +} + +#[tokio::test] +async fn values_lowers_to_one_row_per_values_row() { + let qe = lower("SELECT * FROM (VALUES (1, 'a'), (2, 'b')) AS v(n, s)").await; + let values = OperatorNode::reachable(&qe) + .into_iter() + .find(|n| matches!(op(n), NonASAPOp::Values { .. })) + .expect("expected a Values node"); + let NonASAPOp::Values { rows, schema } = op(&values) else { + unreachable!() + }; + assert_eq!(rows.len(), 2); + assert_eq!( + rows[1], + vec![ + ScalarExpr::Literal(ScalarValue::Int64(2)), + ScalarExpr::Literal(ScalarValue::Utf8("b".into())), + ] + ); + assert_eq!(schema.fields.len(), 2); + assert_eq!(schema.fields[0].dtype, DataType::Int64); + assert_eq!(schema.fields[1].dtype, DataType::Utf8); + assert_eq!( + qe.schema + .fields + .iter() + .map(|f| f.name.as_str()) + .collect::>(), + ["n", "s"] + ); +} + +#[tokio::test] +async fn unary_minus_lowers_to_negative() { + // `-x` over a column is the `Negative` scalar (a negative *literal* is + // folded by DataFusion's planner before lowering). + let qe = lower("SELECT -latency FROM metrics").await; + let NonASAPOp::Project { cols, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert_eq!( + cols[0].expr, + ScalarExpr::Negative { + expr: Box::new(ScalarExpr::Column(2)), + semantics: ExprSemantics::Sql, + } + ); + assert_eq!(qe.schema.fields[0].dtype, DataType::Float64); +} + +#[tokio::test] +async fn sql_comparisons_and_arithmetic_carry_sql_semantics() { + let qe = lower("SELECT bytes * 8 FROM metrics WHERE latency > 1.5").await; + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert!( + matches!( + &cols[0].expr, + ScalarExpr::Arithmetic { + semantics: ExprSemantics::Sql, + .. + } + ), + "got {:?}", + cols[0].expr + ); + let NonASAPOp::Scan { predicates, .. } = op(child) else { + panic!("expected the WHERE folded onto the Scan, got {child:?}"); + }; + assert!( + matches!( + &predicates[0].0, + ScalarExpr::Compare { + semantics: ExprSemantics::Sql, + .. + } + ), + "got {:?}", + predicates[0].0 + ); +} + // ── Issue #115: Quantile / Cardinality carry their input column ───────────── #[tokio::test] @@ -1286,20 +1457,20 @@ async fn time_bucketing_group_by_lowers_to_a_derived_key() { let qe = lower("SELECT date_trunc('minute', ts) AS m, SUM(bytes) FROM metrics GROUP BY m").await; let node = find_aggregate_node(&qe).expect("expected an Aggregate"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = node + } = op(node) else { unreachable!() }; assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(op(child), NonASAPOp::Project { .. }), "expected a materializing Project beneath the Aggregate" ); - let schema = child.output_schema().expect("child schema"); + let schema = &child.schema; assert_eq!(reduction, &Reduction::by(vec![0])); assert!( schema.fields[0].name.contains("date_trunc"), @@ -1322,14 +1493,14 @@ async fn time_bucketing_keeps_the_scan_predicate() { WHERE bytes > 10 GROUP BY m", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_has_predicate(child), + fn scan_has_predicate(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => scan_has_predicate(child), _ => false, } } @@ -1346,13 +1517,13 @@ async fn a_plain_group_by_inserts_no_projection() { "SELECT COUNT(*) FROM metrics", ] { let qe = lower(q).await; - let QueryExpr::Aggregate { child, .. } = - find_aggregate_node(&qe).expect("expected an Aggregate") + let NonASAPOp::Aggregate { child, .. } = + op(find_aggregate_node(&qe).expect("expected an Aggregate")) else { unreachable!() }; assert!( - !matches!(**child, QueryExpr::Project { .. }), + !matches!(op(child), NonASAPOp::Project { .. }), "{q} should not gain a projection" ); } @@ -1361,14 +1532,14 @@ async fn a_plain_group_by_inserts_no_projection() { #[tokio::test] async fn a_shared_expression_is_materialized_once() { let qe = lower("SELECT SUM(bytes * 2), MIN(bytes * 2) FROM metrics").await; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = find_aggregate_node(&qe).expect("expected an Aggregate") + } = op(find_aggregate_node(&qe).expect("expected an Aggregate")) else { unreachable!() }; assert_eq!( - child.output_schema().expect("child schema").fields.len(), + child.schema.fields.len(), 1, "the two reducers should share one derived column" ); @@ -1378,38 +1549,32 @@ async fn a_shared_expression_is_materialized_once() { // ── Issue #118: multi-level grouping expands into one Aggregate per level ─── /// The branches of the first `Concat` along the single-child spine. -fn merge_branches(qe: &QueryExpr) -> &Vec { - fn find(qe: &QueryExpr) -> Option<&Vec> { - match qe { - QueryExpr::Concat { children, .. } => Some(children), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find(child), +fn merge_branches(node: &OperatorNode) -> &Vec> { + fn find(node: &OperatorNode) -> Option<&Vec>> { + match op(node) { + NonASAPOp::Concat { children, .. } => Some(children), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => find(child), _ => None, } } - find(qe).expect("expected a Concat") + find(node).expect("expected a Concat") } /// `(group keys, column names)` of each merged grouping level. -fn grouping_levels(qe: &QueryExpr) -> Vec<(GroupKeys, Vec)> { - merge_branches(qe) +fn grouping_levels(node: &OperatorNode) -> Vec<(GroupKeys, Vec)> { + merge_branches(node) .iter() .map(|b| { - let QueryExpr::Project { child, .. } = b else { + let NonASAPOp::Project { child, .. } = op(b) else { panic!("expected a Project per level, got {b:?}"); }; - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { reduction, .. } = op(child) else { panic!("expected an Aggregate under the Project, got {child:?}"); }; - let names = b - .output_schema() - .expect("level schema") - .fields - .iter() - .map(|c| c.name.clone()) - .collect(); + let names = b.schema.fields.iter().map(|c| c.name.clone()).collect(); (reduction.expect_reduce().clone(), names) }) .collect() @@ -1468,9 +1633,7 @@ async fn omitted_grouping_keys_become_typed_nulls() { } // The `()` level projects `service` as a Utf8 null, not a Float64 one. - let schema = merge_branches(&qe)[1] - .output_schema() - .expect("level schema"); + let schema = &merge_branches(&qe)[1].schema; assert_eq!(schema.fields[0].name, "service"); assert_eq!( schema.fields[0].dtype, @@ -1488,8 +1651,7 @@ async fn grouping_levels_are_union_compatible() { let shapes: Vec<_> = merge_branches(&qe) .iter() .map(|b| { - b.output_schema() - .expect("level schema") + b.schema .fields .iter() .map(|c| (c.name.clone(), c.dtype.clone())) @@ -1539,12 +1701,12 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { // #110's materializing Project sits beneath every level's Aggregate. let qe = lower("SELECT service, SUM(bytes * 8) FROM metrics GROUP BY ROLLUP(service)").await; for b in merge_branches(&qe) { - let QueryExpr::Project { child, .. } = b else { + let NonASAPOp::Project { child, .. } = op(b) else { panic!("expected a Project per level"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = op(child) else { panic!("expected an Aggregate"); }; @@ -1553,7 +1715,7 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { [AggIntent::Sum { col: Some(_) }] )); assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(op(child), NonASAPOp::Project { .. }), "the derived-column projection should sit under each level" ); } @@ -1616,7 +1778,7 @@ async fn array_agg_is_deliberately_rejected() { // ── Issue #225: catalog-driven ClickHouse builtins (countIf, generalizing // uniqExact from #221) ─────────────────────────────────────────────────── -async fn lower_clickhouse(sql: &str) -> QueryExpr { +async fn lower_clickhouse(sql: &str) -> Rc { lower_sql_dialect( sql, &catalog(), @@ -1627,20 +1789,20 @@ async fn lower_clickhouse(sql: &str) -> QueryExpr { .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) } -fn temporal_aggregate(qe: &QueryExpr) -> (&AggIntent, std::time::Duration, &QueryExpr) { - match qe { - QueryExpr::Aggregate { +fn temporal_aggregate(node: &OperatorNode) -> (&AggIntent, std::time::Duration, &OperatorNode) { + match op(node) { + NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, child, .. } => { - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = op(child) else { panic!("temporal Aggregate must directly wrap TimeRange, got {child:?}"); }; (&measures[0], *range, child) } - QueryExpr::Project { child, .. } | QueryExpr::Filter { child, .. } => { + NonASAPOp::Project { child, .. } | NonASAPOp::Filter { child, .. } => { temporal_aggregate(child) } other => panic!("expected temporal Aggregate, got {other:?}"), @@ -1661,15 +1823,15 @@ async fn explicit_temporal_aggregates_share_promql_intents_and_timerange() { let (intent, range, child) = temporal_aggregate(&qe); assert_eq!(intent, &expected); assert_eq!(range, std::time::Duration::from_secs(300)); - assert!(matches!(child, QueryExpr::Project { child, .. } - if matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1))); + assert!(matches!(op(child), NonASAPOp::Project { child, .. } + if matches!(op(child), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1))); - let QueryExpr::Project { cols, .. } = &qe else { + let NonASAPOp::Project { cols, .. } = op(&qe) else { panic!("SELECT list must remain a Project, got {qe:?}"); }; - assert!(matches!(cols[0].expr, QueryExpr::Column(2))); + assert!(matches!(cols[0].expr, ScalarExpr::Column(2))); assert_eq!(cols[1].alias.as_deref(), Some("v")); - assert!(matches!(cols[1].expr, QueryExpr::Column(1))); + assert!(matches!(cols[1].expr, ScalarExpr::Column(1))); } } @@ -1830,20 +1992,20 @@ async fn project_filter_and_outer_aggregate_preserve_temporal_child() { ) r WHERE v >= 0", ) .await; - let QueryExpr::Project { child, .. } = &qe else { + let NonASAPOp::Project { child, .. } = op(&qe) else { panic!("expected outer SELECT Project, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: Reduction::Reduce(_), measures, child, .. - } = child.as_ref() + } = op(child) else { panic!("expected outer Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Filter { child, .. } = child.as_ref() else { + let NonASAPOp::Filter { child, .. } = op(child) else { panic!("derived-table WHERE must remain above the inner query, got {child:?}"); }; let (intent, range, _) = temporal_aggregate(child); @@ -2007,13 +2169,13 @@ async fn lag_in_frame_lowers_to_its_own_kind_not_lag() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LagInFrame); assert_eq!( args, - &vec![QueryExpr::Column(3)], + &vec![ScalarExpr::Column(3)], "lagInFrame(bytes) → arg col 3" ); } @@ -2026,7 +2188,7 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LeadInFrame); @@ -2039,16 +2201,16 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { async fn now_in_predicate_lowers_to_current_timestamp() { // WHERE folds onto Scan.predicates (no explicit Filter node). let qe = lower("SELECT * FROM metrics WHERE ts < NOW()").await; - let QueryExpr::Project { child, .. } = &qe else { + let NonASAPOp::Project { child, .. } = op(&qe) else { panic!("expected Project at root, got {qe:?}"); }; - let QueryExpr::Scan { predicates, .. } = child.as_ref() else { + let NonASAPOp::Scan { predicates, .. } = op(child) else { panic!("expected Scan under the projection, got {child:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } + if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), "NOW() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2059,16 +2221,16 @@ async fn now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { let qe = lower_clickhouse("SELECT * FROM metrics WHERE ts < now()").await; - let QueryExpr::Project { child, .. } = &qe else { + let NonASAPOp::Project { child, .. } = op(&qe) else { panic!("expected Project at root, got {qe:?}"); }; - let QueryExpr::Scan { predicates, .. } = child.as_ref() else { + let NonASAPOp::Scan { predicates, .. } = op(child) else { panic!("expected Scan under the projection, got {child:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } + if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), "now() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2077,12 +2239,16 @@ async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn current_timestamp_lowers_to_typed_current_timestamp_leaf() { let qe = lower("SELECT CURRENT_TIMESTAMP FROM metrics").await; - let QueryExpr::Project { cols, .. } = &qe else { + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { panic!("expected Project at root, got {qe:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); - let schema = cols[0].expr.output_schema().expect("timestamp schema"); - assert_eq!(schema.fields[0].dtype, DataType::Timestamp); + assert!(matches!(&cols[0].expr, ScalarExpr::CurrentTimestamp)); + let (dtype, _) = cols[0] + .expr + .scalar_type(&child.schema) + .expect("timestamp type"); + assert_eq!(dtype, DataType::Timestamp); + assert_eq!(qe.schema.fields[0].dtype, DataType::Timestamp); } // A `count` over a non-null input is a plain row count; over a nullable @@ -2125,10 +2291,7 @@ async fn count_null_semantics_become_a_measure_filter() { aggregate_filters(&qe) ); }; - assert!( - matches!(cond.as_ref(), QueryExpr::IsNotNull(_)), - "{sql}: {cond:?}" - ); + assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{sql}: {cond:?}"); } // Only the second measure is filtered. let qe = lower_sql( @@ -2175,7 +2338,7 @@ async fn grouped_map_column_preserves_map_type() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().fields[0].dtype, map); + assert_eq!(query.schema.fields[0].dtype, map); } #[tokio::test] @@ -2213,10 +2376,7 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { .await .unwrap(); assert_eq!(function, operator, "{call}"); - assert_eq!( - function.output_schema().unwrap(), - operator.output_schema().unwrap() - ); + assert_eq!(function.schema, operator.schema); } let nullable = lower_sql_dialect( "SELECT modulo(n, 3) AS value FROM numbers", @@ -2226,8 +2386,8 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { ) .await .unwrap() - .output_schema() - .unwrap(); + .schema + .clone(); assert_eq!(nullable.fields[0].dtype, DataType::Int64); assert!(nullable.fields[0].nullable); } @@ -2266,7 +2426,7 @@ async fn original_o11y_map_queries_lower_with_typed_results() { ) .await .unwrap_or_else(|e| panic!("{sql}: {e}")); - let schema = query.output_schema().unwrap(); + let schema = &query.schema; assert!( schema .fields @@ -2298,7 +2458,7 @@ async fn clickhouse_modulo_preserves_projection_names_and_outer_references() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().fields[0].name, name); + assert_eq!(query.schema.fields[0].name, name); } } @@ -2328,7 +2488,7 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio ) .await .unwrap(); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[0].name, "arrayElement(labels, 'job')"); assert_eq!(output.fields[0].dtype, DataType::Utf8); assert!(!output.fields[0].nullable); @@ -2380,7 +2540,7 @@ async fn arg_selector_result_schema_tracks_selected_argument() { ) .await .unwrap(); - let schema = query.output_schema().unwrap(); + let schema = &query.schema; assert_eq!(schema.fields[0].dtype, dtype); assert_eq!(schema.fields[0].nullable, nullable); } @@ -2414,7 +2574,7 @@ async fn clickhouse_list_element_uses_canonical_typed_access() { ) .await .unwrap(); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[0].dtype, DataType::Int64); assert_eq!(output.fields[0].nullable, nullable); let serialized = serde_json::to_string(&query).unwrap(); @@ -2473,7 +2633,7 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { ) .await .unwrap(); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[0].dtype, dtype); assert_eq!(output.fields[0].nullable, nullable); assert!(serde_json::to_string(&query) @@ -2500,7 +2660,7 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { #[tokio::test] async fn corr_result_is_nullable_float() { let query = lower("SELECT corr(latency, bytes) AS correlation FROM metrics").await; - let schema = query.output_schema().unwrap(); + let schema = &query.schema; assert_eq!(schema.fields[0].name, "correlation"); assert_eq!(schema.fields[0].dtype, DataType::Float64); assert!(schema.fields[0].nullable); @@ -2524,8 +2684,8 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = - find_aggregate_node(&composite).expect("expected an Aggregate") + let NonASAPOp::Aggregate { measures, .. } = + op(find_aggregate_node(&composite).expect("expected an Aggregate")) else { unreachable!() }; @@ -2541,8 +2701,8 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = - find_aggregate_node(&single).expect("expected an Aggregate") + let NonASAPOp::Aggregate { measures, .. } = + op(find_aggregate_node(&single).expect("expected an Aggregate")) else { unreachable!() }; @@ -2600,8 +2760,10 @@ async fn distinct_with_derived_sibling() { // ── Issue #466: per-measure FILTER predicates ───────────────────────────────── /// The first `Aggregate`'s `filters`, positional against its child. -fn aggregate_filters(qe: &QueryExpr) -> &[Option] { - let Some(QueryExpr::Aggregate { filters, .. }) = find_aggregate_node(qe) else { +fn aggregate_filters(qe: &OperatorNode) -> &[Option] { + let Some(NonASAPOp::Aggregate { filters, .. }) = + find_aggregate_node(qe).map(|n| n.expect_non_asap()) + else { panic!("expected an Aggregate, got {qe:?}"); }; filters @@ -2631,15 +2793,17 @@ async fn conditional_count_lowers_to_a_filtered_measure() { panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); }; assert!( - matches!(cond.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Gt, .. } - if matches!(left.as_ref(), QueryExpr::Column(2))), + matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Gt, .. } + if matches!(left.as_ref(), ScalarExpr::Column(2))), "latency > 1.0 against the scan, got {cond:?}" ); - let Some(QueryExpr::Aggregate { child, .. }) = find_aggregate_node(&qe) else { + let Some(NonASAPOp::Aggregate { child, .. }) = + find_aggregate_node(&qe).map(|n| n.expect_non_asap()) + else { unreachable!() }; assert!( - matches!(child.as_ref(), QueryExpr::Scan { .. }), + matches!(child.expect_non_asap(), NonASAPOp::Scan { .. }), "{child:?}" ); } @@ -2653,9 +2817,9 @@ async fn filter_clause_lowers_to_a_measure_filter() { panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); }; assert!( - matches!(cond.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Eq, right } - if matches!(left.as_ref(), QueryExpr::Column(1)) - && matches!(right.as_ref(), QueryExpr::Literal(ScalarValue::Utf8(s)) if s == "a")), + matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, .. } + if matches!(left.as_ref(), ScalarExpr::Column(1)) + && matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(s)) if s == "a")), "{cond:?}" ); } @@ -2669,7 +2833,7 @@ async fn count_of_a_nullable_expression_filters_nulls() { let [Some(Predicate(cond))] = aggregate_filters(&qe) else { panic!("expected [Some], got {:?}", aggregate_filters(&qe)); }; - assert!(matches!(cond.as_ref(), QueryExpr::IsNotNull(_)), "{cond:?}"); + assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{cond:?}"); assert!( matches!( find_aggregate(&qe).unwrap().1.as_slice(), @@ -2684,23 +2848,25 @@ async fn count_of_a_nullable_expression_filters_nulls() { #[tokio::test] async fn measure_filter_columns_survive_a_derived_column_projection() { let qe = lower("SELECT sum(bytes * 2) FILTER (WHERE latency > 1.0) FROM metrics").await; - let Some(QueryExpr::Aggregate { child, .. }) = find_aggregate_node(&qe) else { + let Some(NonASAPOp::Aggregate { child, .. }) = + find_aggregate_node(&qe).map(|n| n.expect_non_asap()) + else { unreachable!() }; assert!( - matches!(child.as_ref(), QueryExpr::Project { .. }), + matches!(child.expect_non_asap(), NonASAPOp::Project { .. }), "{child:?}" ); let [Some(Predicate(cond))] = aggregate_filters(&qe) else { panic!("expected [Some], got {:?}", aggregate_filters(&qe)); }; - let QueryExpr::Compare { left, .. } = cond.as_ref() else { + let ScalarExpr::Compare { left, .. } = cond else { panic!("{cond:?}"); }; - let QueryExpr::Column(id) = left.as_ref() else { + let ScalarExpr::Column(id) = left.as_ref() else { panic!("{left:?}"); }; - assert_eq!(child.output_schema().unwrap().fields[*id].name, "latency"); + assert_eq!(child.schema.fields[*id].name, "latency"); } // `GROUP BY ROLLUP` fans one measure list out into one `Aggregate` per level; diff --git a/crates/frontend-sql/tests/temporal_types.rs b/crates/frontend-sql/tests/temporal_types.rs index a6b4f1187..7e12c6f33 100644 --- a/crates/frontend-sql/tests/temporal_types.rs +++ b/crates/frontend-sql/tests/temporal_types.rs @@ -38,10 +38,7 @@ async fn date_shifts_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Date - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Date); } } // Interval literals and explicit interval casts must both cross the Arrow bridge. @@ -54,10 +51,7 @@ async fn interval_cast_lowers_like_interval_literal() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Interval - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Interval); } } @@ -90,11 +84,7 @@ async fn negative_intervals_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Interval, - "{query}" - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Interval, "{query}"); } } @@ -108,9 +98,6 @@ async fn sql_date_literals_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Date - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Date); } } diff --git a/crates/frontend-sql/tests/unified_sql_lowering.rs b/crates/frontend-sql/tests/unified_sql_lowering.rs deleted file mode 100644 index 066434a8c..000000000 --- a/crates/frontend-sql/tests/unified_sql_lowering.rs +++ /dev/null @@ -1,2885 +0,0 @@ -//! End-to-end SQL → unresolved → resolved operator DAG lowering tests. -//! -//! Validates the DataFusion front end: SQL parses + plans, lowers directly to -//! the name-based `UnresolvedOp` tree (issue #179), and the shared -//! `resolve_root` produces the positional, canonical `OperatorNode` DAG (the -//! same resolver the PromQL path uses). Every node's schema is derived during -//! resolution, so a successful `lower` already proves schema derivation is -//! total over the tree. - -use ::asap_frontend_sql::unified as asap_frontend_sql; -use asap_types::ir::Predicate; -use std::rc::Rc; - -use asap_frontend_common::{UnresolvedOp, UnresolvedScalar}; -use asap_frontend_sql::{ - lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError, SqlLowerer, -}; -use asap_types::ir::{ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr}; -use asap_types::pre_asap::schema::{DataType, Field, FieldDataType, Schema}; -use asap_types::pre_asap::{ - AggIntent, CompareOpKind, GroupKeys, JoinKind, Reduction, ScalarValue, Source, - WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, -}; -use asap_types::types::AccuracyTarget; -use asap_types::workload::SqlDialect; - -fn col(name: &str, dtype: DataType) -> Field { - Field::plain(name, dtype, false) -} - -/// `metrics(ts, service, latency, bytes)` + `hosts(service, region)`. -fn catalog() -> SqlCatalog { - SqlCatalog::new() - .with_table( - "metrics", - Schema::with_time_index( - vec![ - col("ts", DataType::Timestamp), - col("service", DataType::Utf8), - col("latency", DataType::Float64), - col("bytes", DataType::Int64), - ], - 0, - vec![vec![0, 1]], - ), - ) - .with_table( - "hosts", - Schema::new(vec![ - col("service", DataType::Utf8), - col("region", DataType::Utf8), - ]), - ) -} - -async fn lower(sql: &str) -> Rc { - lower_sql(sql, &catalog(), AccuracyTarget::Exact) - .await - .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) -} - -/// The operator of a front-end node: a front-end DAG never holds an ASAP node. -fn op(node: &OperatorNode) -> &NonASAPOp { - node.expect_non_asap() -} - -#[tokio::test] -async fn planning_subquery_bridge_rejects_a_relation_without_vector_conversion() { - let result = lower_sql("SELECT max(value) FROM (SELECT asap_promql_subquery(21600000, 60000) AS value FROM (SELECT sum(bytes) AS value FROM metrics))", &catalog(), AccuracyTarget::Exact).await; - assert!(result.is_err()); -} - -#[tokio::test] -async fn planning_histogram_bridge_reuses_classic_bucket_intent() { - let query = lower( - "SELECT asap_histogram_quantile(0.95) AS value FROM (\ - SELECT service AS le, sum(bytes) AS value FROM metrics GROUP BY service)", - ) - .await; - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = op(&query) - else { - panic!("expected canonical histogram aggregate"); - }; - // One histogram over all rows; the bucket bound is the child's column 0. - assert!(reduction.expect_reduce().keys().is_empty()); - assert!(matches!( - measures.as_slice(), - [AggIntent::HistogramQuantile { q, le: 0 }] if (*q - 0.95).abs() < 1e-12 - )); - assert!(matches!(op(child), NonASAPOp::Project { .. })); -} - -#[tokio::test] -async fn planning_relation_bridges_reject_ambiguous_shapes() { - let missing_alias = lower_sql( - "SELECT asap_promql_subquery(300000, 60000) FROM metrics", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .unwrap_err(); - assert!(missing_alias.to_string().contains("must have an alias")); - - let histogram_with_extra_column = lower_sql( - "SELECT service, asap_histogram_quantile(0.95) AS value FROM metrics", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .unwrap_err(); - assert!(histogram_with_extra_column - .to_string() - .contains("only expression")); - - let invalid_q = lower_sql( - "SELECT asap_histogram_quantile(1.5) AS value FROM metrics", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .unwrap_err(); - assert!(invalid_q.to_string().contains("finite and in [0,1]")); -} - -/// Find the first `Aggregate` node along the single-child spine. -fn find_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { - match op(node) { - NonASAPOp::Aggregate { - reduction, - measures, - .. - } => Some((reduction.expect_reduce(), measures)), - NonASAPOp::Project { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Dedup { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } - | NonASAPOp::PromqlSubquery { child, .. } => find_aggregate(child), - _ => None, - } -} - -/// The first `Aggregate` node itself, for tests that need its child. -fn find_aggregate_node(node: &OperatorNode) -> Option<&OperatorNode> { - match op(node) { - NonASAPOp::Aggregate { .. } => Some(node), - NonASAPOp::Project { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } => find_aggregate_node(child), - _ => None, - } -} - -/// The names of the columns the first `Aggregate`'s reducers read, resolved -/// against its child's schema, plus whether that child is a materializing -/// `Project` (issue #110). -fn reducer_input_names(node: &OperatorNode) -> (Vec, bool) { - let NonASAPOp::Aggregate { - measures, child, .. - } = op(find_aggregate_node(node).expect("expected an Aggregate")) - else { - unreachable!() - }; - let schema = &child.schema; - let names = measures - .iter() - .flat_map(|a| a.input_cols()) - .map(|id| schema.fields[id].name.clone()) - .collect(); - (names, matches!(op(child), NonASAPOp::Project { .. })) -} - -/// Find the first `Join` node along the single-child spine. -fn find_join(node: &OperatorNode) -> Option<&OperatorNode> { - match op(node) { - NonASAPOp::Join { .. } => Some(node), - NonASAPOp::Project { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Aggregate { child, .. } - | NonASAPOp::Dedup { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } - | NonASAPOp::PromqlSubquery { child, .. } => find_join(child), - _ => None, - } -} - -/// The first `Filter` node along the single-child spine. -fn find_filter(node: &OperatorNode) -> Option<&OperatorNode> { - match op(node) { - NonASAPOp::Filter { .. } => Some(node), - NonASAPOp::Project { child, .. } - | NonASAPOp::Aggregate { child, .. } - | NonASAPOp::Dedup { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } - | NonASAPOp::PromqlSubquery { child, .. } => find_filter(child), - _ => None, - } -} - -#[tokio::test] -async fn where_folds_predicate_onto_scan() { - // WHERE folds onto the Scan predicates, below the SELECT projection. - let qe = lower("SELECT * FROM metrics WHERE service = 'api'").await; - let NonASAPOp::Project { child, .. } = op(&qe) else { - panic!("expected Project at root, got {qe:?}"); - }; - let NonASAPOp::Scan { - source, - predicates, - schema, - } = op(child) - else { - panic!("expected Scan under the projection, got {child:?}"); - }; - assert!(matches!(source, Source::Table { table_ref } if table_ref == "metrics")); - assert_eq!(predicates.len(), 1, "WHERE clause folded onto the scan"); - assert!( - schema.closed, - "a catalog-backed SQL scan has a closed schema" - ); -} - -#[tokio::test] -async fn multi_aggregate_group_by_binds_columns_positionally() { - // SUM(bytes)=col 3, AVG(latency)=col 2, GROUP BY service=col 1. - let qe = lower("SELECT service, SUM(bytes), AVG(latency) FROM metrics GROUP BY service").await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate in the tree"); - assert_eq!(by, &vec![1], "GROUP BY service → column 1"); - assert!( - measures.contains(&AggIntent::Sum { col: Some(3) }), - "SUM(bytes) → Sum{{col:3}}, got {measures:?}" - ); - assert!( - measures.contains(&AggIntent::Avg { col: Some(2) }), - "AVG(latency) → Avg{{col:2}}, got {measures:?}" - ); -} - -#[tokio::test] -async fn projection_over_aggregate_resolves_output_types_via_output_names() { - // The enclosing Projection references the aggregates by DataFusion's - // generated names (e.g. "sum(metrics.bytes)"); output_names threads those - // onto the canonical Aggregate so the Project resolves real types — not - // the Utf8 fallback that an unresolved column would get. - let qe = lower("SELECT SUM(bytes), AVG(latency) FROM metrics").await; - let schema = &qe.schema; - assert_eq!(schema.fields.len(), 2); - assert_eq!( - schema.fields[0].dtype, - DataType::Int64, - "SUM(bytes:Int64) resolves to Int64, not the Utf8 fallback" - ); - assert_eq!( - schema.fields[1].dtype, - DataType::Float64, - "AVG(latency) resolves to Float64" - ); -} - -#[tokio::test] -async fn single_agg_group_by_keeps_key_in_output_schema() { - // A tabular single-aggregate GROUP BY routes through the positional - // Aggregate.by path (not the PromQL fused-Partition shape), so the group - // key is a real output column the enclosing SELECT projection resolves. - let qe = lower("SELECT service, SUM(bytes) FROM metrics GROUP BY service").await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate (not a Partition)"); - assert_eq!(by, &vec![1], "GROUP BY service → Aggregate.by column 1"); - assert!(matches!( - measures.as_slice(), - [AggIntent::Sum { col: Some(3) }] - )); - - // Both the group key and the aggregate resolve in the root projection schema. - let schema = &qe.schema; - assert_eq!(schema.fields.len(), 2); - assert_eq!( - schema.fields[0].dtype, - DataType::Utf8, - "service is in the output" - ); - assert_eq!(schema.fields[1].dtype, DataType::Int64, "SUM(bytes)"); -} - -#[tokio::test] -async fn count_ranked_topk_is_heavy_hitter() { - // `ORDER BY COUNT(*) DESC LIMIT k` over a single COUNT aggregate is the one - // case the heavy-hitter (frequency) sketch is correct for. The shared - // `canonicalize` pass (issue #34) promotes it to the canonical two-level - // form: an outer global `TopK` (by: []) over the explicit inner `Count` - // grouped by `service`. - let qe = lower( - "SELECT service, COUNT(*) FROM metrics GROUP BY service ORDER BY COUNT(*) DESC LIMIT 10", - ) - .await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - by.is_empty(), - "outer TopK is a global ranking (by: []), got {by:?}" - ); - assert!( - matches!(measures.as_slice(), [AggIntent::TopK { k: 10, .. }]), - "count-ranked topk → heavy-hitter TopK, got {measures:?}" - ); - // The inner child is the explicit Count, grouped by service (col 1). - let NonASAPOp::Aggregate { child, .. } = op(&qe) else { - panic!("expected outer Aggregate, got {qe:?}"); - }; - let (inner_by, inner_measures) = find_aggregate(child).expect("expected inner Count aggregate"); - assert_eq!(inner_by, &vec![1], "inner Count grouped by service → col 1"); - assert!( - matches!(inner_measures.as_slice(), [AggIntent::Count { .. }]), - "inner aggregate is the explicit Count, got {inner_measures:?}" - ); -} - -#[tokio::test] -async fn count_ranked_topk_via_alias_is_also_heavy_hitter() { - // Regression for #20: aliasing `COUNT(*)` in the ORDER BY used to defeat the - // SQL front-end gate. The positional `canonicalize` pass now promotes it too, - // so the aliased and inline forms produce an identical canonical tree. - let inline = lower( - "SELECT service, COUNT(*) FROM metrics GROUP BY service ORDER BY COUNT(*) DESC LIMIT 10", - ) - .await; - let aliased = lower( - "SELECT service, COUNT(*) AS cnt FROM metrics GROUP BY service ORDER BY cnt DESC LIMIT 10", - ) - .await; - assert_eq!( - inline, aliased, - "aliased count-ranked topk must match the inline form" - ); - let (_, measures) = find_aggregate(&aliased).expect("expected an Aggregate"); - assert!( - matches!(measures.as_slice(), [AggIntent::TopK { k: 10, .. }]), - "aliased count-ranked topk → heavy-hitter TopK, got {measures:?}" - ); -} - -#[tokio::test] -async fn non_count_ranked_limit_keeps_the_aggregate() { - // Ranking by AVG (not a count) must NOT become a frequency heavy-hitter — - // the AVG aggregate has to survive as a generic Sort+Limit. - let qe = lower( - "SELECT service, AVG(latency) AS a FROM metrics GROUP BY service ORDER BY a DESC LIMIT 10", - ) - .await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - measures.iter().any(|a| matches!(a, AggIntent::Avg { .. })), - "AVG must be preserved, got {measures:?}" - ); - assert!( - !measures.iter().any(|a| matches!(a, AggIntent::TopK { .. })), - "AVG ranking must not become a frequency heavy-hitter, got {measures:?}" - ); -} - -#[tokio::test] -async fn distinct_value_reducer_is_rejected_not_dropped() { - // The canonical intent algebra has no distinct-Sum; SUM(DISTINCT x) must - // be rejected, not silently lowered as SUM(x). - let res = lower_sql( - "SELECT SUM(DISTINCT bytes) FROM metrics", - &catalog(), - AccuracyTarget::Exact, - ) - .await; - assert!(res.is_err(), "SUM(DISTINCT ...) should be rejected"); -} - -#[tokio::test] -async fn aggregate_over_an_expression_reduces_a_derived_column() { - // The canonical `AggIntent` reduces a column, not an arbitrary expression. - // `SUM(bytes + 1)` used to be rejected for that reason; since #110 the - // expression is materialized as a derived column in a `Project` beneath - // the aggregate, and reduced there. - let qe = lower("SELECT SUM(bytes + 1) FROM metrics").await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - matches!(measures.as_slice(), [AggIntent::Sum { col: Some(_) }]), - "expected Sum bound to the derived column, got {measures:?}" - ); - let (names, materialized) = reducer_input_names(&qe); - assert!(materialized, "expected a materializing Project"); - assert!( - names[0].contains("bytes") && names[0].contains('1'), - "the reduced column should be the projected `bytes + 1`, got {names:?}" - ); -} - -#[tokio::test] -async fn count_star_is_count_intent() { - let qe = lower("SELECT COUNT(*) FROM metrics").await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!(by.is_empty()); - assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); -} - -#[tokio::test] -async fn count_distinct_is_cardinality() { - let qe = lower("SELECT COUNT(DISTINCT service) FROM metrics").await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!(matches!( - measures.as_slice(), - [AggIntent::Cardinality { .. }] - )); -} - -#[tokio::test] -async fn select_distinct_lowers_to_distinct_with_positional_cols() { - // SELECT DISTINCT → a `Dedup` node whose `cols` are positional ColumnIds - // (not name-based ColumnRefs). DataFusion's `Distinct::All` dedups on every - // column, so `cols` is empty here — but the field type is now `Vec`. - let qe = lower("SELECT DISTINCT service FROM metrics").await; - let NonASAPOp::Dedup { cols, .. } = op(&qe) else { - panic!("expected a Dedup at the root, got {qe:?}"); - }; - let _: &Vec = cols; // compile-time: positional ids, not ColumnRefs - assert!(cols.is_empty(), "DISTINCT * dedups on all columns"); -} - -#[tokio::test] -async fn inner_join_lowers_to_join_over_two_scans() { - // INNER JOIN over two distinct tables → a canonical Join with both leaves as Scans. - let qe = lower( - "SELECT metrics.bytes, hosts.region \ - FROM metrics JOIN hosts ON metrics.service = hosts.service", - ) - .await; - let join = find_join(&qe).expect("expected a Join in the tree"); - let NonASAPOp::Join { - kind, left, right, .. - } = op(join) - else { - unreachable!("find_join only returns Join"); - }; - assert_eq!(*kind, JoinKind::Inner); - assert!(matches!(op(left), NonASAPOp::Scan { .. })); - assert!(matches!(op(right), NonASAPOp::Scan { .. })); -} - -/// The two `ColumnId`s an equijoin predicate `Column(l) = Column(r)` binds to, -/// returned sorted so the assertion is independent of left/right ordering. -fn join_eq_columns(join: &OperatorNode) -> [usize; 2] { - let NonASAPOp::Join { pred, .. } = op(join) else { - unreachable!("expected a Join"); - }; - let ScalarExpr::Compare { - left, - op: CompareOpKind::Eq, - right, - .. - } = &pred.0 - else { - panic!("expected an equijoin Compare, got {:?}", pred.0); - }; - match (left.as_ref(), right.as_ref()) { - (ScalarExpr::Column(l), ScalarExpr::Column(r)) => { - let mut cols = [*l, *r]; - cols.sort_unstable(); - cols - } - other => panic!("expected Field = Field, got {other:?}"), - } -} - -#[tokio::test] -async fn join_predicate_disambiguates_shared_column_name() { - // Issue #7: `metrics.service = hosts.service` shares a column name across the - // join. The qualified refs must bind to two *distinct* positions in the - // concatenated schema, not collapse onto the first `service`. - // metrics(ts,service,latency,bytes) ++ hosts(service,region) - // → metrics.service = col 1, hosts.service = col 4. - let qe = lower( - "SELECT metrics.bytes, hosts.region \ - FROM metrics JOIN hosts ON metrics.service = hosts.service", - ) - .await; - let join = find_join(&qe).expect("expected a Join in the tree"); - assert_eq!( - join_eq_columns(join), - [1, 4], - "join key must bind to distinct positions, not the same `service`" - ); -} - -#[tokio::test] -async fn derived_table_join_disambiguates_via_alias() { - // Issue #66: a join over two *derived tables* must bind its keys to distinct - // positions. Before the fix the derived output columns lost their qualifier, - // so `a.service` and `b.service` both fell back to the first bare `service` - // (col 0) — `service = service`, always true → a silent cross product. - // Concatenated: a[service,region] ++ b[service,region] → a.service=0, b.service=2. - let qe = lower( - "SELECT a.region, b.region \ - FROM (SELECT service, region FROM hosts) a \ - JOIN (SELECT service, region FROM hosts) b ON a.service = b.service", - ) - .await; - let join = find_join(&qe).expect("expected a Join in the tree"); - assert_eq!( - join_eq_columns(join), - [0, 2], - "derived-table join keys must bind to distinct positions, not both to the first `service`" - ); -} - -#[tokio::test] -async fn derived_table_select_star_join_disambiguates_via_alias() { - // Same as above but `SELECT *` derived tables (the non-Projection path that - // wraps the inner plan in an identity re-qualifying projection). - let qe = lower( - "SELECT a.region, b.region \ - FROM (SELECT * FROM hosts) a JOIN (SELECT * FROM hosts) b \ - ON a.service = b.service", - ) - .await; - let join = find_join(&qe).expect("expected a Join in the tree"); - let [l, r] = join_eq_columns(join); - assert_ne!( - l, r, - "SELECT * derived-table join keys must not collapse to one column" - ); -} - -#[tokio::test] -async fn self_join_disambiguates_via_aliases() { - // A self-join shares *every* column name; the alias qualifiers (`a`/`b`) are - // the only way to tell the two `service` columns apart. - // metrics ++ metrics → a.service = col 1, b.service = col 5 (4 cols/side). - let qe = lower( - "SELECT a.bytes, b.latency \ - FROM metrics a JOIN metrics b ON a.service = b.service", - ) - .await; - let join = find_join(&qe).expect("expected a self-Join in the tree"); - assert_eq!( - join_eq_columns(join), - [1, 5], - "self-join keys must bind to distinct sides" - ); -} - -#[tokio::test] -async fn qualified_where_over_join_resolves_to_right_side() { - // Issue #7 beyond the join key: a WHERE on the *duplicated* column name - // (`service` exists on both sides) must bind to the qualified side, not the - // first match. metrics.service = col 1, hosts.service = col 4 → `hosts.service` - // must resolve to 4. (Unoptimized plan keeps the Filter above the Join — no - // predicate pushdown — so it binds against the concatenated schema.) - let qe = lower( - "SELECT metrics.bytes FROM metrics JOIN hosts ON metrics.service = hosts.service \ - WHERE hosts.service = 'api'", - ) - .await; - let filter = find_filter(&qe).expect("expected a Filter over the join"); - let NonASAPOp::Filter { pred, .. } = op(filter) else { - unreachable!("find_filter only returns Filter"); - }; - assert!( - matches!(&pred.0, ScalarExpr::Compare { left, op: CompareOpKind::Eq, .. } - if matches!(left.as_ref(), ScalarExpr::Column(4))), - "hosts.service must bind to concatenated position 4 (not the first `service`), got {:?}", - pred.0 - ); -} - -#[tokio::test] -async fn self_join_group_by_disambiguates_via_qualifier() { - // Group-key qualifier fix: GROUP BY on the *duplicated* column over a - // self-join must bind to the qualified side, not first-match. metrics ⋈ - // metrics → a.service = col 1, b.service = col 5. (Without qualified keys, - // both `GROUP BY a.service` and `GROUP BY b.service` collapsed to col 1.) - let qe_b = lower( - "SELECT b.service, COUNT(*) FROM metrics a JOIN metrics b \ - ON a.service = b.service GROUP BY b.service", - ) - .await; - let (by, _) = find_aggregate(&qe_b).expect("expected an Aggregate over the self-join"); - assert_eq!( - by, - &vec![5], - "GROUP BY b.service binds to the b side (col 5)" - ); - - let qe_a = lower( - "SELECT a.service, COUNT(*) FROM metrics a JOIN metrics b \ - ON a.service = b.service GROUP BY a.service", - ) - .await; - let (by, _) = find_aggregate(&qe_a).expect("expected an Aggregate over the self-join"); - assert_eq!( - by, - &vec![1], - "GROUP BY a.service binds to the a side (col 1)" - ); -} - -#[tokio::test] -async fn aggregate_over_join_binds_against_concatenated_schema() { - // GROUP BY a right-table column over a join: the key must resolve against - // the concatenated schema, exercising the bottom-up converter end to end. - // Two aggregates → the multi-agg path, which carries GROUP BY keys as - // positional `Aggregate.by` (as does every reducing GROUP BY). - let qe = lower( - "SELECT hosts.region, SUM(metrics.bytes), COUNT(*) \ - FROM metrics JOIN hosts ON metrics.service = hosts.service \ - GROUP BY hosts.region", - ) - .await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate over the join"); - // metrics(ts,service,latency,bytes) ++ hosts(service,region) → - // region is column 5, bytes is column 3 of the concatenated schema. - assert_eq!( - by, - &vec![5], - "GROUP BY hosts.region → concatenated column 5" - ); - assert!( - measures.contains(&AggIntent::Sum { col: Some(3) }), - "SUM(metrics.bytes) → Sum{{col:3}}, got {measures:?}" - ); -} - -// ── Issue #111: IN / EXISTS subquery predicates become semi / anti joins ──── -// -// The front end now leaves them as `UnresolvedScalar::{InSubquery, Exists}` -// filter conjuncts; the shared `canonicalize` pass (run by `resolve_root`) -// lowers each to the semi-/anti-join, so the resolved DAG a test sees is the -// same join shape the front end used to emit directly. - -/// The first `Join` node's `(kind, predicate, left column count)`. -fn join_parts(node: &OperatorNode) -> (&JoinKind, &ScalarExpr, usize) { - let NonASAPOp::Join { - kind, - pred, - left, - right: _, - } = op(find_join(node).expect("expected a Join")) - else { - unreachable!() - }; - (kind, &pred.0, left.schema.fields.len()) -} - -#[tokio::test] -async fn in_subquery_lowers_to_a_semi_join() { - // `metrics(ts, service, latency, bytes)` — service is column 1. - let qe = - lower("SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)").await; - let (kind, pred, left_len) = join_parts(&qe); - assert_eq!(kind, &JoinKind::Semi); - assert_eq!(left_len, 4); - - // The predicate resolves against `left ++ right`. Both relations have a - // `service` column, so a name-based lookup would bind *both* sides to the - // left's — silently making this `service = service`, always true. The key is - // bound positionally to the subquery's column (right after the left's), - // which makes that impossible. - let ScalarExpr::Compare { left, right, .. } = pred else { - panic!("expected a comparison, got {pred:?}"); - }; - assert_eq!(**left, ScalarExpr::Column(1), "outer service"); - assert_eq!( - **right, - ScalarExpr::Column(left_len), - "the subquery key, not the outer column again" - ); -} - -#[tokio::test] -async fn a_semi_join_outputs_only_the_left_schema() { - // The right side is a filter, not a source of columns. - let qe = - lower("SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)").await; - let join = find_join(&qe).expect("expected a Join"); - let names: Vec<_> = join.schema.fields.iter().map(|c| c.name.clone()).collect(); - assert_eq!(names, ["ts", "service", "latency", "bytes"]); -} - -#[tokio::test] -async fn a_subquery_key_that_is_an_expression_still_binds() { - // `SELECT bytes + 1 …` has no column name of its own; the join key binds - // to it positionally rather than through an unreferenceable `col_0`. - let qe = - lower("SELECT service FROM metrics WHERE bytes IN (SELECT bytes + 1 FROM metrics)").await; - assert_eq!(join_parts(&qe).0, &JoinKind::Semi); -} - -#[tokio::test] -async fn a_multi_column_in_subquery_is_rejected() { - let err = lower_sql( - "SELECT service FROM metrics WHERE service IN (SELECT service, region FROM hosts)", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .expect_err("IN must select one column"); - // DataFusion's planner rejects this before `lower_in_subquery`'s own - // arity check; either message names the one-column rule. - assert!(format!("{err}").contains("one column"), "got {err}"); -} - -#[tokio::test] -async fn an_ordinary_conjunct_still_folds_onto_the_scan() { - // The residual filter stays *below* the semi-join, where the converter can - // still fold it onto the Scan. A semi-join only drops left rows, so the - // orders agree. - let qe = lower( - "SELECT service FROM metrics WHERE bytes > 10 \ - AND service IN (SELECT service FROM hosts)", - ) - .await; - fn scan_has_predicate(node: &OperatorNode) -> bool { - match op(node) { - NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), - NonASAPOp::Project { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Aggregate { child, .. } => scan_has_predicate(child), - NonASAPOp::Join { left, right, .. } => { - scan_has_predicate(left) || scan_has_predicate(right) - } - _ => false, - } - } - assert_eq!(join_parts(&qe).0, &JoinKind::Semi); - assert!( - scan_has_predicate(&qe), - "WHERE bytes > 10 should reach the Scan" - ); -} - -/// Find the first `SQLWindowFunc` node along the single-child spine. -fn find_windowfunc(node: &OperatorNode) -> Option<&OperatorNode> { - match op(node) { - NonASAPOp::SQLWindowFunc { .. } => Some(node), - NonASAPOp::Project { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Aggregate { child, .. } - | NonASAPOp::Dedup { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } - | NonASAPOp::PromqlSubquery { child, .. } => find_windowfunc(child), - _ => None, - } -} - -#[tokio::test] -async fn window_function_lowers_to_positional_windowfunc() { - // ROW_NUMBER() OVER (PARTITION BY service ORDER BY bytes DESC). - let qe = lower( - "SELECT service, ROW_NUMBER() OVER (PARTITION BY service ORDER BY bytes DESC) \ - FROM metrics", - ) - .await; - let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let NonASAPOp::SQLWindowFunc { - func, - partition_by, - order_by, - .. - } = op(win) - else { - unreachable!("find_windowfunc only returns SQLWindowFunc"); - }; - assert_eq!(*func, WindowFuncKind::RowNumber); - assert_eq!(partition_by, &vec![1], "PARTITION BY service → col 1"); - assert_eq!(order_by.len(), 1); - assert_eq!( - order_by[0].expr, - ScalarExpr::Column(3), - "ORDER BY bytes → col 3" - ); - assert!(!order_by[0].ascending, "DESC"); - - // The window output column is appended to the schema (Int64 for ROW_NUMBER), - // and the enclosing projection resolves it (output_name threading). - let schema = &qe.schema; - assert!( - schema.fields.iter().any(|c| c.dtype == DataType::Int64), - "row_number output column present, got {:?}", - schema.fields - ); -} - -#[tokio::test] -async fn window_aggregate_lowers_to_windowfunc() { - let qe = lower("SELECT service, SUM(bytes) OVER (PARTITION BY service) FROM metrics").await; - let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { - unreachable!(); - }; - assert_eq!(*func, WindowFuncKind::Sum); - assert_eq!(args, &vec![ScalarExpr::Column(3)], "SUM(bytes) → arg col 3"); -} - -// ── Window frames (issue #268) ─────────────────────────────────────────────── - -/// The frame clause must actually reach the IR, not just the display string: -/// three window frames that differ semantically must lower to different -/// `SQLWindowFunc.frame` values. -#[tokio::test] -async fn window_frame_is_captured_not_dropped() { - let default_frame = - lower("SELECT service, SUM(latency) OVER (PARTITION BY service ORDER BY ts) FROM metrics") - .await; - let two_preceding = lower( - "SELECT service, SUM(latency) OVER (PARTITION BY service ORDER BY ts \ - ROWS BETWEEN 2 PRECEDING AND CURRENT ROW) FROM metrics", - ) - .await; - let unbounded_following = lower( - "SELECT service, SUM(latency) OVER (PARTITION BY service ORDER BY ts \ - ROWS BETWEEN CURRENT ROW AND UNBOUNDED FOLLOWING) FROM metrics", - ) - .await; - - let frame_of = |node: &OperatorNode| { - let NonASAPOp::SQLWindowFunc { frame, .. } = op(find_windowfunc(node).unwrap()) else { - unreachable!(); - }; - frame - .clone() - .expect("newly lowered SQL always records a frame") - }; - let (a, b, c) = ( - frame_of(&default_frame), - frame_of(&two_preceding), - frame_of(&unbounded_following), - ); - assert_ne!(a, b, "default frame vs ROWS 2 PRECEDING must differ"); - assert_ne!( - a, c, - "default frame vs ROWS CURRENT..UNBOUNDED FOLLOWING must differ" - ); - assert_ne!(b, c); - - assert_eq!(b.units, WindowFrameUnits::Rows); - assert_eq!( - b.start_bound, - WindowFrameBound::Preceding(WindowFrameOffset::Scalar(ScalarValue::Int64(2))) - ); - assert_eq!(b.end_bound, WindowFrameBound::CurrentRow); - - assert_eq!(c.start_bound, WindowFrameBound::CurrentRow); - assert_eq!( - c.end_bound, - WindowFrameBound::Following(WindowFrameOffset::Scalar(ScalarValue::Null)) - ); -} - -#[tokio::test] -async fn range_interval_frame_is_preserved() { - let qe = lower( - "SELECT service, SUM(latency) OVER (PARTITION BY service ORDER BY ts \ - RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW) FROM metrics", - ) - .await; - let NonASAPOp::SQLWindowFunc { - frame: Some(frame), .. - } = op(find_windowfunc(&qe).unwrap()) - else { - panic!("expected a window function with a concrete frame"); - }; - - assert_eq!(frame.units, WindowFrameUnits::Range); - assert_eq!( - frame.start_bound, - WindowFrameBound::Preceding(WindowFrameOffset::Interval { - months: 0, - days: 0, - nanoseconds: 3_600_000_000_000, - }) - ); - assert_eq!(frame.end_bound, WindowFrameBound::CurrentRow); -} - -#[tokio::test] -async fn range_numeric_frames_remain_scalar_offsets() { - let integer = lower( - "SELECT SUM(bytes) OVER (ORDER BY bytes \ - RANGE BETWEEN 2 PRECEDING AND CURRENT ROW) FROM metrics", - ) - .await; - let fractional = lower( - "SELECT SUM(latency) OVER (ORDER BY latency \ - RANGE BETWEEN 1.5 PRECEDING AND CURRENT ROW) FROM metrics", - ) - .await; - - let start_bound = |node: &OperatorNode| { - let NonASAPOp::SQLWindowFunc { - frame: Some(frame), .. - } = op(find_windowfunc(node).unwrap()) - else { - panic!("expected a window function with a concrete frame"); - }; - frame.start_bound.clone() - }; - - assert_eq!( - start_bound(&integer), - WindowFrameBound::Preceding(WindowFrameOffset::Scalar(ScalarValue::Int64(2))) - ); - assert_eq!( - start_bound(&fractional), - WindowFrameBound::Preceding(WindowFrameOffset::Scalar(ScalarValue::Float64(1.5))) - ); -} - -/// `GROUPS` frames aren't in this repo's SQL corpora and nothing downstream -/// interprets frame semantics yet — rejected explicitly rather than silently -/// mis-lowered. -#[tokio::test] -async fn groups_frame_is_rejected() { - let err = lower_sql( - "SELECT service, SUM(latency) OVER (PARTITION BY service ORDER BY ts \ - GROUPS BETWEEN 2 PRECEDING AND CURRENT ROW) FROM metrics", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .expect_err("GROUPS frame unit must be rejected"); - assert!(format!("{err}").contains("GROUPS"), "got {err}"); -} - -// ── Nested query functions: derived tables / inline views (issue #27) ─────────── - -/// Collect every `AggIntent` in the DAG, root-to-leaf (every reachable node, -/// including operators referenced from scalar positions). -fn all_intents(root: &Rc) -> Vec { - OperatorNode::reachable(root) - .iter() - .filter_map(|node| match op(node) { - NonASAPOp::Aggregate { measures, .. } => Some(measures.clone()), - _ => None, - }) - .flatten() - .collect() -} - -#[tokio::test] -async fn derived_table_aggregate_over_aggregate_nests() { - // `MAX(s)` over a derived table `(SELECT service, SUM(bytes) AS s … GROUP BY - // service)` — the SQL counterpart of PromQL function nesting (issue #27). - // Both reductions survive into the canonical tree: an outer `Max` over - // the inner `Sum`. - let qe = lower( - "SELECT MAX(s) FROM \ - (SELECT service, SUM(bytes) AS s FROM metrics GROUP BY service) t", - ) - .await; - let intents = all_intents(&qe); - assert!( - intents.iter().any(|i| matches!(i, AggIntent::Max { .. })), - "outer MAX survives, got {intents:?}" - ); - assert!( - intents.iter().any(|i| matches!(i, AggIntent::Sum { .. })), - "inner SUM survives, got {intents:?}" - ); - // The whole nested tree's output schema derives (positional resolution - // is total across the derived-table boundary). - assert_eq!(qe.schema.fields.len(), 1); -} - -#[tokio::test] -async fn derived_table_outer_avg_over_inner_percentile() { - // Outer exact `AVG` over an inner approximate `Quantile` — each layer keeps - // its own intent (the per-node sketch-vs-exact choice is a post-ASAP decision). - let qe = lower( - "SELECT AVG(p) FROM \ - (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ - FROM metrics GROUP BY service) t", - ) - .await; - let intents = all_intents(&qe); - assert!(intents.iter().any(|i| matches!(i, AggIntent::Avg { .. }))); - assert!(intents - .iter() - .any(|i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9))); -} - -#[tokio::test] -async fn filter_over_derived_aggregate_resolves_alias_column() { - // `WHERE t.s > 100` over a derived aggregate — the qualified ref `t.s` - // resolves by bare name against the derived output schema, and the Filter - // sits above the inner Aggregate. - let qe = lower( - "SELECT t.service, t.s FROM \ - (SELECT service, SUM(bytes) AS s FROM metrics GROUP BY service) t \ - WHERE t.s > 100", - ) - .await; - assert!( - find_filter(&qe).is_some(), - "the outer WHERE lowers to a Filter, got {qe:?}" - ); - assert!(all_intents(&qe) - .iter() - .any(|i| matches!(i, AggIntent::Sum { .. }))); - // Schema derivation is total across the boundary: the root carries one. - assert_eq!(qe.schema.fields.len(), 2); -} - -#[tokio::test] -async fn scalar_subquery_in_predicate_lowers_through_a_cross_join() { - let qe = - lower("SELECT service FROM metrics WHERE bytes > (SELECT AVG(bytes) FROM metrics)").await; - let filter = find_filter(&qe).unwrap(); - let NonASAPOp::Filter { pred, child } = op(filter) else { - panic!() - }; - assert!(matches!(op(child), NonASAPOp::Scan { .. })); - assert!( - matches!(&pred.0,ScalarExpr::Compare { right,.. } if matches!(right.as_ref(),ScalarExpr::ScalarSubquery(_))) - ); - qe.validate_structure().unwrap(); -} - -#[tokio::test] -async fn correlated_exists_lifts_its_correlation_into_the_join() { - // `EXISTS (SELECT 1 FROM hosts h WHERE h.service = m.service)` → a semi-join - // on `h.service = m.service`. The `SELECT 1` projection is dropped: a - // semi-join keeps no right columns, and it would have projected away the - // very column the correlation needs. - let qe = lower( - "SELECT service FROM metrics m WHERE EXISTS \ - (SELECT 1 FROM hosts h WHERE h.service = m.service)", - ) - .await; - let (kind, pred, left_len) = join_parts(&qe); - assert_eq!(kind, &JoinKind::Semi); - let ScalarExpr::Compare { left, right, .. } = pred else { - panic!("expected the correlation as a comparison, got {pred:?}"); - }; - assert_eq!( - **left, - ScalarExpr::Column(left_len), - "h.service (right side)" - ); - assert_eq!(**right, ScalarExpr::Column(1), "m.service (left side)"); -} - -#[tokio::test] -async fn not_exists_lowers_to_an_anti_join() { - let qe = lower( - "SELECT service FROM metrics m WHERE NOT EXISTS \ - (SELECT 1 FROM hosts h WHERE h.service = m.service)", - ) - .await; - assert_eq!(join_parts(&qe).0, &JoinKind::Anti); -} - -#[tokio::test] -async fn an_uncorrelated_exists_is_an_unconditional_semi_join() { - // No correlation → keep every left row iff the right side has any row. - let qe = lower("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; - let (kind, pred, _) = join_parts(&qe); - assert_eq!(kind, &JoinKind::Semi); - assert_eq!(*pred, ScalarExpr::Literal(ScalarValue::Boolean(true))); -} - -#[tokio::test] -async fn where_exists_resolves_to_a_semi_join_over_the_subquery() { - // The front end emits `Filter { Exists(s) }`; the resolved DAG is the - // `Semi` join with the subquery (a filtered `hosts` scan) on the right. - let qe = lower( - "SELECT service FROM metrics WHERE EXISTS (SELECT service FROM hosts WHERE region = 'eu')", - ) - .await; - let NonASAPOp::Project { child, .. } = op(&qe) else { - panic!("expected the SELECT list as a Project, got {qe:?}"); - }; - let NonASAPOp::Join { - kind, - pred, - left, - right, - } = op(child) - else { - panic!("expected the Semi join directly under the Project, got {child:?}"); - }; - assert_eq!(*kind, JoinKind::Semi); - assert_eq!(pred.0, ScalarExpr::Literal(ScalarValue::Boolean(true))); - assert!( - matches!(op(left), NonASAPOp::Scan { .. }), - "left is metrics" - ); - let NonASAPOp::Project { child: scan, .. } = op(right) else { - panic!("expected the subquery's projection on the right, got {right:?}"); - }; - assert!( - matches!(op(scan), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1), - "the subquery's WHERE stays on its own Scan, got {scan:?}" - ); - assert_eq!( - child.schema.fields.len(), - 4, - "a semi join outputs the left's columns alone" - ); -} - -#[tokio::test] -async fn not_in_subquery_is_rejected_rather_than_mislowered_as_an_anti_join() { - let qe = - lower("SELECT service FROM metrics WHERE service NOT IN (SELECT service FROM hosts)").await; - let filter = find_filter(&qe).unwrap(); - let NonASAPOp::Filter { pred, .. } = op(filter) else { - panic!() - }; - assert!(matches!( - pred.0, - ScalarExpr::InSubquery { negated: true, .. } - )); - qe.validate_structure().unwrap(); -} - -#[tokio::test] -async fn a_correlated_in_subquery_is_rejected() { - let err = lower_sql( - "SELECT service FROM metrics m WHERE service IN \ - (SELECT h.service FROM hosts h WHERE h.region = m.service)", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .expect_err("correlated IN needs both a key match and a correlation"); - assert!(format!("{err}").contains("correlated IN"), "got {err}"); -} - -// ── Subquery-valued expressions at the `UnresolvedOp` level ───────────────── - -/// `SqlLowerer::lower` output, before `resolve_root`. -async fn lower_unresolved(sql: &str) -> UnresolvedOp { - let catalog = catalog(); - SqlLowerer::new(&catalog) - .lower(sql, &AccuracyTarget::Exact) - .await - .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) -} - -#[tokio::test] -async fn scalar_subquery_in_projection_lowers_to_a_scalar_subquery_item() { - // An uncorrelated `(SELECT max(v) FROM t2)` in the SELECT list is a - // `ScalarSubquery` projection item reading its own lowered plan; the - // cross-join rewrite is `canonicalize`'s job, not the front end's. - let tree = lower_unresolved("SELECT (SELECT max(latency) FROM metrics) FROM hosts").await; - let UnresolvedOp::Project { cols, child, .. } = &tree else { - panic!("expected the SELECT list as a Project, got {tree:?}"); - }; - assert!( - matches!(child.as_ref(), UnresolvedOp::Scan { source: Source::Table { table_ref }, .. } - if table_ref == "hosts"), - "the outer relation stays the projection's child, got {child:?}" - ); - assert_eq!(cols.len(), 1); - let UnresolvedScalar::ScalarSubquery(sub) = &cols[0].expr else { - panic!("expected a ScalarSubquery item, got {:?}", cols[0].expr); - }; - let UnresolvedOp::Project { child: inner, .. } = sub.as_ref() else { - panic!("expected the subquery's own SELECT list, got {sub:?}"); - }; - assert!( - matches!(inner.as_ref(), UnresolvedOp::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Max { .. }])), - "the subquery plan is lowered as a root of its own, got {inner:?}" - ); -} - -#[tokio::test] -async fn exists_and_in_subqueries_lower_to_scalar_filter_conjuncts() { - // The front end no longer builds the semi join itself: `EXISTS` / `IN - // (…)` are `Filter` predicates reading the subquery operator. - let tree = - lower_unresolved("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; - let UnresolvedOp::Project { child, .. } = &tree else { - panic!("expected a Project, got {tree:?}"); - }; - assert!( - matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } - if matches!(pred.0, UnresolvedScalar::Exists { negated: false, .. })), - "expected Filter {{ Exists }}, got {child:?}" - ); - - let tree = lower_unresolved( - "SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)", - ) - .await; - let UnresolvedOp::Project { child, .. } = &tree else { - panic!("expected a Project, got {tree:?}"); - }; - assert!( - matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } - if matches!(pred.0, UnresolvedScalar::InSubquery { negated: false, .. })), - "expected Filter {{ InSubquery }}, got {child:?}" - ); -} - -// ── `SELECT` without `FROM`, unary minus, SQL expression semantics ────────── - -#[tokio::test] -async fn select_without_from_projects_over_one_empty_row() { - // `SELECT 1` has no table: DataFusion's `EmptyRelation` is one empty - // input row, which the SELECT list projects a literal over. - let qe = lower("SELECT 1").await; - let NonASAPOp::Project { cols, child, .. } = op(&qe) else { - panic!("expected Project at root, got {qe:?}"); - }; - assert_eq!(cols.len(), 1); - assert_eq!(cols[0].expr, ScalarExpr::Literal(ScalarValue::Int64(1))); - let NonASAPOp::Values { rows, schema } = op(child) else { - panic!("expected Values under the Project, got {child:?}"); - }; - assert_eq!(rows, &vec![Vec::::new()], "one empty row"); - assert!(schema.fields.is_empty() && schema.closed); - assert_eq!(qe.schema.fields.len(), 1); - assert_eq!(qe.schema.fields[0].dtype, DataType::Int64); -} - -#[tokio::test] -async fn values_lowers_to_one_row_per_values_row() { - let qe = lower("SELECT * FROM (VALUES (1, 'a'), (2, 'b')) AS v(n, s)").await; - let values = OperatorNode::reachable(&qe) - .into_iter() - .find(|n| matches!(op(n), NonASAPOp::Values { .. })) - .expect("expected a Values node"); - let NonASAPOp::Values { rows, schema } = op(&values) else { - unreachable!() - }; - assert_eq!(rows.len(), 2); - assert_eq!( - rows[1], - vec![ - ScalarExpr::Literal(ScalarValue::Int64(2)), - ScalarExpr::Literal(ScalarValue::Utf8("b".into())), - ] - ); - assert_eq!(schema.fields.len(), 2); - assert_eq!(schema.fields[0].dtype, DataType::Int64); - assert_eq!(schema.fields[1].dtype, DataType::Utf8); - assert_eq!( - qe.schema - .fields - .iter() - .map(|f| f.name.as_str()) - .collect::>(), - ["n", "s"] - ); -} - -#[tokio::test] -async fn unary_minus_lowers_to_negative() { - // `-x` over a column is the `Negative` scalar (a negative *literal* is - // folded by DataFusion's planner before lowering). - let qe = lower("SELECT -latency FROM metrics").await; - let NonASAPOp::Project { cols, .. } = op(&qe) else { - panic!("expected Project at root, got {qe:?}"); - }; - assert_eq!( - cols[0].expr, - ScalarExpr::Negative { - expr: Box::new(ScalarExpr::Column(2)), - semantics: ExprSemantics::Sql, - } - ); - assert_eq!(qe.schema.fields[0].dtype, DataType::Float64); -} - -#[tokio::test] -async fn sql_comparisons_and_arithmetic_carry_sql_semantics() { - let qe = lower("SELECT bytes * 8 FROM metrics WHERE latency > 1.5").await; - let NonASAPOp::Project { cols, child, .. } = op(&qe) else { - panic!("expected Project at root, got {qe:?}"); - }; - assert!( - matches!( - &cols[0].expr, - ScalarExpr::Arithmetic { - semantics: ExprSemantics::Sql, - .. - } - ), - "got {:?}", - cols[0].expr - ); - let NonASAPOp::Scan { predicates, .. } = op(child) else { - panic!("expected the WHERE folded onto the Scan, got {child:?}"); - }; - assert!( - matches!( - &predicates[0].0, - ScalarExpr::Compare { - semantics: ExprSemantics::Sql, - .. - } - ), - "got {:?}", - predicates[0].0 - ); -} - -// ── Issue #115: Quantile / Cardinality carry their input column ───────────── - -#[tokio::test] -async fn quantile_carries_its_input_column() { - // `metrics(ts=0, service=1, latency=2, bytes=3)`. Two quantiles over - // different columns must not compare equal — a workload-level dedupe pass - // would compare on `AggIntent` equality, so a col-less intent would - // collapse them. - let qe = lower( - "SELECT approx_percentile_cont(latency, 0.5), \ - approx_percentile_cont(bytes, 0.5) FROM metrics", - ) - .await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - matches!( - measures.as_slice(), - [ - AggIntent::Quantile { col: Some(2), .. }, - AggIntent::Quantile { col: Some(3), .. } - ] - ), - "quantiles must bind their own column, got {measures:?}" - ); - assert_ne!( - measures[0], measures[1], - "distinct-column quantiles must not compare equal" - ); -} - -#[tokio::test] -async fn count_distinct_carries_its_input_column() { - let qe = lower("SELECT COUNT(DISTINCT service), COUNT(DISTINCT bytes) FROM metrics").await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - matches!( - measures.as_slice(), - [ - AggIntent::Cardinality { cols: c1, .. }, - AggIntent::Cardinality { cols: c2, .. } - ] if c1 == &[1] && c2 == &[3] - ), - "cardinalities must bind their own column, got {measures:?}" - ); - assert_ne!( - measures[0], measures[1], - "distinct-column cardinalities must not compare equal" - ); -} - -#[tokio::test] -async fn quantile_and_count_distinct_over_an_expression_bind_the_derived_column() { - // A SQL aggregate has no "sample value" to fall back on, so an expression - // argument must never reach the canonical tree as `col: None` (#115). - // Since #110 it reaches the canonical tree as `col: Some(derived)` - // instead of being rejected. - for q in [ - "SELECT approx_percentile_cont(bytes * 8, 0.95) FROM metrics", - "SELECT COUNT(DISTINCT bytes * 8) FROM metrics", - "SELECT approx_distinct(bytes * 8) FROM metrics", - ] { - let qe = lower(q).await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - !measures[0].input_cols().is_empty(), - "{q} must bind a column, never the implicit input, got {measures:?}" - ); - let (names, materialized) = reducer_input_names(&qe); - assert!(materialized, "{q} expected a materializing Project"); - assert!( - names[0].contains("bytes"), - "{q} should reduce the projected `bytes * 8`, got {names:?}" - ); - } -} - -// ── Issue #111: median / approx_median → the φ=0.5 quantile ───────────────── - -#[tokio::test] -async fn median_lowers_to_the_half_quantile() { - // `metrics(ts=0, service=1, latency=2, bytes=3)`. - for sql in [ - "SELECT median(latency) FROM metrics", - "SELECT approx_median(latency) FROM metrics", - ] { - let qe = lower(sql).await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - matches!( - measures.as_slice(), - [AggIntent::Quantile { col: Some(2), q, .. }] if (*q - 0.5).abs() < 1e-9 - ), - "{sql} should lower to Quantile(0.5) over latency, got {measures:?}" - ); - } -} - -#[tokio::test] -async fn median_is_the_same_intent_as_an_explicit_half_percentile() { - // Two spellings of one intent: CSE should be able to merge them. - let m = lower("SELECT median(latency) FROM metrics").await; - let p = lower("SELECT approx_percentile_cont(latency, 0.5) FROM metrics").await; - let (_, m_measures) = find_aggregate(&m).expect("expected an Aggregate"); - let (_, p_measures) = find_aggregate(&p).expect("expected an Aggregate"); - assert_eq!(m_measures, p_measures); -} - -#[tokio::test] -async fn median_threads_the_accuracy_target() { - // The `approx_` prefix does not decide: the AccuracyTarget does. - let qe = lower_sql( - "SELECT approx_median(latency) FROM metrics", - &catalog(), - AccuracyTarget::Epsilon(0.01), - ) - .await - .expect("approx_median should lower"); - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - matches!( - measures.as_slice(), - [AggIntent::Quantile { accuracy: AccuracyTarget::Epsilon(e), .. }] - if (*e - 0.01).abs() < 1e-12 - ), - "median must carry the workload's accuracy target, got {measures:?}" - ); -} - -#[tokio::test] -async fn median_over_an_expression_binds_the_derived_column() { - // Was rejected when filed (#111); supported since #110 materialized the - // expression. What must still hold is the #115 rule: never `col: None`. - let qe = lower("SELECT median(bytes * 8) FROM metrics").await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!( - matches!(measures.as_slice(), [AggIntent::Quantile { col: Some(_), q, .. }] if (*q - 0.5).abs() < 1e-9), - "expected Quantile(0.5) bound to the derived column, got {measures:?}" - ); -} - -// ── Issue #110: expression GROUP BY (time bucketing) ──────────────────────── - -#[tokio::test] -async fn time_bucketing_group_by_lowers_to_a_derived_key() { - // The canonical time-series shape: `GROUP BY date_trunc(...)`. The bucket - // expression is materialized beneath the aggregate and grouped on. - let qe = - lower("SELECT date_trunc('minute', ts) AS m, SUM(bytes) FROM metrics GROUP BY m").await; - let node = find_aggregate_node(&qe).expect("expected an Aggregate"); - let NonASAPOp::Aggregate { - reduction, - measures, - child, - .. - } = op(node) - else { - unreachable!() - }; - assert!( - matches!(op(child), NonASAPOp::Project { .. }), - "expected a materializing Project beneath the Aggregate" - ); - let schema = &child.schema; - assert_eq!(reduction, &Reduction::by(vec![0])); - assert!( - schema.fields[0].name.contains("date_trunc"), - "group key should be the projected bucket, got {:?}", - schema.fields[0].name - ); - // The reducer still binds its own column, not the bucket. - assert!(matches!( - measures.as_slice(), - [AggIntent::Sum { col: Some(1) }] - )); -} - -#[tokio::test] -async fn time_bucketing_keeps_the_scan_predicate() { - // The projection is inserted above the scan, so a WHERE clause still folds - // onto the Scan rather than being stranded. - let qe = lower( - "SELECT date_trunc('minute', ts) AS m, SUM(bytes) FROM metrics \ - WHERE bytes > 10 GROUP BY m", - ) - .await; - fn scan_has_predicate(node: &OperatorNode) -> bool { - match op(node) { - NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), - NonASAPOp::Project { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Aggregate { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } => scan_has_predicate(child), - _ => false, - } - } - assert!(scan_has_predicate(&qe), "WHERE should stay on the Scan"); -} - -#[tokio::test] -async fn a_plain_group_by_inserts_no_projection() { - // Queries that lowered before #110 must keep their exact tree shape — the - // projection appears only when something actually needs materializing. - for q in [ - "SELECT service, SUM(bytes) FROM metrics GROUP BY service", - "SELECT SUM(bytes) FROM metrics", - "SELECT COUNT(*) FROM metrics", - ] { - let qe = lower(q).await; - let NonASAPOp::Aggregate { child, .. } = - op(find_aggregate_node(&qe).expect("expected an Aggregate")) - else { - unreachable!() - }; - assert!( - !matches!(op(child), NonASAPOp::Project { .. }), - "{q} should not gain a projection" - ); - } -} - -#[tokio::test] -async fn a_shared_expression_is_materialized_once() { - let qe = lower("SELECT SUM(bytes * 2), MIN(bytes * 2) FROM metrics").await; - let NonASAPOp::Aggregate { - measures, child, .. - } = op(find_aggregate_node(&qe).expect("expected an Aggregate")) - else { - unreachable!() - }; - assert_eq!( - child.schema.fields.len(), - 1, - "the two reducers should share one derived column" - ); - assert_eq!(measures[0].input_cols(), measures[1].input_cols()); -} - -// ── Issue #118: multi-level grouping expands into one Aggregate per level ─── - -/// The branches of the first `Concat` along the single-child spine. -fn merge_branches(node: &OperatorNode) -> &Vec> { - fn find(node: &OperatorNode) -> Option<&Vec>> { - match op(node) { - NonASAPOp::Concat { children, .. } => Some(children), - NonASAPOp::Project { child, .. } - | NonASAPOp::Filter { child, .. } - | NonASAPOp::Sort { child, .. } - | NonASAPOp::Limit { child, .. } => find(child), - _ => None, - } - } - find(node).expect("expected a Concat") -} - -/// `(group keys, column names)` of each merged grouping level. -fn grouping_levels(node: &OperatorNode) -> Vec<(GroupKeys, Vec)> { - merge_branches(node) - .iter() - .map(|b| { - let NonASAPOp::Project { child, .. } = op(b) else { - panic!("expected a Project per level, got {b:?}"); - }; - let NonASAPOp::Aggregate { reduction, .. } = op(child) else { - panic!("expected an Aggregate under the Project, got {child:?}"); - }; - let names = b.schema.fields.iter().map(|c| c.name.clone()).collect(); - (reduction.expect_reduce().clone(), names) - }) - .collect() -} - -#[tokio::test] -async fn rollup_expands_to_one_aggregate_per_prefix() { - // ROLLUP(a, b) → (a,b), (a), () — three levels, widest first. - let qe = - lower("SELECT service, bytes, SUM(latency) FROM metrics GROUP BY ROLLUP(service, bytes)") - .await; - let levels = grouping_levels(&qe); - let keys: Vec<_> = levels.iter().map(|(by, _)| by.clone()).collect(); - assert_eq!( - keys, - vec![ - GroupKeys::by(vec![1, 3]), - GroupKeys::by(vec![1]), - GroupKeys::none(), - ] - ); -} - -#[tokio::test] -async fn cube_expands_to_the_power_set() { - // CUBE(a, b) → (a,b), (a), (b), () — four levels. - let qe = - lower("SELECT service, bytes, SUM(latency) FROM metrics GROUP BY CUBE(service, bytes)") - .await; - assert_eq!(grouping_levels(&qe).len(), 4); -} - -#[tokio::test] -async fn a_mixed_grouping_set_is_normalized_by_datafusion() { - // `GROUP BY g, ROLLUP(d)` arrives as one GroupingSets, not a plain key - // alongside a grouping set — so there is only one shape to handle. - let qe = - lower("SELECT service, bytes, SUM(latency) FROM metrics GROUP BY service, ROLLUP(bytes)") - .await; - assert_eq!(grouping_levels(&qe).len(), 2); -} - -#[tokio::test] -async fn omitted_grouping_keys_become_typed_nulls() { - // Every level must emit every key — as NULL where the level omits it — or - // `Concat` (which takes the first child's schema) would misdescribe the rest. - // The null is *cast*: a bare Null literal infers as Float64. - let qe = lower("SELECT service, SUM(bytes) FROM metrics GROUP BY ROLLUP(service)").await; - let levels = grouping_levels(&qe); - assert_eq!(levels.len(), 2); - for (_, names) in &levels { - assert_eq!( - names, - &["service".to_string(), "sum(metrics.bytes)".to_string()] - ); - } - - // The `()` level projects `service` as a Utf8 null, not a Float64 one. - let schema = &merge_branches(&qe)[1].schema; - assert_eq!(schema.fields[0].name, "service"); - assert_eq!( - schema.fields[0].dtype, - DataType::Utf8, - "the omitted key must keep its declared type" - ); -} - -#[tokio::test] -async fn grouping_levels_are_union_compatible() { - let qe = lower( - "SELECT service, bytes, SUM(latency) FROM metrics GROUP BY GROUPING SETS ((service),(bytes),())", - ) - .await; - let shapes: Vec<_> = merge_branches(&qe) - .iter() - .map(|b| { - b.schema - .fields - .iter() - .map(|c| (c.name.clone(), c.dtype.clone())) - .collect::>() - }) - .collect(); - assert!( - shapes.windows(2).all(|w| w[0] == w[1]), - "levels disagree: {shapes:?}" - ); -} - -#[tokio::test] -async fn grouping_function_is_rejected() { - // `__grouping_id` is dropped when the levels are expanded. It is observable - // only through `GROUPING(col)`, so dropping it loses nothing representable — - // this test is what makes that true. - let err = lower_sql( - "SELECT service, SUM(bytes), GROUPING(service) FROM metrics GROUP BY ROLLUP(service)", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .expect_err("GROUPING() must be rejected while __grouping_id is dropped"); - assert!(format!("{err}").contains("grouping"), "got {err}"); -} - -#[tokio::test] -async fn a_non_column_key_inside_a_grouping_set_is_rejected() { - // The #110 derived-column machinery covers plain `GROUP BY `; inside a - // grouping set the key also has to be reinstatable as a typed null. - let err = lower_sql( - "SELECT date_trunc('minute', ts) AS m, SUM(bytes) FROM metrics GROUP BY ROLLUP(m)", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .expect_err("expression key inside ROLLUP must be rejected"); - assert!( - format!("{err}").contains("non-column key inside a multi-level grouping"), - "got {err}" - ); -} - -#[tokio::test] -async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { - // #110's materializing Project sits beneath every level's Aggregate. - let qe = lower("SELECT service, SUM(bytes * 8) FROM metrics GROUP BY ROLLUP(service)").await; - for b in merge_branches(&qe) { - let NonASAPOp::Project { child, .. } = op(b) else { - panic!("expected a Project per level"); - }; - let NonASAPOp::Aggregate { - measures, child, .. - } = op(child) - else { - panic!("expected an Aggregate"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::Sum { col: Some(_) }] - )); - assert!( - matches!(op(child), NonASAPOp::Project { .. }), - "the derived-column projection should sit under each level" - ); - } -} - -#[tokio::test] -async fn an_ambiguous_passthrough_column_is_rejected_only_when_projecting() { - // A `Project` carries one relation qualifier for all its columns, so `a.k` - // and `b.k` cannot both survive it. That only matters once a projection is - // inserted: without a derived column the join keys resolve as before. - let ok = lower_sql( - "SELECT m.service, h.service, SUM(m.bytes) FROM metrics m \ - JOIN hosts h ON m.service = h.service GROUP BY m.service, h.service", - &catalog(), - AccuracyTarget::Exact, - ) - .await; - assert!( - ok.is_ok(), - "no derived column ⇒ no projection ⇒ no ambiguity" - ); - - let err = lower_sql( - "SELECT m.service, h.service, SUM(m.bytes * 2) FROM metrics m \ - JOIN hosts h ON m.service = h.service GROUP BY m.service, h.service", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .expect_err("ambiguous passthrough must be rejected, not silently resolved"); - assert!(format!("{err}").contains("ambiguous column"), "got {err}"); -} - -// ── Issue #111: array_agg is deliberately not an intent (WONTFIX) ─────────── - -#[tokio::test] -async fn array_agg_is_deliberately_rejected() { - // Not a coverage gap. `AggIntent` exists so the planner can bind a sketch or - // a mergeable accumulator per node; `array_agg` pre-aggregates nothing (its - // output is O(input rows)), has no bounded-memory approximate form, and its - // partial state *is* the data. An `AggIntent::ArrayAgg` would force every - // arm of `plan::boundary::realize` — an exhaustive match — to answer - // `PassThrough`. Contrast `median`, which is `Quantile { q: 0.5 }` and does - // feed the sketch path. - // - // This test exists so the rejection reads as a decision rather than a gap. - let err = lower_sql( - "SELECT array_agg(service) FROM metrics", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .expect_err("array_agg must not lower to an intent"); - assert!( - format!("{err}").contains("unsupported aggregate: array_agg"), - "expected a clean UnsupportedAggregate, got {err}" - ); -} - -// ── Issue #225: catalog-driven ClickHouse builtins (countIf, generalizing -// uniqExact from #221) ─────────────────────────────────────────────────── - -async fn lower_clickhouse(sql: &str) -> Rc { - lower_sql_dialect( - sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) -} - -fn temporal_aggregate(node: &OperatorNode) -> (&AggIntent, std::time::Duration, &OperatorNode) { - match op(node) { - NonASAPOp::Aggregate { - reduction: Reduction::PerEntity, - measures, - child, - .. - } => { - let NonASAPOp::TimeRange { range, child, .. } = op(child) else { - panic!("temporal Aggregate must directly wrap TimeRange, got {child:?}"); - }; - (&measures[0], *range, child) - } - NonASAPOp::Project { child, .. } | NonASAPOp::Filter { child, .. } => { - temporal_aggregate(child) - } - other => panic!("expected temporal Aggregate, got {other:?}"), - } -} - -#[tokio::test] -async fn explicit_temporal_aggregates_share_promql_intents_and_timerange() { - for (function, expected) in [ - ("asap_rate", AggIntent::Rate), - ("asap_increase", AggIntent::Increase), - ] { - let sql = format!( - "SELECT service, {function}(latency, ts, 300000) AS v \ - FROM metrics WHERE service = 'api' GROUP BY service" - ); - let qe = lower_clickhouse(&sql).await; - let (intent, range, child) = temporal_aggregate(&qe); - assert_eq!(intent, &expected); - assert_eq!(range, std::time::Duration::from_secs(300)); - assert!(matches!(op(child), NonASAPOp::Project { child, .. } - if matches!(op(child), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1))); - - let NonASAPOp::Project { cols, .. } = op(&qe) else { - panic!("SELECT list must remain a Project, got {qe:?}"); - }; - assert!(matches!(cols[0].expr, ScalarExpr::Column(2))); - assert_eq!(cols[1].alias.as_deref(), Some("v")); - assert!(matches!(cols[1].expr, ScalarExpr::Column(1))); - } -} - -#[tokio::test] -async fn temporal_aggregate_rejects_non_timestamp_and_non_positive_window() { - for sql in [ - "SELECT asap_rate(latency, bytes, 300000) FROM metrics", - "SELECT asap_rate(latency, ts, 0) FROM metrics", - "SELECT asap_rate(latency, ts, bytes) FROM metrics", - ] { - let err = lower_sql_dialect( - sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect_err("invalid temporal arguments must fail closed"); - assert!( - format!("{err}").contains("timestamp argument") - || format!("{err}").contains("window_ms"), - "unexpected error for {sql}: {err}" - ); - } -} - -#[tokio::test] -async fn temporal_aggregate_rejects_mixed_reducers() { - let err = lower_sql_dialect( - "SELECT asap_rate(latency, ts, 300000), sum(bytes) FROM metrics", - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect_err("one child cannot carry temporal and ordinary aggregate semantics"); - assert!(format!("{err}").contains("cannot share an Aggregate node")); -} - -#[tokio::test] -async fn last_fails_closed_until_an_executable_summary_exists() { - let err = lower_sql_dialect( - "SELECT service, asap_last(latency, ts, 300000) FROM metrics GROUP BY service", - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect_err("last must not be advertised without an executable physical summary"); - assert!(format!("{err}").contains("Invalid function 'asap_last'")); -} - -#[tokio::test] -async fn temporal_grouping_requires_the_complete_declared_series_identity() { - let multi_series = SqlCatalog::new().with_table( - "samples", - Schema::with_time_index( - vec![ - col("ts", DataType::Timestamp), - col("service", DataType::Utf8), - col("instance", DataType::Utf8), - col("value", DataType::Float64), - ], - 0, - vec![vec![0, 1, 2]], - ), - ); - for sql in [ - "SELECT asap_rate(value, ts, 300000) FROM samples", - "SELECT service, asap_rate(value, ts, 300000) FROM samples GROUP BY service", - ] { - let err = lower_sql_dialect( - sql, - &multi_series, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect_err("partial identity must not merge counter series"); - assert!(format!("{err}").contains("declared series identity")); - } - - lower_sql_dialect( - "SELECT service, instance, asap_rate(value, ts, 300000) \ - FROM samples GROUP BY service, instance", - &multi_series, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect("the complete declared series identity is safe"); - - let row_id_only = SqlCatalog::new().with_table( - "samples", - Schema::with_time_index( - vec![ - col("ts", DataType::Timestamp), - col("service", DataType::Utf8), - col("value", DataType::Float64), - ], - 0, - vec![vec![1]], - ), - ); - lower_sql_dialect( - "SELECT service, asap_rate(value, ts, 300000) FROM samples GROUP BY service", - &row_id_only, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect_err("a row key without time does not prove a series identity"); -} - -#[tokio::test] -async fn temporal_grouping_rejects_value_time_and_duplicate_resolved_columns() { - for sql in [ - "SELECT asap_rate(latency, ts, 300000) FROM metrics GROUP BY ts", - "SELECT asap_rate(latency, ts, 300000) FROM metrics GROUP BY latency", - "SELECT m.service, asap_rate(m.latency, m.ts, 300000) \ - FROM metrics m GROUP BY m.service, service", - ] { - let err = lower_sql_dialect( - sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect_err("unsafe or duplicate resolved grouping must fail closed"); - let message = format!("{err}"); - assert!( - message.contains("timestamp or value") - || message.contains("same resolved column more than once"), - "unexpected error for {sql}: {message}" - ); - } -} - -#[tokio::test] -async fn qualified_columns_are_validated_by_resolved_identity() { - let qe = lower_clickhouse( - "SELECT m.service, asap_increase(m.latency, m.ts, 300000) AS v \ - FROM metrics AS m GROUP BY m.service", - ) - .await; - let (intent, range, _) = temporal_aggregate(&qe); - assert_eq!(intent, &AggIntent::Increase); - assert_eq!(range, std::time::Duration::from_secs(300)); -} - -#[tokio::test] -async fn project_filter_and_outer_aggregate_preserve_temporal_child() { - let qe = lower_clickhouse( - "SELECT max(v) FROM (\ - SELECT service, asap_rate(latency, ts, 300000) AS v \ - FROM metrics WHERE bytes > 0 GROUP BY service\ - ) r WHERE v >= 0", - ) - .await; - let NonASAPOp::Project { child, .. } = op(&qe) else { - panic!("expected outer SELECT Project, got {qe:?}"); - }; - let NonASAPOp::Aggregate { - reduction: Reduction::Reduce(_), - measures, - child, - .. - } = op(child) - else { - panic!("expected outer Aggregate, got {child:?}"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let NonASAPOp::Filter { child, .. } = op(child) else { - panic!("derived-table WHERE must remain above the inner query, got {child:?}"); - }; - let (intent, range, _) = temporal_aggregate(child); - assert_eq!(intent, &AggIntent::Rate); - assert_eq!(range, std::time::Duration::from_secs(300)); -} - -#[tokio::test] -async fn count_if_lowers_to_a_sum_over_a_derived_indicator_column() { - // ClickHouse's `countIf(cond)` has no DataFusion equivalent at all, so it - // goes through the same stub-UDAF + catalog-driven `FunctionRewrite` - // mechanism `uniqExact` (#221) does — rewritten, before `lower_agg_intent` - // ever runs, to `sum(CASE WHEN cond THEN 1 ELSE 0 END)`. A per-measure - // filter (#466) could express it as a filtered `Count` now; that move is - // a follow-up, so the indicator sum is still the shape to expect. - let qe = lower_clickhouse("SELECT countIf(bytes > 100) AS big FROM metrics").await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!(by.is_empty()); - assert!( - matches!(measures.as_slice(), [AggIntent::Sum { col: Some(_) }]), - "expected a Sum bound to the derived indicator column, got {measures:?}" - ); - let (_, materialized) = reducer_input_names(&qe); - assert!( - materialized, - "the indicator expression must be materialized in a Project beneath the Aggregate" - ); -} - -#[tokio::test] -async fn two_count_ifs_with_different_conditions_stay_distinct_reducers() { - // The corpus pattern (`countIf(operation = 'A'), countIf(operation = 'W')` - // in one GROUP BY) needs each call's own condition to survive as its own - // derived column, not collapse onto a shared one. - let qe = lower_clickhouse( - "SELECT service, countIf(bytes > 100) AS big, countIf(bytes <= 100) AS small \ - FROM metrics GROUP BY service", - ) - .await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert_eq!(*by, GroupKeys::by(vec![0])); - assert!( - matches!( - measures.as_slice(), - [ - AggIntent::Sum { col: Some(a) }, - AggIntent::Sum { col: Some(b) } - ] if a != b - ), - "expected two distinct Sum reducers, got {measures:?}" - ); -} - -#[tokio::test] -async fn count_if_composes_with_group_by() { - let qe = lower_clickhouse( - "SELECT service, countIf(bytes > 100) AS big FROM metrics GROUP BY service", - ) - .await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert_eq!(*by, GroupKeys::by(vec![0])); - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); -} - -// ── Issue #232: argMax/argMin -- AggIntent::Extension, not a first-class -// core variant. A repo-wide search (PromQL front end, other SQL dialects, -// docs) turned up no second deployment model wanting this two-column, -// row-selecting shape, so per `AggIntent::Extension`'s own "core only grows -// for intents ≥2 deployment models actually use" bar, it stays an opaque -// `Extension` rather than a new `ArgMax`/`ArgMin` core variant. Unlike -// `countIf`/`uniqExact`, there is no native DataFusion aggregate shape to -// rewrite to (`RewriteKind::PassThrough`) -- `lower_agg_intent` builds the -// `AggIntent` directly from the ClickHouse name. ───────────────────────── - -#[tokio::test] -async fn arg_max_lowers_to_an_extension_intent() { - // No existing `AggIntent` reducer fits: every one folds one column to a - // value derived from itself, while `argMax(arg, val)` returns a - // *different* column's value, selected by which row maximizes a second. - let qe = lower_clickhouse( - "SELECT service, argMax(service, latency) AS busiest FROM metrics GROUP BY service", - ) - .await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert_eq!( - *by, - GroupKeys::by(vec![1]), - "grouped by `service` (schema index 1)" - ); - assert!( - matches!( - measures.as_slice(), - [AggIntent::Extension { ext_kind, .. }] if ext_kind == "arg_max" - ), - "expected Extension {{ ext_kind: \"arg_max\", .. }}, got {measures:?}" - ); -} - -#[tokio::test] -async fn arg_min_lowers_to_its_own_extension_kind() { - let qe = lower_clickhouse("SELECT argMin(service, latency) FROM metrics").await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - assert!(by.is_empty()); - assert!( - matches!( - measures.as_slice(), - [AggIntent::Extension { ext_kind, .. }] if ext_kind == "arg_min" - ), - "expected Extension {{ ext_kind: \"arg_min\", .. }}, got {measures:?}" - ); -} - -#[tokio::test] -async fn arg_max_payload_preserves_both_column_names() { - // Core never resolves an `Extension`'s payload, so both columns are kept - // as validated bare-column `ColumnRef`s in `payload`, not run through - // positional `ColumnId` binding -- see `lower_arg_selector`'s doc. - let qe = lower_clickhouse("SELECT argMax(service, latency) AS m FROM metrics").await; - let (_, measures) = find_aggregate(&qe).expect("expected an Aggregate"); - let AggIntent::Extension { payload, .. } = &measures[0] else { - panic!("expected an Extension intent, got {:?}", measures[0]); - }; - let named = |key: &str| { - payload - .get(key) - .and_then(|c| c.get("Named")) - .and_then(|n| n.as_str()) - .map(str::to_string) - }; - assert_eq!(named("arg_col"), Some("service".to_string())); - assert_eq!(named("val_col"), Some("latency".to_string())); -} - -#[tokio::test] -async fn arg_max_rejects_a_non_column_argument() { - // Same "bare column only" rule as every other reducer (`reducer_col`, - // issue #115) -- an expression argument is rejected, not silently - // dropped or materialized into the wrong column. - let err = lower_sql_dialect( - "SELECT argMax(service, latency * 2) FROM metrics", - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect_err("argMax over a non-column expression must be rejected"); - assert!( - format!("{err}").contains("non-column expression"), - "got {err}" - ); -} - -// ── Issue #267: lagInFrame/leadInFrame get distinct WindowFuncKind variants, -// not conflated with ANSI Lag/Lead ────────────────────────────────────────── - -#[tokio::test] -async fn lag_in_frame_lowers_to_its_own_kind_not_lag() { - let qe = lower_clickhouse( - "SELECT service, lagInFrame(bytes) OVER (PARTITION BY service ORDER BY ts) \ - FROM metrics", - ) - .await; - let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { - unreachable!(); - }; - assert_eq!(*func, WindowFuncKind::LagInFrame); - assert_eq!( - args, - &vec![ScalarExpr::Column(3)], - "lagInFrame(bytes) → arg col 3" - ); -} - -#[tokio::test] -async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { - let qe = lower_clickhouse( - "SELECT service, leadInFrame(bytes) OVER (PARTITION BY service ORDER BY ts) \ - FROM metrics", - ) - .await; - let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let NonASAPOp::SQLWindowFunc { func, .. } = op(win) else { - unreachable!(); - }; - assert_eq!(*func, WindowFuncKind::LeadInFrame); -} - -/// Issue #184: `NOW()` in a predicate must lower to the timestamp-typed -/// `CurrentTimestamp` leaf, not the semantically-opaque function catch-all or -/// PromQL's Float64 Unix-seconds `EvalTimestamp`. -#[tokio::test] -async fn now_in_predicate_lowers_to_current_timestamp() { - // WHERE folds onto Scan.predicates (no explicit Filter node). - let qe = lower("SELECT * FROM metrics WHERE ts < NOW()").await; - let NonASAPOp::Project { child, .. } = op(&qe) else { - panic!("expected Project at root, got {qe:?}"); - }; - let NonASAPOp::Scan { predicates, .. } = op(child) else { - panic!("expected Scan under the projection, got {child:?}"); - }; - assert_eq!(predicates.len(), 1); - assert!( - matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } - if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), - "NOW() must lower to CurrentTimestamp, got {:?}", - predicates[0].0 - ); -} - -/// Same for ClickHouse's `now()`, since #184 was raised specifically against -/// the ClickHouse dialect. -#[tokio::test] -async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { - let qe = lower_clickhouse("SELECT * FROM metrics WHERE ts < now()").await; - let NonASAPOp::Project { child, .. } = op(&qe) else { - panic!("expected Project at root, got {qe:?}"); - }; - let NonASAPOp::Scan { predicates, .. } = op(child) else { - panic!("expected Scan under the projection, got {child:?}"); - }; - assert_eq!(predicates.len(), 1); - assert!( - matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } - if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), - "now() must lower to CurrentTimestamp, got {:?}", - predicates[0].0 - ); -} - -#[tokio::test] -async fn current_timestamp_lowers_to_typed_current_timestamp_leaf() { - let qe = lower("SELECT CURRENT_TIMESTAMP FROM metrics").await; - let NonASAPOp::Project { cols, child, .. } = op(&qe) else { - panic!("expected Project at root, got {qe:?}"); - }; - assert!(matches!(&cols[0].expr, ScalarExpr::CurrentTimestamp)); - let (dtype, _) = cols[0] - .expr - .scalar_type(&child.schema) - .expect("timestamp type"); - assert_eq!(dtype, DataType::Timestamp); - assert_eq!(qe.schema.fields[0].dtype, DataType::Timestamp); -} - -// A `count` over a non-null input is a plain row count; over a nullable -// input it keeps SQL's NULL-skipping as the measure's own filter (#466), and -// only the multi-level grouping path, which cannot carry one, still rejects it. -#[tokio::test] -async fn count_null_semantics_become_a_measure_filter() { - let catalog = SqlCatalog::new().with_table( - "samples", - Schema::new(vec![ - Field::plain("nullable_value", DataType::Float64, true), - Field::plain("value", DataType::Float64, false), - ]), - ); - for sql in [ - "SELECT count(*) FROM samples", - "SELECT count(1) FROM samples", - "SELECT count(value) FROM samples", - "SELECT count(value + 1) FROM samples", - ] { - let qe = lower_sql(sql, &catalog, AccuracyTarget::Exact) - .await - .unwrap_or_else(|error| panic!("{sql}: {error}")); - assert!( - aggregate_filters(&qe).is_empty(), - "{sql}: unfiltered row count" - ); - } - for sql in [ - "SELECT count(nullable_value) FROM samples", - "SELECT count(NULL) FROM samples", - "SELECT count(nullable_value + 1) FROM samples", - ] { - let qe = lower_sql(sql, &catalog, AccuracyTarget::Exact) - .await - .unwrap_or_else(|error| panic!("{sql}: {error}")); - let [Some(Predicate(cond))] = aggregate_filters(&qe) else { - panic!( - "{sql}: expected one filtered Count, got {:?}", - aggregate_filters(&qe) - ); - }; - assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{sql}: {cond:?}"); - } - // Only the second measure is filtered. - let qe = lower_sql( - "SELECT count(*), count(nullable_value) FROM samples", - &catalog, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - assert!(matches!(aggregate_filters(&qe), [None, Some(_)])); - let error = lower_sql( - "SELECT count(nullable_value) FROM samples GROUP BY ROLLUP(value)", - &catalog, - AccuracyTarget::Exact, - ) - .await - .unwrap_err(); - assert!( - matches!(error, LoweringError::UnsupportedFeature(_)), - "{error}" - ); -} - -/// A native SQL map grouping key retains its typed key/value schema. -#[tokio::test] -async fn grouped_map_column_preserves_map_type() { - let map = DataType::Map { - key: Box::new(DataType::Utf8), - value: Box::new(DataType::Utf8), - value_nullable: false, - }; - let catalog = SqlCatalog::new().with_table( - "raw_samples", - Schema::new(vec![ - col("labels", map.clone()), - col("value", DataType::Float64), - ]), - ); - let query = lower_sql_dialect( - "SELECT labels, max(value) AS value FROM raw_samples GROUP BY labels ORDER BY labels", - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - assert_eq!(query.schema.fields[0].dtype, map); -} - -#[tokio::test] -async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { - let catalog = SqlCatalog::new().with_table( - "numbers", - Schema::new(vec![ - Field::plain("i", DataType::Int64, false), - Field::plain("n", DataType::Int64, true), - Field::plain("f", DataType::Float64, false), - ]), - ); - for (call, native) in [ - ("modulo(i, 3)", "i % 3"), - ("modulo(n, -3)", "n % -3"), - ("modulo(f, 2.5)", "f % 2.5"), - ("modulo(-7, 3)", "-7 % 3"), - ("modulo(i, 0)", "i % 0"), - ("modulo(modulo(i, 5), 2)", "(i % 5) % 2"), - ] { - let function = lower_sql_dialect( - &format!("SELECT {call} AS value FROM numbers"), - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - let operator = lower_sql_dialect( - &format!("SELECT {native} AS value FROM numbers"), - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - assert_eq!(function, operator, "{call}"); - assert_eq!(function.schema, operator.schema); - } - let nullable = lower_sql_dialect( - "SELECT modulo(n, 3) AS value FROM numbers", - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap() - .schema - .clone(); - assert_eq!(nullable.fields[0].dtype, DataType::Int64); - assert!(nullable.fields[0].nullable); -} - -#[tokio::test] -async fn original_o11y_map_queries_lower_with_typed_results() { - let catalog = SqlCatalog::new().with_table( - "raw_samples", - Schema::new(vec![ - Field::plain("metric", DataType::Utf8, false), - Field::plain("ts_ms", DataType::Int64, false), - Field::plain("value", DataType::Float64, false), - Field::plain( - "labels", - DataType::Map { - key: Box::new(DataType::Utf8), - value: Box::new(DataType::Utf8), - value_nullable: false, - }, - false, - ), - ]), - ); - for sql in [ - include_str!("data/o11y_q10.sql"), - include_str!("data/o11y_q27.sql"), - include_str!("data/o11y_q07.sql"), - include_str!("data/o11y_q09.sql"), - include_str!("data/o11y_q12.sql"), - ] { - let query = lower_sql_dialect( - sql, - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap_or_else(|e| panic!("{sql}: {e}")); - let schema = &query.schema; - assert!( - schema - .fields - .iter() - .any(|column| matches!(column.dtype, FieldDataType::Plain(DataType::Map { .. }))), - "{schema:?}" - ); - } -} - -#[tokio::test] -async fn clickhouse_modulo_preserves_projection_names_and_outer_references() { - for (sql, name) in [ - ("SELECT modulo(bytes, 3) FROM metrics", "modulo(bytes, 3)"), - ( - "SELECT modulo(bytes, 3) AS remainder FROM metrics", - "remainder", - ), - ( - "SELECT \"modulo(bytes, 3)\" FROM (SELECT modulo(bytes, 3) FROM metrics) t", - "modulo(bytes, 3)", - ), - ] { - let query = lower_sql_dialect( - sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - assert_eq!(query.schema.fields[0].name, name); - } -} - -#[tokio::test] -async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercion() { - let catalog = SqlCatalog::new().with_table( - "t", - Schema::new(vec![ - Field::plain( - "labels", - DataType::Map { - key: Box::new(DataType::Utf8), - value: Box::new(DataType::Utf8), - value_nullable: false, - }, - false, - ), - Field::plain("integer", DataType::Int64, false), - Field::plain("floating", DataType::Float64, false), - ]), - ); - let query = lower_sql_dialect( - "SELECT labels['job'] FROM t", - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - let output = &query.schema; - assert_eq!(output.fields[0].name, "arrayElement(labels, 'job')"); - assert_eq!(output.fields[0].dtype, DataType::Utf8); - assert!(!output.fields[0].nullable); - assert!(lower_sql_dialect( - "SELECT map()['a'] FROM t", - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .is_err()); - assert!(lower_sql_dialect( - "SELECT map('a', integer, 'b', floating) FROM t", - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact - ) - .await - .is_err()); -} - -#[tokio::test] -async fn arg_selector_result_schema_tracks_selected_argument() { - let catalog = SqlCatalog::new().with_table( - "t", - Schema::new(vec![ - Field::plain("v", DataType::Float64, false), - Field::plain("text", DataType::Utf8, true), - Field::plain("ts", DataType::Int64, true), - ]), - ); - for (sql, dtype, nullable) in [ - ( - "SELECT argMax(v, ts) AS value FROM t", - DataType::Float64, - false, - ), - ( - "SELECT argMin(text, ts) AS value FROM t", - DataType::Utf8, - true, - ), - ] { - let query = lower_sql_dialect( - sql, - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - let schema = &query.schema; - assert_eq!(schema.fields[0].dtype, dtype); - assert_eq!(schema.fields[0].nullable, nullable); - } -} - -#[tokio::test] -async fn clickhouse_list_element_uses_canonical_typed_access() { - let catalog = SqlCatalog::new().with_table( - "t", - Schema::new(vec![ - Field::plain( - "samples", - DataType::List { - element: Box::new(Field::new("item", DataType::Int64, false)), - }, - false, - ), - Field::plain("index", DataType::Int64, true), - ]), - ); - for (sql, nullable) in [ - ("SELECT samples[1] AS selected FROM t", false), - ("SELECT arrayElement(samples, -1) AS selected FROM t", false), - ("SELECT samples[index] AS selected FROM t", true), - ] { - let query = lower_sql_dialect( - sql, - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - let output = &query.schema; - assert_eq!(output.fields[0].dtype, DataType::Int64); - assert_eq!(output.fields[0].nullable, nullable); - let serialized = serde_json::to_string(&query).unwrap(); - assert!(serialized.contains("asap_element_access"), "{serialized}"); - } - for sql in ["SELECT samples[0] FROM t", "SELECT samples['bad'] FROM t"] { - assert!( - lower_sql_dialect( - sql, - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact - ) - .await - .is_err(), - "{sql}" - ); - } -} - -#[tokio::test] -async fn clickhouse_tuple_element_preserves_declared_field_metadata() { - let catalog = SqlCatalog::new().with_table( - "t", - Schema::new(vec![ - Field::plain( - "sample", - DataType::Struct { - fields: vec![ - Field::new("time", DataType::Int64, false), - Field::new("value", DataType::Float64, true), - ], - }, - false, - ), - Field::plain("index", DataType::Int64, false), - ]), - ); - for (sql, dtype, nullable) in [ - ( - "SELECT tupleElement(sample, 1) AS chosen FROM t", - DataType::Int64, - false, - ), - ( - "SELECT tupleElement(sample, 'value') AS chosen FROM t", - DataType::Float64, - true, - ), - ] { - let query = lower_sql_dialect( - sql, - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - let output = &query.schema; - assert_eq!(output.fields[0].dtype, dtype); - assert_eq!(output.fields[0].nullable, nullable); - assert!(serde_json::to_string(&query) - .unwrap() - .contains("asap_struct_field")); - } - for selector in ["0", "-1", "3", "'missing'", "index"] { - let sql = format!("SELECT tupleElement(sample, {selector}) FROM t"); - assert!( - lower_sql_dialect( - &sql, - &catalog, - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact - ) - .await - .is_err(), - "{sql}" - ); - } -} - -/// Correlation lowers to a nullable numeric result instead of UnsupportedAggregate. -#[tokio::test] -async fn corr_result_is_nullable_float() { - let query = lower("SELECT corr(latency, bytes) AS correlation FROM metrics").await; - let schema = &query.schema; - assert_eq!(schema.fields[0].name, "correlation"); - assert_eq!(schema.fields[0].dtype, DataType::Float64); - assert!(schema.fields[0].nullable); -} - -// A multi-column DISTINCT counts tuples; one column stays the single-column -// intent, so neither form can be mistaken for the other downstream. -#[tokio::test] -async fn composite_distinct_counts_tuples() { - let cat = SqlCatalog::new().with_table( - "t", - Schema::new(vec![ - Field::plain("a", DataType::Int64, false), - Field::plain("b", DataType::Int64, false), - ]), - ); - let composite = lower_sql( - "SELECT COUNT(DISTINCT a, b) FROM t", - &cat, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - let NonASAPOp::Aggregate { measures, .. } = - op(find_aggregate_node(&composite).expect("expected an Aggregate")) - else { - unreachable!() - }; - assert!( - matches!(measures.as_slice(), [AggIntent::Cardinality { cols, .. }] if cols == &[0, 1]), - "{measures:?}" - ); - - let single = lower_sql( - "SELECT COUNT(DISTINCT a) FROM t", - &cat, - AccuracyTarget::Exact, - ) - .await - .unwrap(); - let NonASAPOp::Aggregate { measures, .. } = - op(find_aggregate_node(&single).expect("expected an Aggregate")) - else { - unreachable!() - }; - assert!( - matches!(measures.as_slice(), [AggIntent::Cardinality { cols, .. }] if cols == &[0]), - "{measures:?}" - ); -} - -// An expression argument has no column identity to hash, so it is rejected -// rather than silently reduced over a probe column. -#[tokio::test] -async fn composite_distinct_rejects_expression_arguments() { - let cat = SqlCatalog::new().with_table( - "t", - Schema::new(vec![ - Field::plain("a", DataType::Int64, false), - Field::plain("b", DataType::Int64, false), - ]), - ); - let error = lower_sql( - "SELECT COUNT(DISTINCT a, b + 1) FROM t", - &cat, - AccuracyTarget::Exact, - ) - .await - .unwrap_err(); - assert!( - matches!(&error, LoweringError::UnsupportedAggregate(reason) - if reason.contains("non-column expression")), - "{error}" - ); -} - -// DISTINCT inputs survive projections introduced by sibling aggregates. -#[tokio::test] -async fn distinct_with_derived_sibling() { - let catalog = SqlCatalog::new().with_table( - "t", - Schema::new(vec![ - Field::plain("a", DataType::Int64, false), - Field::plain("b", DataType::Int64, false), - ]), - ); - for sql in [ - "SELECT count(DISTINCT a), sum(b + 1) FROM t", - "SELECT count(DISTINCT a, b), sum(b + 1) FROM t", - "SELECT count(DISTINCT a, b), corr(a,b) FROM t", - ] { - let result = lower_sql(sql, &catalog, AccuracyTarget::Exact).await; - assert!(result.is_ok(), "{sql}: {result:?}"); - } -} - -// ── Issue #466: per-measure FILTER predicates ───────────────────────────────── - -/// The first `Aggregate`'s `filters`, positional against its child. -fn aggregate_filters(qe: &OperatorNode) -> &[Option] { - let Some(NonASAPOp::Aggregate { filters, .. }) = - find_aggregate_node(qe).map(|n| n.expect_non_asap()) - else { - panic!("expected an Aggregate, got {qe:?}"); - }; - filters -} - -// The motivating query: one scan, one grouping, one conditional count next to -// a plain sum — a single `Aggregate` whose Count carries the condition, with no -// `Join` and no derived column for the `CASE`. -#[tokio::test] -async fn conditional_count_lowers_to_a_filtered_measure() { - let qe = lower( - "SELECT service, count(CASE WHEN latency > 1.0 THEN 1 END), sum(bytes) \ - FROM metrics GROUP BY service", - ) - .await; - assert!(find_join(&qe).is_none(), "no join: {qe:?}"); - let (by, measures) = find_aggregate(&qe).unwrap(); - assert_eq!(by.keys(), &[1]); - assert!( - matches!( - measures.as_slice(), - [AggIntent::Count { .. }, AggIntent::Sum { col: Some(3) }] - ), - "{measures:?}" - ); - let [Some(Predicate(cond)), None] = aggregate_filters(&qe) else { - panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); - }; - assert!( - matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Gt, .. } - if matches!(left.as_ref(), ScalarExpr::Column(2))), - "latency > 1.0 against the scan, got {cond:?}" - ); - let Some(NonASAPOp::Aggregate { child, .. }) = - find_aggregate_node(&qe).map(|n| n.expect_non_asap()) - else { - unreachable!() - }; - assert!( - matches!(child.expect_non_asap(), NonASAPOp::Scan { .. }), - "{child:?}" - ); -} - -// `FILTER (WHERE …)` parses under the DataFusion dialect and lands on exactly -// the measure it annotates. -#[tokio::test] -async fn filter_clause_lowers_to_a_measure_filter() { - let qe = lower("SELECT sum(bytes) FILTER (WHERE service = 'a'), count(*) FROM metrics").await; - let [Some(Predicate(cond)), None] = aggregate_filters(&qe) else { - panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); - }; - assert!( - matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, .. } - if matches!(left.as_ref(), ScalarExpr::Column(1)) - && matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(s)) if s == "a")), - "{cond:?}" - ); -} - -// SQL `count(expr)` skips NULLs; canonical `Count` counts rows and never sees -// `expr`, so a nullable argument becomes the measure filter `expr IS NOT NULL` -// instead of being rejected (the pre-#466 behavior) or silently over-counted. -#[tokio::test] -async fn count_of_a_nullable_expression_filters_nulls() { - let qe = lower("SELECT count(nullif(bytes, 0)) FROM metrics").await; - let [Some(Predicate(cond))] = aggregate_filters(&qe) else { - panic!("expected [Some], got {:?}", aggregate_filters(&qe)); - }; - assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{cond:?}"); - assert!( - matches!( - find_aggregate(&qe).unwrap().1.as_slice(), - [AggIntent::Count { .. }] - ), - "still a row count" - ); -} - -// The columns a measure filter reads must survive the derived-column -// `Project` a reducer expression inserts beneath the aggregate. -#[tokio::test] -async fn measure_filter_columns_survive_a_derived_column_projection() { - let qe = lower("SELECT sum(bytes * 2) FILTER (WHERE latency > 1.0) FROM metrics").await; - let Some(NonASAPOp::Aggregate { child, .. }) = - find_aggregate_node(&qe).map(|n| n.expect_non_asap()) - else { - unreachable!() - }; - assert!( - matches!(child.expect_non_asap(), NonASAPOp::Project { .. }), - "{child:?}" - ); - let [Some(Predicate(cond))] = aggregate_filters(&qe) else { - panic!("expected [Some], got {:?}", aggregate_filters(&qe)); - }; - let ScalarExpr::Compare { left, .. } = cond else { - panic!("{cond:?}"); - }; - let ScalarExpr::Column(id) = left.as_ref() else { - panic!("{left:?}"); - }; - assert_eq!(child.schema.fields[*id].name, "latency"); -} - -// `GROUP BY ROLLUP` fans one measure list out into one `Aggregate` per level; -// a filtered measure there is rejected rather than silently unfiltered. -#[tokio::test] -async fn measure_filter_inside_a_rollup_is_rejected() { - let err = lower_sql( - "SELECT service, count(*) FILTER (WHERE latency > 1.0) FROM metrics GROUP BY ROLLUP(service)", - &catalog(), - AccuracyTarget::Exact, - ) - .await - .unwrap_err(); - assert!(matches!(err, LoweringError::UnsupportedFeature(_)), "{err}"); -} diff --git a/crates/integration-tests/Cargo.toml b/crates/integration-tests/Cargo.toml index 5a5de9902..afa7559b4 100644 --- a/crates/integration-tests/Cargo.toml +++ b/crates/integration-tests/Cargo.toml @@ -10,6 +10,7 @@ asap-frontend-sql = { path = "../frontend-sql" } asap-aware-mapping = { path = "../asap-aware-mapping" } [dev-dependencies] +asap-planner = { path = "../planner" } asap_sketchlib = { workspace = true } serde_json = "1" tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } diff --git a/crates/integration-tests/src/lib.rs b/crates/integration-tests/src/lib.rs index be8e259bc..35822710d 100644 --- a/crates/integration-tests/src/lib.rs +++ b/crates/integration-tests/src/lib.rs @@ -12,21 +12,34 @@ //! here derives or computes expected outputs. pub mod fixtures { - use asap_frontend_promql::lower_promql_workload; + + use asap_types::ir::OperatorNode; use asap_types::pre_asap::schema::{DataType, Field, Schema}; - use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; + use std::rc::Rc; /// Lower one query through the plan-ready workload API using the test /// suite's declared one-second source cadence. pub fn lower_promql( query: &str, accuracy: AccuracyTarget, - ) -> Result { + ) -> Result, asap_frontend_promql::PromqlError> { + match lower_promql_root(query, accuracy)? { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + _ => Err(asap_frontend_promql::PromqlError::UnsupportedFeature( + "expected vector root".into(), + )), + } + } + + pub fn lower_promql_root( + query: &str, + accuracy: AccuracyTarget, + ) -> Result { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -51,7 +64,7 @@ pub mod fixtures { ..Default::default() }), }; - let mut lowered = lower_promql_workload(&workload, 0)?; + let mut lowered = asap_frontend_promql::lower_promql_query_workload(&workload, 0)?; Ok(lowered.remove(0)) } @@ -83,3 +96,26 @@ pub mod fixtures { } } } + +/// Timing and export helpers for post-ASAP plans. +pub mod post_asap { + use asap_types::ir::physical_export::{compile_physical_asap_dag, PhysicalASAPDAG}; + use asap_types::ir::{apply_lifecycle_timings, LifecycleAssignment, OperatorNode, TimingMemo}; + use std::rc::Rc; + + /// Time `root` under the default (every summary maintained) lifecycle + /// assignment. Returns the timed copy; read `node.timing` on it. + pub fn timed(root: &Rc) -> Rc { + apply_lifecycle_timings( + root, + &LifecycleAssignment::default_maintained(), + &mut TimingMemo::new(), + ) + .expect("default lifecycle timing failed") + } + + /// Time `root` (default assignment), then export the wire-6 DAG. + pub fn post_asap_dag(root: &Rc) -> PhysicalASAPDAG { + compile_physical_asap_dag(&timed(root)).expect("post-ASAP DAG export failed") + } +} diff --git a/crates/integration-tests/tests/aggregate.rs b/crates/integration-tests/tests/aggregate.rs index 051eeb6b7..13d2875b9 100644 --- a/crates/integration-tests/tests/aggregate.rs +++ b/crates/integration-tests/tests/aggregate.rs @@ -1,47 +1,54 @@ -//! `QueryExpr::Aggregate` — cross-series aggregation tests. +//! `NonASAPOp::Aggregate` — cross-series aggregation tests. //! //! topk/bottomk are omitted — dispatch is deferred. //! -//! Cross-series aggregates lower to a single `Aggregate` node with no -//! `TimeRange` child (range functions use `TimeRange` — see `time_range.rs`). -//! Group keys land on `Aggregate.by` as positional `ColumnId`s. -//! Single-stat PromQL aggregates always get `output_names: [""]` (no alias) -//! and `having: None`. +//! Cross-series aggregates lower to a single `Aggregate` node over the +//! instant-selector `TimeRange` (range functions use a `Range` selector — +//! see `time_range.rs`). Group keys land on `Aggregate.reduction` as +//! positional `ColumnId`s. Single-stat PromQL aggregates always get +//! `output_names: [""]` (no alias) and `having: None`. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; +use asap_types::pre_asap::{AggIntent, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(child), + kind: TimeRangeKind::Instant, + child, }), - } + }) } // #5 — sum with no group keys @@ -169,7 +176,7 @@ fn q_stdvar_no_group() { ); } -// #10 — cross-series quantile; no TimeRange node (no range window) +// #10 — cross-series quantile; instant selector, no range window #[test] fn q10_quantile_cross_series() { assert_eq!( diff --git a/crates/integration-tests/tests/binary_op.rs b/crates/integration-tests/tests/binary_op.rs index 35f1c632d..284501080 100644 --- a/crates/integration-tests/tests/binary_op.rs +++ b/crates/integration-tests/tests/binary_op.rs @@ -1,92 +1,121 @@ -//! `QueryExpr::BinaryOp` — arithmetic, comparison, and vector-match tests. +//! `NonASAPOp::BinaryOp` — arithmetic, comparison, and vector-match tests. //! //! Each side of a `BinaryOp` is bound independently by the SchemaResolver, so each //! gets its own scan schema derived from the labels it references. -//! `VectorMatch` labels (e.g. `on(job)`) are carried as strings on the node -//! and are NOT resolved to column ids — the SchemaResolver does not see them. +//! `VectorMatch` labels (e.g. `on(job)`) are carried as strings on the +//! operator and are NOT resolved to column ids — the SchemaResolver does not +//! see them. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind}; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, GroupSide, QueryExpr, Reduction, - Source, VectorGrouping, VectorMatch, VectorMatchKind, + AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, GroupSide, Reduction, Source, + VectorGrouping, VectorMatch, VectorMatchKind, }; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::TimeRange { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(source_scan(metric, labels)), - } + kind: TimeRangeKind::Instant, + child: source_scan(metric, labels), + }) } -fn source_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn source_scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn rate_agg(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn rate_agg(metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Rate], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(source_scan(metric, &[])), + kind: TimeRangeKind::Range, + child: source_scan(metric, &[]), }), - } + }) } -fn sum_by_job(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn sum_by_job(metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(scan(metric, &["job"])), - } + child: scan(metric, &["job"]), + }) +} + +/// A PromQL binary operator: no checked-division flags, no `bool` modifier. +fn binary( + kind: BinaryOpKind, + vector_match: Option, + lhs: Rc, + rhs: Rc, +) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + }, + return_bool: false, + lhs, + rhs, + }) } // #18 — arithmetic binary op between two bare scans; no vector match #[test] fn q18_div_bare_scans() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + scan("http_requests_total", &[]), + scan("http_requests_total", &[]), + ); assert_eq!(lower("http_requests_total / http_requests_total"), expected); } // #19 — add with on(job) vector match; match labels are strings, not column ids #[test] fn q19_add_with_on_match() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: Some(VectorMatch { + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: None, }), - }; + scan("http_requests_total", &[]), + scan("http_requests_total", &[]), + ); assert_eq!( lower("http_requests_total + on(job) http_requests_total"), expected @@ -96,12 +125,12 @@ fn q19_add_with_on_match() { // #20 — divide two rate aggregates over different metrics #[test] fn q20_div_two_rates() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(rate_agg("http_requests_total")), - rhs: Rc::new(rate_agg("http_errors_total")), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + rate_agg("http_requests_total"), + rate_agg("http_errors_total"), + ); assert_eq!( lower("rate(http_requests_total[5m]) / rate(http_errors_total[5m])"), expected, @@ -113,12 +142,12 @@ fn q20_div_two_rates() { fn q_gt_comparison() { assert_eq!( lower("http_requests_total > http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Gt), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -126,12 +155,12 @@ fn q_gt_comparison() { fn q_lt_comparison() { assert_eq!( lower("http_requests_total < http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Lt), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Lt), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -139,12 +168,12 @@ fn q_lt_comparison() { fn q_ge_comparison() { assert_eq!( lower("http_requests_total >= http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Ge), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Ge), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -152,12 +181,12 @@ fn q_ge_comparison() { fn q_le_comparison() { assert_eq!( lower("http_requests_total <= http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Le), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Le), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -166,16 +195,16 @@ fn q_le_comparison() { fn q_add_with_ignoring() { assert_eq!( lower("http_requests_total + ignoring(job) http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + Some(VectorMatch { kind: VectorMatchKind::Ignoring, labels: vec!["job".into()], grouping: None, }), - } + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -184,11 +213,9 @@ fn q_add_with_ignoring() { fn q_mul_group_left() { assert_eq!( lower("http_requests_total * on(job) group_left() node_info"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("node_info", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: Some(VectorGrouping { @@ -196,7 +223,9 @@ fn q_mul_group_left() { labels: vec![], }), }), - } + scan("http_requests_total", &[]), + scan("node_info", &[]), + ) ); } @@ -205,11 +234,9 @@ fn q_mul_group_left() { fn q_mul_group_right() { assert_eq!( lower("node_info * on(job) group_right() http_requests_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("node_info", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: Some(VectorGrouping { @@ -217,7 +244,9 @@ fn q_mul_group_right() { labels: vec![], }), }), - } + scan("node_info", &[]), + scan("http_requests_total", &[]), + ) ); } @@ -225,12 +254,12 @@ fn q_mul_group_right() { // each side: Aggregate{Sum, by=[2]} over Scan([ts, value, job]) #[test] fn q21_div_two_sum_by_job() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(sum_by_job("http_requests_total")), - rhs: Rc::new(sum_by_job("http_errors_total")), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + sum_by_job("http_requests_total"), + sum_by_job("http_errors_total"), + ); assert_eq!( lower("sum by (job) (http_requests_total) / sum by (job) (http_errors_total)"), expected, @@ -238,34 +267,29 @@ fn q21_div_two_sum_by_job() { } // #36 — unary negation lowers as `expr * -1`: a Mul BinaryOp of the vector -// against PromqlScalarBridge(-1), no vector match. The vector side keeps its schema. +// against a `ScalarExpr(-1)` leaf, no vector match. The vector side keeps +// its schema. #[test] fn q36_unary_negation_is_multiply_by_minus_one() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("some_metric", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), - vector_match: None, + let root = lower("-some_metric"); + let NonASAPOp::Project { cols, child, .. } = root.expect_non_asap() else { + panic!() }; - assert_eq!(lower("-some_metric"), expected); + assert!(child.schema.has_promql_series_identity()); + assert!(matches!(&cols[1].expr, ScalarExpr::Negative { .. })); } // #36 — negation nested inside an aggregate argument (issue #27 nesting): // `sum(-m)` → Aggregate{Sum} over the `m * -1` BinaryOp. #[test] fn q36_sum_of_negation_nests() { - let expected = QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec!["".into()], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("node_cpu_seconds_total", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), - vector_match: None, - }), + let root = lower("sum(-node_cpu_seconds_total)"); + let NonASAPOp::Aggregate { + child, measures, .. + } = root.expect_non_asap() + else { + panic!() }; - assert_eq!(lower("sum(-node_cpu_seconds_total)"), expected); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); } diff --git a/crates/integration-tests/tests/cse.rs b/crates/integration-tests/tests/cse.rs index bb11eee2d..56a0e657b 100644 --- a/crates/integration-tests/tests/cse.rs +++ b/crates/integration-tests/tests/cse.rs @@ -2,14 +2,14 @@ //! #223). //! //! Drives the full staged pipeline this issue lands: two independently -//! lowered `QueryExpr` DAGs → `share_common_sub_dags` (stage 1, -//! `asap-types::pre_asap::cse`, run internally by `search_workload`) → +//! lowered `OperatorNode` DAGs → `share_common_sub_dags` (stage 1, +//! `asap-types::ir::cse`, run internally by `search_workload`) → //! `search_workload` (stage 2, `asap-aware-mapping`) — and asserts the //! sharing that stage 1 decides survives into stage 2's discovered //! `CandidateLogicalASAPDAGs` as one genuinely shared `TargetSubDAGCandidates`, not just one shared -//! `Rc`. This is the "real caller" the issue's landing plan +//! `Rc`. This is the "real caller" the issue's landing plan //! requires before `share_common_sub_dags` is allowed to exist at all (its -//! predecessor, `asap-plan::cse::dedupe_subtrees`, was deleted in #192 for +//! predecessor, `asap-plan::cse::dedupe_sub-DAGs`, was deleted in #192 for //! being unwired dead code). //! //! Committing to one final, physically-materialized answer for a whole @@ -24,14 +24,14 @@ use std::rc::Rc; -use asap_aware_mapping::{search_workload, Replacement}; +use asap_aware_mapping::{is_logical_rewrite, search_workload, Replacement}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::NonASAPOp; use asap_types::types::AccuracyTarget; /// Two workload entries that happen to submit the exact same query (a /// realistic case — two dashboards, or a query fired both standalone and as -/// part of a larger batch) collapse onto one shared `Rc` after +/// part of a larger batch) collapse onto one shared `Rc` after /// `search_workload`'s internal `share_common_sub_dags` pass, and onto one /// genuinely-shared [`TargetSubDAGCandidates`](asap_aware_mapping::TargetSubDAGCandidates) — carrying /// every candidate discovered for it exactly once, not once per root — no @@ -41,7 +41,7 @@ use asap_types::types::AccuracyTarget; fn duplicate_workload_queries_collapse_onto_one_memo_group() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema carries // a provable unique key — the legality gate `share_common_sub_dags` - // enforces (see `asap-types::pre_asap::cse`'s module doc) — and its + // enforces (see `asap-types::ir::cse`'s module doc) — and its // `ExactAggregate(Sum)` realization is deterministic regardless of the // accuracy target, so this pins the sharing mechanism itself rather than // any one particular summary-family choice. @@ -57,17 +57,17 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { "fixture sanity: identical query text lowers identically" ); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); // roots[0] and roots[1] must have merged onto the same Rc — the // `share_common_sub_dags` pass `search_workload` runs internally. assert!( Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1), - "search_workload must collapse the two identical roots onto one Rc" + "search_workload must collapse the two identical roots onto one Rc" ); // The single shared root is one discovered TargetSubDAG, holding one - // TargetSubDAGCandidates with consumer_count 2 — SketchAlgorithmStrategy's one + // TargetSubDAGCandidates with consumer_count 2 — ASAPStrategies's one // ExactAggregate candidate *and* SharedSubDAGStrategy's share-vs- // recompute pair, exactly as `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // (asap-aware-mapping::replacement's own equivalent, internal test) @@ -79,19 +79,21 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { assert_eq!( group.candidates.len(), 3, - "1 ExactAggregate Summary + 2 Rewrite (share/recompute): {:?}", + "1 ExactAggregate summary + 2 logical rewrites (share/recompute): {:?}", group.candidates ); + // A bound summary is a `Subtree` with an ASAP operator in it; a logical + // rewrite is a `Subtree` with none (`is_logical_rewrite`). let summary_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Summary(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap())) .count(); let rewrite_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if is_logical_rewrite(n))) .count(); assert_eq!(summary_count, 1); assert_eq!(rewrite_count, 2); @@ -100,12 +102,14 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // "false-positive dedup" failure mode `is_duplicate_rewrite` exists to // prevent): one shares the group's own target `Rc`, the other is a // structurally-identical but independently-built `Rc`. - let one_is_the_target = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target)), - ); - let one_is_not = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if !Rc::ptr_eq(rc, &group.target)), - ); + let one_is_the_target = group.candidates.iter().any(|c| { + matches!(&c.replacement, Replacement::SubDAG(rc) + if is_logical_rewrite(rc) && Rc::ptr_eq(rc, &group.target)) + }); + let one_is_not = group.candidates.iter().any(|c| { + matches!(&c.replacement, Replacement::SubDAG(rc) + if is_logical_rewrite(rc) && !Rc::ptr_eq(rc, &group.target)) + }); assert!(one_is_the_target && one_is_not); } @@ -121,7 +125,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { .expect("query b failed to lower"); assert_ne!(a, b, "fixture sanity: the two queries differ"); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); assert!(!Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let group_a = space @@ -140,7 +144,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { /// Single-query CSE (a repeated sub-expression within one query) also /// survives through `search_workload`: the two grouped-`Aggregate` branches -/// of a `BinaryOp` collapse to one shared `Rc` in the internal +/// of a `BinaryOp` collapse to one shared `Rc` in the internal /// `share_common_sub_dags` pass, and to one shared `TargetSubDAGCandidates` (with /// `consumer_count == 2`, one per branch) here. #[test] @@ -148,16 +152,16 @@ fn single_query_repeated_subexpression_shares_one_memo_group() { let query = "sum by (job) (http_requests_total) / sum by (job) (http_requests_total)"; let expr = lower_promql(query, AccuracyTarget::Exact).expect("query failed to lower"); - let space = search_workload(vec![("q", Rc::new(expr))]); + let space = search_workload(vec![("q", expr)]); let [(_, root)] = space.roots.as_slice() else { panic!("expected 1 root"); }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = root.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { panic!("expected a BinaryOp root, got {root:?}"); }; assert!( Rc::ptr_eq(lhs, rhs), - "the two identical sum-by-job branches must collapse onto one Rc" + "the two identical sum-by-job branches must collapse onto one Rc" ); let group = space diff --git a/crates/integration-tests/tests/exact_composition.rs b/crates/integration-tests/tests/exact_composition.rs index 917ef7b3a..3193026ab 100644 --- a/crates/integration-tests/tests/exact_composition.rs +++ b/crates/integration-tests/tests/exact_composition.rs @@ -1,5 +1,5 @@ //! Issue #171 — composing exact operators with summary plans across -//! explicit update/readout boundaries, end to end through +//! explicit update/evaluation boundaries, end to end through //! `search_workload_with` → `CandidateLogicalASAPDAGs::global_selection` → //! `GlobalSelection::assemble_selected_dag` → `dag_export`. //! @@ -16,27 +16,37 @@ use asap_aware_mapping::cost_model::{ CostProvenance, CostUnit, ExactCompositionCostInputs, ExactCompositionCostRequest, ValueOperationCapabilities, }; +use asap_aware_mapping::exact_composition::ExactOperation; use asap_aware_mapping::replacement::{ - default_strategies_with, search_workload_with, Replacement, ReplacementProvenance, - ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, + default_strategies_with, search_workload_with, ASAPStrategies, Replacement, + ReplacementProvenance, ReplacementStrategy, TargetSubDAG, }; use asap_aware_mapping::{ CostModel, DefaultCostModel, EvaluationRate, ExplanationKind, OperationPlacement, }; use asap_integration_tests::fixtures::lower_promql; +use asap_integration_tests::post_asap::{post_asap_dag, timed}; use asap_types::dag_export; +use asap_types::ir::operator_properties::{Reduction, Source}; +use asap_types::ir::physical_export::PhysicalASAPOperatorPayload; +use asap_types::ir::timing::data_state; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, TimeRangeKind}; use asap_types::post_asap::{ - validate_execution_data_states, ExactKind, ExactOperation, ExecutionDataState, ExecutionTiming, - FieldDataType, SketchAlgorithm, SummaryExpr, SummaryNode, SummaryUpdate, + ExactKind, ExecutionDataState, ExecutionTiming, FieldDataType, SketchAlgorithm, SummaryUpdate, }; use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; + use asap_types::types::AccuracyTarget; // ── fixtures ──────────────────────────────────────────────────────────── -fn metric_scan(labels: &[&str]) -> QueryExpr { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn metric_scan(labels: &[&str]) -> Rc { let mut columns = vec![ Field::plain("ts", DataType::Timestamp, false), Field::plain("value", DataType::Float64, false), @@ -46,17 +56,17 @@ fn metric_scan(labels: &[&str]) -> QueryExpr { .iter() .map(|n| Field::plain(*n, DataType::Utf8, true)), ); - QueryExpr::Scan { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "latency".into(), }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), - } + }) } -fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -66,8 +76,8 @@ fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc }) } -fn per_entity(intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn per_entity(intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec![], @@ -78,11 +88,11 @@ fn per_entity(intent: AggIntent, child: Rc) -> Rc { } /// `quantile by (zone, host) (latency)` — the fine-grained inner summary. -fn fine_quantile() -> Rc { +fn fine_quantile() -> Rc { agg( vec![2, 3], default_quantile(0.99), - Rc::new(metric_scan(&["zone", "host"])), + metric_scan(&["zone", "host"]), ) } @@ -96,7 +106,7 @@ struct StatsModel; fn custom_accuracy_rule_survives_root_target_and_materialization() { use asap_aware_mapping::{AccuracyModel, DefaultAccuracyModel, PropagationStats}; use asap_types::post_asap::{ - AccuracyError, CompositionOperator, ExactOperation, ResultGuarantee, SketchStatistic, + AccuracyError, CompositionOperator, ResultGuarantee, SketchStatistic, }; struct Model; impl AccuracyModel for Model { @@ -278,34 +288,44 @@ fn unknown_runtime_capability_keeps_candidate_but_prevents_selection() { } fn plan( - roots: Vec<(&'static str, Rc)>, + roots: Vec<(&'static str, Rc)>, cost_model: &dyn CostModel, ) -> asap_aware_mapping::CandidateLogicalASAPDAGs<&'static str> { search_workload_with(roots, &default_strategies_with(cost_model)) } -fn is_plain(node: &SummaryNode) -> bool { +fn is_plain(node: &OperatorNode) -> bool { node.schema .fields .iter() .all(|f| matches!(f.dtype, FieldDataType::Plain(_))) } -fn names(node: &SummaryNode) -> Vec<&str> { +fn names(node: &OperatorNode) -> Vec<&str> { node.schema.fields.iter().map(|f| f.name.as_str()).collect() } +/// The composed query-time shape: an exact `Aggregate` directly over a +/// summary evaluation, at query time. +fn is_query_time_fold(node: &OperatorNode) -> bool { + matches!( + node.non_asap(), + Some(NonASAPOp::Aggregate { child, .. }) + if matches!(child.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) + ) +} + // ── step 1: pin every already-supported exact-accumulator nesting ─────── #[test] fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { use std::time::Duration; - let cases: Vec<(Rc, ExactKind)> = vec![ + let cases: Vec<(Rc, ExactKind)> = vec![ ( agg( vec![2], AggIntent::Sum { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Sum, ), @@ -315,7 +335,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { AggIntent::Count { accuracy: AccuracyTarget::Exact, }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Count, ), @@ -323,7 +343,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { agg( vec![2], AggIntent::Min { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Min, ), @@ -331,16 +351,17 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { agg( vec![2], AggIntent::Max { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Max, ), ( per_entity( AggIntent::Rate, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ), ExactKind::Rate, @@ -348,9 +369,10 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { ( per_entity( AggIntent::Increase, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ), ExactKind::Increase, @@ -359,40 +381,43 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { for (inner, kind) in cases { let outer = agg(vec![], default_quantile(0.9), inner); let target = TargetSubDAG::new(&outer); - let candidates = SketchAlgorithmStrategy::default_cost_model().replacements(&target); - let Replacement::Summary(root) = &candidates[0].replacement else { + let candidates = ASAPStrategies::default_cost_model().replacements(&target); + let Replacement::SubDAG(root) = &candidates[0].replacement else { unreachable!() }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected KLL readout, got {:?}", root.expr); + // Timing is not stored on the plan: time it (default lifecycle, + // which also validates every edge) and inspect the timed copy. + let root = timed(root); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected KLL evaluation, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("expected outer SummaryAgg"); }; - let SummaryExpr::ValueOperation { - child, - operation: asap_types::post_asap::ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } = &child.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: finalized }) = &child.operator else { panic!("{kind:?}: missing maintenance finalization"); }; + assert_eq!( + child.timing, + Some(ExecutionTiming::IngestionTime), + "{kind:?}: finalization runs at maintenance time" + ); assert!( matches!( - &child.expr, - SummaryExpr::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. } if *k == kind + &finalized.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. }) if *k == kind ), "{kind:?}: expected the exact accumulator under its finalization, got {:?}", - child.expr + finalized.operator ); - validate_execution_data_states(root).expect("accumulator state composes under maintenance"); } } -// ── direction 1: outer exact fold over an inner summary readout ──────── +// ── direction 1: outer exact fold over an inner summary evaluation ──────── -/// Before this PR both `max`/`avg` over a quantile collapsed into one -/// opaque `KeepPreAsap`. Now: the outer group holds an `ValueOperationAtQueryTime` +/// `max`/`avg` over a quantile does not collapse into one opaque kept +/// sub-DAG: the outer group holds an `ValueOperationAtQueryTime` /// candidate referencing the inner target, the inner group keeps its own /// sketch candidates, and with statistics the pair is committed and /// materializes as `ValueOperationAtQueryTime → SummaryEstimate → SummaryAgg`. @@ -402,7 +427,7 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { let root = agg(vec![0], intent.clone(), fine_quantile()); let space = plan(vec![("q", Rc::clone(&root))], &StatsModel); let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: inner, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child: inner, .. }) = root.non_asap() else { unreachable!() }; @@ -419,9 +444,9 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { inner_group .candidates .iter() - .any(|c| matches!(&c.replacement, Replacement::Summary(n) - if matches!(n.expr, SummaryExpr::SummaryEstimate { .. }))), - "{intent:?}: the inner quantile keeps its own readout candidates" + .any(|c| matches!(&c.replacement, Replacement::SubDAG(n) + if matches!(n.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })))), + "{intent:?}: the inner quantile keeps its own evaluation candidates" ); let selection = space.global_selection(&StatsModel); @@ -449,18 +474,16 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { )); let composed = selection.assemble_selected_dag(&root).unwrap().unwrap(); - let SummaryExpr::ValueOperation { - child, - timing: ExecutionTiming::QueryTime, - .. - } = &composed.expr - else { + let Some(NonASAPOp::Aggregate { child, .. }) = composed.non_asap() else { panic!( "{intent:?}: expected ValueOperationAtQueryTime root, got {:?}", - composed.expr + composed.operator ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); assert!( child.guarantee.is_some(), "child has its KLL rank guarantee" @@ -472,15 +495,18 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { assert!(is_plain(&composed)); assert_eq!( names(&composed), - root.output_schema() - .unwrap() + root.schema .fields .iter() .map(|c| c.name.as_str()) .collect::>(), "the composed plan's schema is the pre-ASAP target's own" ); - validate_execution_data_states(&composed).unwrap(); + assert_eq!( + timed(&composed).timing, + Some(ExecutionTiming::QueryTime), + "{intent:?}: the exact fold runs at query time" + ); } } @@ -490,11 +516,7 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { fn avg_over_quantile_keeps_the_sum_over_count_rewrite_as_a_competitor() { // `by (zone)` over `by (zone)`: the averaged column resolves to the // non-null quantile output, which is what the rewrite requires. - let inner = agg( - vec![2], - default_quantile(0.99), - Rc::new(metric_scan(&["zone"])), - ); + let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); let root = agg(vec![0], AggIntent::Avg { col: None }, inner); let space = plan(vec![("q", root)], &StatsModel); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); @@ -508,11 +530,7 @@ fn avg_over_quantile_keeps_the_sum_over_count_rewrite_as_a_competitor() { /// is the same, only the fold's row multiplicity differs. #[test] fn identity_and_genuine_multi_row_folds_both_compose() { - let identity_inner = agg( - vec![2], - default_quantile(0.99), - Rc::new(metric_scan(&["zone"])), - ); + let identity_inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); for (label, inner) in [ ("identity", identity_inner), ("fine-to-coarse", fine_quantile()), @@ -527,14 +545,17 @@ fn identity_and_genuine_multi_row_folds_both_compose() { .unwrap(); assert!( matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } + composed.non_asap(), + Some(NonASAPOp::Aggregate { child, .. }) + if matches!(child.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) ), "{label}: {:?}", - composed.expr + composed.operator + ); + assert_eq!( + timed(&composed).timing, + Some(ExecutionTiming::QueryTime), + "{label}" ); assert_eq!(names(&composed), vec!["zone", "max"], "{label}"); } @@ -543,7 +564,7 @@ fn identity_and_genuine_multi_row_folds_both_compose() { /// One inner quantile consumed by two outer folds in two queries: CSE /// collapses the inner target onto one `Rc`, both compositions commit to /// the *same* child candidate, and both materializations share one -/// `Rc` for it — the summary is maintained once. +/// `Rc` for it — the summary is maintained once. #[test] fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let max = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); @@ -551,9 +572,9 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let space = plan(vec![("max", max), ("min", min)], &StatsModel); let selection = space.global_selection(&StatsModel); - let roots: Vec> = space.roots.iter().map(|(_, r)| Rc::clone(r)).collect(); - let inner_of = |r: &Rc| match r.as_ref() { - QueryExpr::Aggregate { child, .. } => Rc::clone(child), + let roots: Vec> = space.roots.iter().map(|(_, r)| Rc::clone(r)).collect(); + let inner_of = |r: &Rc| match r.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) => Rc::clone(child), _ => unreachable!(), }; assert!( @@ -589,17 +610,20 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { .iter() .map(|r| selection.assemble_selected_dag(r).unwrap().unwrap()) .collect(); - let child_of = |n: &Rc| match &n.expr { - SummaryExpr::ValueOperation { - child, - timing: ExecutionTiming::QueryTime, - .. - } => Rc::clone(child), - other => panic!("expected ValueOperationAtQueryTime, got {other:?}"), + let child_of = |n: &Rc| match n.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) + if matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ) => + { + Rc::clone(child) + } + _ => panic!("expected ValueOperationAtQueryTime, got {:?}", n.operator), }; assert!( Rc::ptr_eq(&child_of(&composed[0]), &child_of(&composed[1])), - "both folds compose over the same Rc" + "both folds compose over the same Rc" ); } @@ -614,15 +638,16 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { use std::time::Duration; let deriv = per_entity( AggIntent::Deriv, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ); let root = agg(vec![], default_quantile(0.99), deriv); let space = plan(vec![("q", root)], &StatsModel); let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: deriv, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child: deriv, .. }) = root.non_asap() else { unreachable!() }; assert!(space @@ -643,33 +668,25 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { assert!(decision.cost_rate < decision.baseline_rate); let composed = selection.assemble_selected_dag(&root).unwrap().unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &composed.expr else { - panic!("expected readout root, got {:?}", composed.expr); + // Walk the timed copy: timing is written by the lifecycle assignment. + let composed = timed(&composed); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &composed.operator else { + panic!("expected evaluation root, got {:?}", composed.operator); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("expected SummaryAgg"); }; - let SummaryExpr::ValueOperation { - child: raw, - timing: ExecutionTiming::IngestionTime, - .. - } = &child.expr - else { + let Some(NonASAPOp::Aggregate { child: raw, .. }) = child.non_asap() else { panic!( "expected ValueOperationAtIngestionTime under the maintained summary, got {:?}", - child.expr + child.operator ); }; - assert!(matches!(raw.expr, SummaryExpr::KeepPreAsap(_))); - let assignment = validate_execution_data_states(&composed).unwrap(); - assert_eq!( - assignment.data_state_of(child), - Some(ExecutionDataState::INGESTION_ROWS) - ); - assert_eq!( - assignment.data_state_of(raw), - Some(ExecutionDataState::INGESTION_ROWS) - ); + // The raw input is kept as-is. + assert!(matches!(raw.non_asap(), Some(NonASAPOp::TimeRange { .. }))); + assert!(!raw.contains_asap()); + assert_eq!(data_state(child), Some(ExecutionDataState::INGESTION_ROWS)); + assert_eq!(data_state(raw), Some(ExecutionDataState::INGESTION_ROWS)); } // ── rejection, capability, statistics ─────────────────────────────────── @@ -685,24 +702,28 @@ fn summary_construction_follows_its_value_input_phase() { .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let illegal = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: post, - family: FieldDataType::ExactAggregate( - ExactKind::Max, - asap_types::post_asap::ExactParams::Max, - ), - input: SummaryUpdate::column(asap_types::pre_asap::ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: Default::default(), - filter: None, - }, - schema: asap_types::post_asap::Schema::lifted(vec![], None), - guarantee: None, - }); - let state = asap_types::post_asap::produced_data_state(&illegal.expr).unwrap(); + let illegal = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: post, + family: FieldDataType::ExactAggregate( + ExactKind::Max, + asap_types::post_asap::ExactParams::Max, + ), + input: SummaryUpdate::column(asap_types::pre_asap::ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: Default::default(), + filter: None, + }), + Schema::lifted(vec![], None), + ) + .with_guarantee(None), + ); + // Even under an ingestion-time consumer the state is built at query + // time, because a evaluation sits below it. + let state = asap_types::ir::planned_data_state(&illegal, ExecutionTiming::IngestionTime); assert_eq!(state.timing, ExecutionTiming::QueryTime); - asap_types::post_asap::validate_execution_data_states_at(&illegal, state).unwrap(); + asap_types::ir::validate_default(&illegal, state.timing).unwrap(); } #[test] @@ -718,9 +739,9 @@ fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { let selection = space.global_selection(&NoCapabilityModel); assert!(selection.for_target(&root).unwrap().composition.is_none()); let node = selection.assemble_selected_dag(&root).unwrap().unwrap(); - assert!(!matches!(node.expr, SummaryExpr::ValueOperation { .. })); + assert!(!is_query_time_fold(&node)); // The inner quantile is still independently selectable. - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = root.non_asap() else { unreachable!() }; assert!(selection.for_target(child).unwrap().chosen.is_some()); @@ -731,7 +752,7 @@ fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { /// site keeps a non-composed alternative, and the inner summary stays /// independently selectable. #[test] -fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { +fn missing_cost_statistics_preserve_the_conservative_retain_exact() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); let space = plan(vec![("q", root)], &DefaultCostModel); let root = Rc::clone(&space.roots[0].1); @@ -749,9 +770,9 @@ fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { Some(Replacement::ExactComposition(_)) )); let node = selection.assemble_selected_dag(&root).unwrap().unwrap(); - assert!(!matches!(node.expr, SummaryExpr::ValueOperation { .. })); + assert!(!is_query_time_fold(&node)); - let explanations = asap_aware_mapping::explain_replacements(vec![("q", (*root).clone())]); + let explanations = asap_aware_mapping::explain_replacements(vec![("q", Rc::clone(&root))]); assert!(explanations .iter() .any(|e| e.kind == ExplanationKind::ExactComposition)); @@ -771,12 +792,17 @@ fn dag_export_carries_explicit_stage_and_plain_schema_for_a_composed_plan() { .unwrap(); let dag = dag_export::export_summary(&composed); let node = &dag.nodes[dag.root as usize]; - assert_eq!(node.kind, "ValueOperation"); - assert_eq!(node.detail["timing"], "query_time"); - assert!(node.detail["operation"] - .as_str() - .unwrap() - .starts_with("Exact(Aggregate")); + assert_eq!(node.kind, "aggregate"); + assert!(node.detail["measures"].is_array()); + // Timing is explicit in the wire-6 DAG: the root is a relational + // aggregate placed at query time. + let wire = post_asap_dag(&composed); + let wire_root = wire.nodes.iter().find(|n| n.id == wire.roots[0]).unwrap(); + assert!(matches!( + wire_root.payload, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Aggregate { .. }) + )); + assert_eq!(wire_root.output_state.timing, ExecutionTiming::QueryTime); // Pre-ASAP export of the same target still describes the same columns. let pre = dag_export::export(root); @@ -798,7 +824,7 @@ fn promql_max_by_zone_over_quantile_over_time_composes() { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let space = plan(vec![("q", Rc::new(expr))], &StatsModel); + let space = plan(vec![("q", expr)], &StatsModel); let root = &space.roots[0].1; let selection = space.global_selection(&StatsModel); let selected = selection.for_target(root).unwrap(); @@ -815,13 +841,8 @@ fn promql_max_by_zone_over_quantile_over_time_composes() { .collect::>() ); let composed = selection.assemble_selected_dag(root).unwrap().unwrap(); - assert!(matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } - )); + assert!(is_query_time_fold(&composed), "{:?}", composed.operator); + assert_eq!(timed(&composed).timing, Some(ExecutionTiming::QueryTime)); assert_eq!( selected.composition.as_ref().map(|d| d.inputs.unit), Some(CostUnit::CostUnitsPerSecond) diff --git a/crates/integration-tests/tests/frontend_timestamps.rs b/crates/integration-tests/tests/frontend_timestamps.rs index 388261da6..300f0cee7 100644 --- a/crates/integration-tests/tests/frontend_timestamps.rs +++ b/crates/integration-tests/tests/frontend_timestamps.rs @@ -1,9 +1,9 @@ //! Cross-frontend evaluation-time semantics (issues #46 and #184). use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_integration_tests::fixtures::lower_promql; +use asap_integration_tests::fixtures::lower_promql_root; +use asap_types::ir::{NonASAPOp, ScalarExpr}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::QueryExpr; use asap_types::types::AccuracyTarget; /// PromQL exposes its evaluation time as Unix seconds, whereas SQL exposes @@ -11,10 +11,21 @@ use asap_types::types::AccuracyTarget; /// but must remain distinguishable in the shared IR and type inference. #[tokio::test] async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { - let promql = lower_promql("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); - assert!(matches!(promql, QueryExpr::EvalTimestamp)); - let promql_schema = promql.output_schema().expect("PromQL time() schema"); - assert_eq!(promql_schema.fields[0].dtype, DataType::Float64); + let promql = lower_promql_root("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); + assert!( + matches!( + promql, + asap_types::ir::QueryRoot::Scalar(ScalarExpr::EvalTimestamp) + ), + "expected a bare evaluation-time scalar, got {promql:?}" + ); + assert_eq!( + ScalarExpr::EvalTimestamp + .scalar_type(&Schema::default()) + .unwrap() + .0, + DataType::Float64 + ); let catalog = SqlCatalog::new().with_table( "metrics", @@ -27,13 +38,14 @@ async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { ) .await .expect("lower SQL CURRENT_TIMESTAMP"); - let QueryExpr::Project { cols, .. } = sql else { + let Some(NonASAPOp::Project { cols, child, .. }) = sql.non_asap() else { panic!("expected SQL projection, got {sql:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); - let sql_schema = cols[0] + assert!(matches!(&cols[0].expr, ScalarExpr::CurrentTimestamp)); + let (sql_dtype, _) = cols[0] .expr - .output_schema() - .expect("SQL CURRENT_TIMESTAMP schema"); - assert_eq!(sql_schema.fields[0].dtype, DataType::Timestamp); + .scalar_type(&child.schema) + .expect("SQL CURRENT_TIMESTAMP type"); + assert_eq!(sql_dtype, DataType::Timestamp); + assert_eq!(sql.schema.fields[0].dtype, DataType::Timestamp); } diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index b48d14bf7..4869d681b 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -1,7 +1,7 @@ //! Maintenance -> stored pane state -> independently bound query execution. mod physical_common; use asap_physical_operators::{ - operators::{Operator, ReadoutQuery}, + operators::{Operator, SummaryEvaluation}, physical_planner::{CompiledPhysicalDAG, InputContract, Source}, plan::{PhysicalDAG, PhysicalOperator, PlanProperties}, runtime::{Input, Limits, OutputStream, RunContext, Scope}, @@ -32,8 +32,8 @@ fn family(k: u32) -> FieldDataType { } fn raw_schema() -> SchemaRef { Arc::new(Schema { - closed: true, unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -98,11 +98,11 @@ fn restore(schema: SchemaRef, states: &[Arc]) -> Batch { ) .unwrap() } -fn readout(schema: SchemaRef, q: f64) -> Operator { - Operator::readout( +fn evaluation(schema: SchemaRef, q: f64) -> Operator { + Operator::evaluation( schema, 0, - ReadoutQuery::Sketch(SketchStatistic::Quantile { q }), + SummaryEvaluation::Sketch(SketchStatistic::Quantile { q }), ) .unwrap() } @@ -159,8 +159,8 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { ), ), (6, (vec![5], merge.clone())), - (7, (vec![6], readout(schema.clone(), 0.5))), - (8, (vec![6], readout(schema.clone(), 0.99))), + (7, (vec![6], evaluation(schema.clone(), 0.5))), + (8, (vec![6], evaluation(schema.clone(), 0.99))), ]), vec![6, 7, 8], ) @@ -247,8 +247,10 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { }, ) .unwrap(); - dag.add(2, vec![1], readout(schema.clone(), 0.5)).unwrap(); - dag.add(3, vec![1], readout(schema.clone(), 0.99)).unwrap(); + dag.add(2, vec![1], evaluation(schema.clone(), 0.5)) + .unwrap(); + dag.add(3, vec![1], evaluation(schema.clone(), 0.99)) + .unwrap(); let outputs = block_on(futures::future::join_all( dag.execute( &[2, 3], diff --git a/crates/integration-tests/tests/nested.rs b/crates/integration-tests/tests/nested.rs index 1e3e34ebd..26ac23a24 100644 --- a/crates/integration-tests/tests/nested.rs +++ b/crates/integration-tests/tests/nested.rs @@ -1,8 +1,8 @@ //! Multi-node pipeline tests — nested `Aggregate`, `TimeRange`, `BinaryOp`, and `Scan`. //! //! Key invariant: `rate`/`increase` are label-preserving (per-series), so an -//! outer `Aggregate.by` resolves its group keys against the inner aggregate's -//! output schema, which still carries all label columns. +//! outer `Aggregate` reduction resolves its group keys against the inner +//! aggregate's output schema, which still carries all label columns. //! //! Label column ordering is always alphabetical, so in a query that references //! both `job` and `status`: @@ -13,56 +13,107 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, TimeRangeKind, +}; use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, GroupKeys, Predicate, - PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, TimeShift, VectorMatch, - VectorMatchKind, + AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, GroupKeys, + PromQLVectorSetOpKind, Reduction, ScalarValue, Source, TimeShift, VectorMatch, VectorMatchKind, }; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str, predicates: Vec, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { + metric: metric.into(), + }, + predicates, + schema: metric_schema(labels), + }) +} + +fn instant(child: Rc) -> Rc { + node(NonASAPOp::TimeRange { + range: Duration::from_secs(1), + kind: TimeRangeKind::Instant, + child, + }) +} + +fn range(secs: u64, child: Rc) -> Rc { + node(NonASAPOp::TimeRange { + range: Duration::from_secs(secs), + kind: TimeRangeKind::Range, + child, + }) +} + +fn eq_pred(col_id: usize, value: &str) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col_id)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Literal(ScalarValue::Utf8(value.into()))), + semantics: ExprSemantics::Promql, + }) +} + +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(child), - } + child, + }) } -fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg_per_entity(intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(child), - } + child, + }) +} + +fn binary( + kind: BinaryOpKind, + vector_match: Option, + lhs: Rc, + rhs: Rc, +) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + }, + return_bool: false, + lhs, + rhs, + }) } // #22 — sum by job over rate; outer by=[2] resolves against rate's // label-preserving output schema [ts, value, job] #[test] fn q22_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["job"])), ); let expected = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); assert_eq!( @@ -76,25 +127,12 @@ fn q22_sum_by_job_over_rate() { // predicate on status (col 3); group key job (col 2) #[test] fn q23_sum_by_job_over_filtered_scan() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; - let expected = agg( - vec![2], - AggIntent::Sum { col: None }, - QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], ); + let expected = agg(vec![2], AggIntent::Sum { col: None }, instant(scan)); assert_eq!( lower(r#"sum by (job) (http_requests_total{status="200"})"#), expected @@ -108,54 +146,30 @@ fn q23_sum_by_job_over_filtered_scan() { // schema [ts, value, job]; outer by=[2] (job) #[test] fn q25_div_over_complex_sub_dags() { - let lhs_scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; + let lhs_scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], + ); let lhs = agg( vec![2], AggIntent::Sum { col: None }, - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(lhs_scan), - }, - ), + agg_per_entity(AggIntent::Rate, range(300, lhs_scan)), ); - let rhs_scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_errors_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; + let rhs_scan = scan("http_errors_total", vec![], &["job"]); let rhs = agg( vec![2], AggIntent::Sum { col: None }, - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(rhs_scan), - }, - ), + agg_per_entity(AggIntent::Rate, range(300, rhs_scan)), ); - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(lhs), - rhs: Rc::new(rhs), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + lhs, + rhs, + ); assert_eq!( lower( r#"sum by (job) (rate(http_requests_total{status="200"}[5m])) / sum by (job) (rate(http_errors_total[5m]))"# @@ -171,19 +185,9 @@ fn q25_div_over_complex_sub_dags() { // label-preserving output schema; the outer `max` has no grouping. #[test] fn q27_max_over_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["job"])), ); let sum_by_job = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); let expected = agg(vec![], AggIntent::Max { col: None }, sum_by_job); @@ -201,25 +205,12 @@ fn q27_max_over_sum_by_job_over_rate() { // Scan schema: [ts(0), value(1), group(2), job(3)] (labels alphabetical). #[test] fn q53_outer_group_key_absent_from_nested_aggregate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api-server".into()))), - }))], - schema: metric_schema(&["group", "job"]), - }; - let inner = agg( - vec![2], - AggIntent::Sum { col: None }, - QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests", + vec![eq_pred(3, "api-server")], + &["group", "job"], ); + let inner = agg(vec![2], AggIntent::Sum { col: None }, instant(scan)); let expected = agg(vec![], AggIntent::Sum { col: None }, inner); assert_eq!( lower(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#), @@ -236,33 +227,26 @@ fn q53_outer_group_key_absent_from_nested_aggregate() { // parser's default `ignoring([])` match modifier. #[test] fn q52_outer_name_label_over_binary_op() { - let side = |metric: &str, env: &str| QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), // env - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(env.into()))), - }))], - schema: metric_schema(&["env", "__name__"]), - }), + let side = |metric: &str, env: &str| { + instant(scan( + metric, + vec![eq_pred(2, env)], // env + &["env", "__name__"], + )) }; let expected = agg( vec![3], // __name__ AggIntent::Sum { col: None }, - QueryExpr::BinaryOp { - op: BinaryOpKind::Set(PromQLVectorSetOpKind::Or), - lhs: Rc::new(side("metric_a", "1")), - rhs: Rc::new(side("metric_b", "2")), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Set(PromQLVectorSetOpKind::Or), + Some(VectorMatch { kind: VectorMatchKind::Ignoring, labels: vec![], grouping: None, }), - }, + side("metric_a", "1"), + side("metric_b", "2"), + ), ); assert_eq!( lower(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#), @@ -276,28 +260,18 @@ fn q52_outer_name_label_over_binary_op() { // the inner rate is label-preserving. Scan schema [ts(0), value(1), instance(2)]. #[test] fn q39_sum_without_instance_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["instance"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["instance"])), ); - let expected = QueryExpr::Aggregate { + let expected = node(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(vec![2])), // exclude `instance` measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(inner_rate), - }; + child: inner_rate, + }); assert_eq!( lower("sum without (instance) (rate(http_requests_total[5m]))"), expected, @@ -310,35 +284,25 @@ fn q39_sum_without_instance_over_rate() { #[test] fn q40_week_over_week_offset() { let rate_over = |shift: Option| { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: metric_schema(&[]), - }; + let scan = scan("m", vec![], &[]); let ranged = match shift { - Some(ms) => QueryExpr::TimeShift { + Some(ms) => node(NonASAPOp::TimeShift { shift: TimeShift { offset_ms: ms, at: None, }, - child: Rc::new(scan), - }, + child: scan, + }), None => scan, }; - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(ranged), - }, - ) - }; - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), - lhs: Rc::new(rate_over(None)), - rhs: Rc::new(rate_over(Some(604_800_000))), // 1w - vector_match: None, + agg_per_entity(AggIntent::Rate, range(300, ranged)) }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + None, + rate_over(None), + rate_over(Some(604_800_000)), // 1w + ); assert_eq!(lower("rate(m[5m]) - rate(m[5m] offset 1w)"), expected,); } @@ -346,22 +310,13 @@ fn q40_week_over_week_offset() { // (seconds → ms); a bare selector wrapped in a `TimeShift` carrying the anchor. #[test] fn q40_at_modifier_absolute() { - let expected = QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(QueryExpr::TimeShift { - shift: TimeShift { - offset_ms: 0, - at: Some(AtModifier::Timestamp(1_609_746_000_000)), - }, - child: Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "up".into(), - }, - predicates: vec![], - schema: metric_schema(&[]), - }), - }), - }; + let expected = instant(node(NonASAPOp::TimeShift { + shift: TimeShift { + offset_ms: 0, + at: Some(AtModifier::Timestamp(1_609_746_000_000)), + }, + child: scan("up", vec![], &[]), + })); assert_eq!(lower("up @ 1609746000"), expected); } @@ -370,24 +325,12 @@ fn q40_at_modifier_absolute() { // so outer sum by job still finds job at col 2 #[test] fn q24_sum_by_job_over_rate_over_filtered_scan() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; - let inner_rate = agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], ); + let inner_rate = agg_per_entity(AggIntent::Rate, range(300, scan)); let expected = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); assert_eq!( lower(r#"sum by (job) (rate(http_requests_total{status="200"}[5m]))"#), @@ -403,35 +346,25 @@ fn q24_sum_by_job_over_rate_over_filtered_scan() { // the whole spine survives verbatim and the schema stays label-preserving. #[test] fn q27_nested_subquery_prometheus_docs_example() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "distance_covered_total".into(), - }, - predicates: vec![], - schema: metric_schema(&[]), - }; let rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(5), - child: Rc::new(scan), - }, + range(5, scan("distance_covered_total", vec![], &[])), ); let deriv = agg_per_entity( AggIntent::Deriv, - QueryExpr::PromqlSubquery { + node(NonASAPOp::PromqlSubquery { range: Duration::from_secs(30), resolution: Some(Duration::from_secs(5)), - child: Rc::new(rate), - }, + child: rate, + }), ); let expected = agg_per_entity( AggIntent::Max { col: None }, - QueryExpr::PromqlSubquery { + node(NonASAPOp::PromqlSubquery { range: Duration::from_secs(600), resolution: None, - child: Rc::new(deriv), - }, + child: deriv, + }), ); assert_eq!( lower("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"), diff --git a/crates/integration-tests/tests/operator_design_examples.rs b/crates/integration-tests/tests/operator_design_examples.rs new file mode 100644 index 000000000..c5f0c66ee --- /dev/null +++ b/crates/integration-tests/tests/operator_design_examples.rs @@ -0,0 +1,419 @@ +//! #511 examples: source text → unified dag → summary rewrite → flat export. +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; +use asap_types::post_asap::{ + ExactKind, ExactParams, FieldDataType, GroupingStrategy, SummaryUpdate, +}; +use asap_types::pre_asap::{AggIntent, ColumnRef, DataType, Field, Schema}; +use asap_types::types::AccuracyTarget; +use std::rc::Rc; +mod physical_common; + +fn catalog() -> SqlCatalog { + SqlCatalog::new() + .with_table( + "requests", + Schema::new(vec![ + Field::plain("bytes", DataType::Int64, true), + Field::plain("status", DataType::Int64, false), + ]), + ) + .with_table( + "lineitem", + Schema::new(vec![Field::plain("l_quantity", DataType::Int64, false)]), + ) +} + +/// The SQL scalar example keeps column scopes and a Boolean row predicate. +#[tokio::test] +async fn sql_filter_projection_example() { + let root = lower_sql( + "SELECT l_quantity * 2 AS q2 FROM lineitem WHERE l_quantity > 10", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert_eq!( + root.schema.fields[0], + Field::plain("q2", DataType::Int64, false) + ); + assert!( + matches!(root.expect_non_asap(),NonASAPOp::Project { cols,.. } if matches!(cols[0].expr,ScalarExpr::Arithmetic { .. })) + ); + let wire = physical_common::compile_physical_asap_dag(&root).unwrap(); + wire.validate().unwrap(); + assert_eq!(wire.nodes.len(), OperatorNode::reachable(&root).len()); +} + +/// SUM's evaluation preserves integer type and SQL NULL behavior across the rewrite. +#[tokio::test] +async fn sql_sum_projection_before_and_after_summary_rewrite() { + let root = lower_sql( + "SELECT SUM(bytes) + 1 AS total_bytes FROM requests WHERE status = 200", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert_eq!( + root.schema.fields[0], + Field::plain("total_bytes", DataType::Int64, true) + ); + fn rewrite(node: &Rc) -> Rc { + if let Some(NonASAPOp::Aggregate { + child, + reduction, + measures, + .. + }) = node.non_asap() + { + let [AggIntent::Sum { col: Some(column) }] = measures.as_slice() else { + panic!() + }; + let state = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(child), + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + input: SummaryUpdate::column(ColumnRef::Named( + child.schema.fields[*column].name.clone(), + )), + reduction: reduction.clone(), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap(), + ); + let finalize = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: state, + }), + node.schema.clone(), + ) + .with_guarantee(None), + ); + return finalize; + } + Rc::new(node.with_new_children(rewrite).unwrap()) + } + let rewritten = rewrite(&root); + rewritten.validate_structure().unwrap(); + assert_eq!(rewritten.schema, root.schema); + let dag = OperatorNode::reachable(&rewritten); + assert!(dag + .iter() + .any(|n| matches!(n.asap(), Some(ASAPOp::SummaryAgg { .. })))); + assert!(dag + .iter() + .any(|n| matches!(n.asap(), Some(ASAPOp::FinalizeExactAccumulator { .. })))); + let wire = physical_common::compile_physical_asap_dag(&rewritten).unwrap(); + wire.validate().unwrap(); + assert_eq!(wire.nodes.len(), dag.len()); + let json = serde_json::to_string(&wire).unwrap(); + assert!(!json.contains("KeepPreAsap") && !json.contains("ScalarBridge")); +} + +/// Scalar subqueries survive normalization with shared, visible producers. +#[tokio::test] +async fn sql_scalar_subquery_retains_its_cardinality_contract() { + for query in [ + "SELECT (SELECT bytes FROM requests) AS v FROM lineitem", + "SELECT l_quantity NOT IN (SELECT bytes FROM requests) AS present FROM lineitem", + ] { + let root = lower_sql(query, &catalog(), AccuracyTarget::Exact) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert!(root.children().len() > 1); + let wire = physical_common::compile_physical_asap_dag(&root).unwrap(); + assert!(wire + .edges + .iter() + .any(|e| e.role == asap_types::ir::physical_export::EdgeRole::ScalarRef)); + } +} + +/// Execute the SQL SUM example for nonempty, empty and all-NULL populations. +#[tokio::test] +async fn sql_sum_example_executes_with_sql_null_semantics() { + use asap_physical_operators::{ + physical_planner::{compile, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + sources::{DataSources, MemorySource}, + values::{Batch, Value}, + }; + use futures::StreamExt; + use std::{collections::BTreeMap, sync::Arc}; + let root = lower_sql( + "SELECT SUM(bytes) + 1 AS total_bytes FROM requests WHERE status = 200", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + let logical_scan = OperatorNode::reachable(&root) + .into_iter() + .find(|node| matches!(node.non_asap(), Some(NonASAPOp::Scan { .. }))) + .unwrap(); + let NonASAPOp::Scan { source, .. } = logical_scan.expect_non_asap() else { + panic!() + }; + let wire = physical_common::compile_physical_asap_dag(&root).unwrap(); + let scan = wire + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + asap_types::ir::physical_export::PhysicalASAPOperatorPayload::NonASAP( + asap_types::ir::NonASAPOp::Scan { .. } + ) + ) + }) + .unwrap(); + let schema = Arc::new(scan.output_schema.clone()); + let plan = compile( + &wire, + BTreeMap::from([(scan.id as u64, InputContract::bounded(schema.clone()))]), + &[wire.roots[0] as u64], + ) + .unwrap(); + for (rows, expected) in [ + ( + vec![ + vec![Value::Int64(10), Value::Int64(200)], + vec![Value::Int64(20), Value::Int64(500)], + vec![Value::Null, Value::Int64(200)], + ], + Value::Int64(11), + ), + (vec![], Value::Null), + (vec![vec![Value::Null, Value::Int64(200)]], Value::Null), + ] { + let mut sources = DataSources::default(); + sources + .register( + source.clone(), + Arc::new( + MemorySource::new( + schema.clone(), + vec![Batch::try_new(schema.clone(), rows).unwrap()], + ) + .unwrap(), + ), + ) + .unwrap(); + let bound = plan + .instantiate(BTreeMap::from([( + scan.id as u64, + Box::new(sources.bind(&logical_scan).unwrap()) as Source<'_>, + )])) + .unwrap(); + let mut stream = bound + .execute( + plan.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].len(), 1); + match (&rows[0][0], expected) { + (Value::Null, Value::Null) => {} + (Value::Int64(actual), Value::Int64(expected)) => assert_eq!(*actual, expected), + other => panic!("{other:?}"), + } + } +} + +/// Empty window frames and filtered groups can yield NULL even on non-NULL input. +#[tokio::test] +async fn sql_window_and_filtered_aggregate_types() { + for query in [ + "SELECT SUM(l_quantity) OVER (ORDER BY l_quantity ROWS BETWEEN 2 PRECEDING AND 1 PRECEDING) AS s FROM lineitem", + "SELECT MIN(l_quantity) OVER (ORDER BY l_quantity ROWS BETWEEN 2 PRECEDING AND 1 PRECEDING) AS s FROM lineitem", + "SELECT SUM(l_quantity) FILTER (WHERE l_quantity < 0) AS s FROM lineitem GROUP BY l_quantity", + ] { + let root = lower_sql(query, &catalog(), AccuracyTarget::Exact).await.unwrap(); + root.validate_structure().unwrap(); + assert_eq!(root.schema.fields[0], Field::plain("s", DataType::Int64, true), "{query}"); + } +} + +/// A real query batch selects one shared SUM producer, retains two result roots, +/// and executes both selected plans. No replacement dag is constructed by the test. +#[tokio::test] +async fn batch_planning_replaces_and_shares_summary_operators() { + use asap_aware_mapping::cost_model::{Cost, DefaultCostModel}; + use asap_aware_mapping::pass::PlanningModels; + use asap_aware_mapping::{ + CostModel, CostRate, LifecycleInput, SummaryMaintenanceLifecycleCapabilities, + SummaryMaintenanceLifecycleCostInputs, + }; + use asap_physical_operators::{ + physical_planner::{compile, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_planner::{e2e_plan, FrontendInput, UserInput}; + use asap_types::post_asap::SketchAlgorithm; + use asap_types::workload::*; + use std::{collections::BTreeMap, sync::Arc}; + struct Costs; + impl CostModel for Costs { + fn rank_candidates( + &self, + intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + DefaultCostModel.rank_candidates(intent, candidates) + } + fn summary_maintenance_lifecycle_cost_inputs( + &self, + _: &OperatorNode, + ) -> SummaryMaintenanceLifecycleCostInputs { + SummaryMaintenanceLifecycleCostInputs { + build_cost: Some(Cost(1.0)), + maintenance_cost_per_update: Some(Cost::ZERO), + summary_read_cost: Some(Cost::ZERO), + retention_cost_rate: Some(CostRate(0.0)), + retirement_cost: Some(Cost::ZERO), + } + } + fn raw_query_recompute_cost(&self, _: &OperatorNode) -> Option { + Some(Cost(1000.0)) + } + } + let queries = [ + "SELECT SUM(bytes) + 1 AS result FROM requests", + "SELECT SUM(bytes) * 2 AS result FROM requests", + ]; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::SQL(SqlDialect::DataFusionSQL), + query_batch: Some( + queries + .iter() + .map(|query| BatchEntry { + query: Query((*query).into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 2, + execute_at: None, + time_selection: TimeSelection::default(), + }) + .collect(), + ), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + arrival: DataArrival::AtRest, + ..Default::default() + }), + }; + let catalog = SqlCatalog::new().with_table( + "requests", + Schema::new(vec![Field::plain("bytes", DataType::Float64, false)]), + ); + let output = e2e_plan(UserInput::new( + &workload, + FrontendInput::Sql { catalog: &catalog }, + PlanningModels::builtin().with_cost(&Costs), + LifecycleInput::new(0, SummaryMaintenanceLifecycleCapabilities::default()), + )) + .await + .unwrap(); + assert_eq!(output.entry_indices(), [0, 1]); + assert_eq!(output.roots().len(), 2); + let states: Vec<_> = output + .operators() + .into_iter() + .filter(|n| matches!(n.asap(), Some(ASAPOp::SummaryAgg { .. }))) + .collect(); + assert_eq!(states.len(), 1, "the batch owns one shared SUM state"); + for (plan, expected) in output.plans.iter().zip([31.0, 60.0]) { + assert!(!plan.plan.selected_raw_recompute); + let root = &plan.plan.root; + root.validate_structure().unwrap(); + assert!(OperatorNode::reachable(root) + .iter() + .any(|n| Rc::ptr_eq(n, &states[0]))); + let wire = physical_common::compile_physical_asap_dag(root).unwrap(); + let scan = wire + .nodes + .iter() + .find(|n| { + matches!( + n.payload, + asap_types::ir::physical_export::PhysicalASAPOperatorPayload::NonASAP( + asap_types::ir::NonASAPOp::Scan { .. } + ) + ) + }) + .unwrap(); + let schema = Arc::new(scan.output_schema.clone()); + let program = compile( + &wire, + BTreeMap::from([(scan.id as u64, InputContract::bounded(schema.clone()))]), + &[wire.roots[0] as u64], + ) + .unwrap(); + let result = physical_common::execute( + &program, + BTreeMap::from([( + scan.id as u64, + Batch::try_new( + schema, + vec![vec![Value::Float64(10.0)], vec![Value::Float64(20.0)]], + ) + .unwrap(), + )]), + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + ); + let rows: Vec<_> = result[0].iter().flat_map(|batch| batch.rows()).collect(); + assert_eq!(rows.len(), 1); + assert!( + matches!(rows[0][0], Value::Float64(v) if v == expected), + "{:?}", + rows + ); + } + // The batch exports as one physical DAG: a root per query and the shared + // SUM state once. + let workload_dag = output.execution_timed_dag().unwrap(); + assert_eq!(workload_dag.roots.len(), 2); + assert_ne!(workload_dag.roots[0], workload_dag.roots[1]); + assert_eq!( + workload_dag + .nodes + .iter() + .filter(|n| matches!( + n.payload, + asap_types::ir::physical_export::PhysicalASAPOperatorPayload::ASAP( + asap_types::ir::ASAPOp::SummaryAgg { .. } + ) + )) + .count(), + 1 + ); +} diff --git a/crates/integration-tests/tests/operator_sharing.rs b/crates/integration-tests/tests/operator_sharing.rs new file mode 100644 index 000000000..6a87a879a --- /dev/null +++ b/crates/integration-tests/tests/operator_sharing.rs @@ -0,0 +1,195 @@ +//! Acceptance tests for operator sharing (issue #468): one operator IR +//! before and after ASAP optimization, so non-ASAP operators sit both above +//! and below summary operators, can share inputs with them, and can carry +//! summaries below set operators. +//! +//! Each test drives SQL text through `lower_sql` → `search_workload` → +//! global selection → `assemble_selected_dag`, the pipeline +//! `sql_to_post_asap.rs` uses. + +use std::rc::Rc; + +use asap_aware_mapping::{search_workload, DefaultCostModel}; +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; +use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::types::AccuracyTarget; + +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) +} + +/// TPC-H `lineitem`, as `frontend-sql/tests/data_quality_check/tpch_deequ.rs` +/// declares it (DECIMAL columns as `Float64`, no time index, no keys). +fn catalog() -> SqlCatalog { + SqlCatalog::new().with_table( + "lineitem", + Schema::new(vec![ + col("l_orderkey", DataType::Int64), + col("l_partkey", DataType::Int64), + col("l_suppkey", DataType::Int64), + col("l_linenumber", DataType::Int64), + col("l_quantity", DataType::Float64), + col("l_extendedprice", DataType::Float64), + col("l_discount", DataType::Float64), + col("l_tax", DataType::Float64), + col("l_returnflag", DataType::Utf8), + col("l_linestatus", DataType::Utf8), + col("l_shipdate", DataType::Date), + col("l_commitdate", DataType::Date), + col("l_receiptdate", DataType::Date), + col("l_shipinstruct", DataType::Utf8), + col("l_shipmode", DataType::Utf8), + col("l_comment", DataType::Utf8), + ]), + ) +} + +/// Lower `sql`, search, select with the default cost model and assemble the +/// selected post-ASAP DAG. +async fn plan(sql: &str, accuracy: AccuracyTarget) -> Rc { + let pre = lower_sql(sql, &catalog(), accuracy) + .await + .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")); + let space = search_workload(vec![("query", pre)]); + let selection = space.global_selection(&DefaultCostModel); + selection + .assemble_selected_dag(&space.roots[0].1) + .expect("materialization failed") + .expect("root must be discovered") +} + +/// Every unique node reachable from `root` whose operator matches `pred`. +fn find_all( + root: &Rc, + pred: impl Fn(&OperatorNode) -> bool, +) -> Vec> { + OperatorNode::reachable(root) + .into_iter() + .filter(|node| pred(node)) + .collect() +} + +fn is_summary_evaluation(node: &OperatorNode) -> bool { + matches!( + node.operator, + Operator::ASAP( + ASAPOp::SummaryEstimate { .. } + | ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::EvaluatePopulation { .. } + ) + ) +} + +fn is_scan(node: &OperatorNode) -> bool { + matches!(node.non_asap(), Some(NonASAPOp::Scan { .. })) +} + +/// The first node reached through single-input non-ASAP operators below +/// `node` (inclusive) that is not one: where an operator chain meets a +/// summary or a multi-input operator. +fn through_unary_non_asap(node: &Rc) -> &Rc { + match node.non_asap().map(|op| op.children()) { + Some(children) if children.len() == 1 => through_unary_non_asap(children[0]), + _ => node, + } +} + +// #468 problem 1: the Project above the summary evaluation and the Scan below +// it are both plain NonASAP nodes (no post-ASAP-only wrapper variant). +#[ignore = "planner chooses no summary here: Avg has no summary realization, so the Aggregate stays a logical pass-through"] +#[tokio::test] +async fn project_above_and_scan_below_a_summary_are_both_non_asap_nodes() { + let root = plan( + "WITH metric AS (SELECT avg(CASE WHEN l_quantity BETWEEN 1 AND 50 THEN 1.0 ELSE 0.0 END) \ + AS in_range FROM lineitem) SELECT in_range, in_range = 1.0 AS ok FROM metric", + AccuracyTarget::Exact, + ) + .await; + // The root is the outer SELECT list: a NonASAP Project. + assert!( + matches!(root.operator, Operator::NonASAP(NonASAPOp::Project { .. })), + "root must be the outer Project, got {:?}", + root.operator + ); + // A summary evaluation sits below the Project chain. + let evaluation = through_unary_non_asap(&root); + assert!( + is_summary_evaluation(evaluation), + "the Project chain must read a summary, got {:?}", + evaluation.operator + ); + // Below the summary the Scan is the same NonASAP operator a front end emits. + let scans = find_all(evaluation, is_scan); + assert_eq!(scans.len(), 1, "one lineitem Scan below the summary"); + assert!(!scans[0].is_asap()); + // The flat plan exports (time first, wire 6). + post_asap_dag(&root); +} + +// #468 problem 2: the exact aggregate and the sketch read one shared Scan +// (`Rc::ptr_eq`), not two copies. +#[ignore = "waits for the binding rule splitting multi-measure aggregates"] +#[tokio::test] +async fn exact_aggregate_and_sketch_share_one_scan() { + let root = plan( + "SELECT avg(l_extendedprice), approx_percentile_cont(l_discount, 0.99) FROM lineitem", + AccuracyTarget::Epsilon(0.01), + ) + .await; + let exact = find_all(&root, |node| { + matches!(node.non_asap(), Some(NonASAPOp::Aggregate { .. })) + }); + let sketch = find_all(&root, |node| { + matches!(node.asap(), Some(ASAPOp::SummaryAgg { .. })) + }); + assert_eq!(exact.len(), 1, "one exact Aggregate for avg: {root:?}"); + assert_eq!( + sketch.len(), + 1, + "one sketch SummaryAgg for the percentile: {root:?}" + ); + let scan_under = |node: &Rc| { + let scans = find_all(node, is_scan); + assert_eq!(scans.len(), 1, "one Scan under {:?}", node.operator); + Rc::clone(&scans[0]) + }; + assert!( + Rc::ptr_eq(&scan_under(&exact[0]), &scan_under(&sketch[0])), + "the exact aggregate and the sketch must read one shared Scan" + ); +} + +// #468 problem 3: a summary can sit below a set operator — each side of the +// UNION ALL holds its own SummaryEstimate. +#[tokio::test] +async fn each_side_of_union_all_holds_a_summary_estimate() { + let root = plan( + "SELECT approx_distinct(l_partkey) FROM lineitem \ + UNION ALL SELECT approx_distinct(l_suppkey) FROM lineitem", + AccuracyTarget::Epsilon(0.01), + ) + .await; + // The SQL front end lowers UNION ALL to `SetOp { all: true }`. + let Some(NonASAPOp::SetOp { + all: true, + left, + right, + .. + }) = root.non_asap() + else { + panic!("root must be the UNION ALL SetOp, got {:?}", root.operator) + }; + for side in [left, right] { + let estimates = find_all(side, |node| { + matches!(node.asap(), Some(ASAPOp::SummaryEstimate { .. })) + }); + assert!( + !estimates.is_empty(), + "UNION ALL side has no SummaryEstimate: {:?}", + side.operator + ); + } + post_asap_dag(&root); +} diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs index 93bebd338..2f03f459c 100644 --- a/crates/integration-tests/tests/physical_common/mod.rs +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -7,6 +7,7 @@ use asap_physical_operators::{ use futures::{executor::block_on, StreamExt}; use std::collections::BTreeMap; +#[allow(dead_code)] pub fn execute( plan: &CompiledPhysicalDAG, inputs: BTreeMap, @@ -40,3 +41,17 @@ pub fn execute( .await }) } + +#[allow(dead_code)] +pub fn compile_physical_asap_dag( + root: &std::rc::Rc, +) -> Result> { + let root = asap_types::ir::apply_lifecycle_timings( + root, + &Default::default(), + &mut Default::default(), + )?; + Ok(asap_types::ir::physical_export::compile_physical_asap_dag( + &root, + )?) +} diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index 9937469cc..55b1fe903 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -1,10 +1,15 @@ //! Planner-selected summaries over raw samples compile as precompute DAGs //! and produce the same estimates as feeding their kernel sample by sample. +mod physical_common; +use asap_types::ir::physical_export::{PhysicalASAPDAG, PhysicalASAPOperatorPayload}; +use asap_types::ir::ASAPOp; +use asap_types::ir::OperatorNode; +use physical_common::compile_physical_asap_dag; use std::{collections::BTreeMap, collections::BTreeSet, rc::Rc, sync::Arc}; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_aware_mapping::{ - search_workload, Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, + search_workload, ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_integration_tests::fixtures::lower_promql; @@ -18,11 +23,10 @@ use asap_physical_operators::{ AggregateCore, KeyByLabelValues, Statistic, }; use asap_types::post_asap::{ - compile_post_asap_dag, EntityIdentity, ExactKind, FieldDataType, PostAsapDAG, - PostAsapOperatorPayload, SketchAlgorithm, SketchStatistic, SummaryInputExpr, SummaryNode, + EntityIdentity, ExactKind, FieldDataType, SketchAlgorithm, SketchStatistic, SummaryInputExpr, SummaryUpdate, }; -use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; +use asap_types::pre_asap::{expr_ir::ColumnRef, Reduction}; use asap_types::types::AccuracyTarget; use futures::{executor::block_on, StreamExt}; @@ -51,14 +55,14 @@ fn canonical(labels: &Series) -> Series { /// Every Planner candidate for `query`: the searched selection plus each /// summary replacement of the root. -fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { - let root = Rc::new(lower_promql(query, accuracy).expect("lowering failed")); - let mut result = SketchAlgorithmStrategy::default_cost_model() +fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { + let root = lower_promql(query, accuracy).expect("lowering failed"); + let mut result = ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&root)) .into_iter() .filter_map(|candidate| match candidate { ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. } => Some(node), _ => None, @@ -75,10 +79,15 @@ fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { } /// Raw-input summary nodes: `(dag, raw source id, summary id)`. -fn raw_summaries(dag: &PostAsapDAG) -> Vec<(u64, u64)> { +fn raw_summaries(dag: &PhysicalASAPDAG) -> Vec<(u64, u64)> { dag.nodes .iter() - .filter(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .filter(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { .. }) + ) + }) .filter_map(|node| { let inputs = dag .edges @@ -89,8 +98,11 @@ fn raw_summaries(dag: &PostAsapDAG) -> Vec<(u64, u64)> { return None; }; let source = dag.nodes.iter().find(|n| n.id == edge.producer)?; - matches!(source.payload, PostAsapOperatorPayload::Fallback { .. }) - .then_some((u64::from(source.id.0), u64::from(node.id.0))) + matches!( + source.payload, + PhysicalASAPOperatorPayload::NonASAP(asap_types::ir::NonASAPOp::TimeRange { .. }) + ) + .then_some((source.id as u64, node.id as u64)) }) .collect() } @@ -114,7 +126,7 @@ fn samples() -> Vec<(Series, i64, f64)> { } fn execute( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, source: u64, root: u64, rows: &[(Series, i64, f64)], @@ -124,7 +136,7 @@ fn execute( "raw summary {root} does not compile: {error}; source {:?}", dag.nodes .iter() - .find(|n| u64::from(n.id.0) == source) + .find(|n| n.id as u64 == source) .map(|n| (&n.output_schema, &n.payload)) ); }); @@ -240,7 +252,7 @@ fn weight(update: &SummaryUpdate, value: f64) -> f64 { } /// Estimates that identify a state's content for comparison. -fn readouts(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { +fn evaluations(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { if let Some(exact) = state.as_any().downcast_ref::() { let FieldDataType::ExactAggregate(kind, _) = family else { unreachable!() @@ -255,7 +267,7 @@ fn readouts(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { other => panic!("unexpected exact kind {other:?}"), }; return vec![exact - .readout(statistic, None, None::<&KeyByLabelValues>) + .evaluation(statistic, None, None::<&KeyByLabelValues>) .unwrap() .unwrap()]; } @@ -277,31 +289,23 @@ fn readouts(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { /// or the family when it has no native state. fn check( query: &str, - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, source: u64, root: u64, rows: &[(Series, i64, f64)], ) -> Result { - let node = dag - .nodes - .iter() - .find(|n| u64::from(n.id.0) == root) - .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let node = dag.nodes.iter().find(|n| n.id as u64 == root).unwrap(); + let PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family, input, reduction, grouping, .. - } = &node.payload + }) = &node.payload else { unreachable!() }; - let source_node = dag - .nodes - .iter() - .find(|n| u64::from(n.id.0) == source) - .unwrap(); + let source_node = dag.nodes.iter().find(|n| n.id as u64 == source).unwrap(); let keys = match reduction { Reduction::Reduce(keys) => keys .keys() @@ -370,8 +374,8 @@ fn check( for (labels, state) in actual { let reference = expected[&labels].snapshot_accumulator(); assert_eq!( - readouts(state.as_ref(), family), - readouts(reference.as_ref(), family), + evaluations(state.as_ref(), family), + evaluations(reference.as_ref(), family), "{query}: {labels:?}" ); } @@ -413,7 +417,7 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { let mut checked = BTreeMap::new(); for (query, accuracy) in queries { for candidate in candidates(query, accuracy.clone()) { - let dag = compile_post_asap_dag(&candidate).unwrap(); + let dag = compile_physical_asap_dag(&candidate).unwrap(); for (source, root) in raw_summaries(&dag) { match check(query, &dag, source, root, &rows) { Ok(family) => { @@ -459,25 +463,21 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { /// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with /// another update, keeping its raw input and reduction. -fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PostAsapDAG, u64, u64) { +fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PhysicalASAPDAG, u64, u64) { let candidate = candidates( "sum by (service) (sum_over_time(m[5m]))", AccuracyTarget::Exact, ) .pop() .unwrap(); - let mut dag = compile_post_asap_dag(&candidate).unwrap(); + let mut dag = compile_physical_asap_dag(&candidate).unwrap(); let (source, root) = raw_summaries(&dag)[0]; - let node = dag - .nodes - .iter_mut() - .find(|n| u64::from(n.id.0) == root) - .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let node = dag.nodes.iter_mut().find(|n| n.id as u64 == root).unwrap(); + let PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: old, input: update, .. - } = &mut node.payload + }) = &mut node.payload else { unreachable!() }; @@ -489,11 +489,7 @@ fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PostAsap *old = family; *update = input; let schema = node.output_schema.clone(); - for edge in dag - .edges - .iter_mut() - .filter(|e| u64::from(e.producer.0) == root) - { + for edge in dag.edges.iter_mut().filter(|e| e.producer as u64 == root) { edge.intermediate_schema = schema.clone(); } (dag, source, root) @@ -587,7 +583,7 @@ fn raw_sample_heaps_resolve_items_from_labels() { // `without` grouping over raw samples drops the listed labels and `__name__`. #[test] fn raw_sample_without_grouping_drops_labels_and_name() { - use asap_types::pre_asap::query_expr::GroupKeys; + use asap_types::pre_asap::GroupKeys; let family = FieldDataType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); let (mut dag, source, root) = @@ -595,19 +591,16 @@ fn raw_sample_without_grouping_drops_labels_and_name() { let service = dag .nodes .iter() - .find(|n| u64::from(n.id.0) == source) + .find(|n| n.id as u64 == source) .unwrap() .output_schema .fields .iter() .position(|f| f.name == "service") .unwrap(); - let node = dag - .nodes - .iter_mut() - .find(|n| u64::from(n.id.0) == root) - .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = &mut node.payload else { + let node = dag.nodes.iter_mut().find(|n| n.id as u64 == root).unwrap(); + let PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &mut node.payload + else { unreachable!() }; *reduction = Reduction::Reduce(GroupKeys::without(vec![service])); diff --git a/crates/integration-tests/tests/promql_numeric_regressions.rs b/crates/integration-tests/tests/promql_numeric_regressions.rs index 692fe9c7a..3ec4d8eac 100644 --- a/crates/integration-tests/tests/promql_numeric_regressions.rs +++ b/crates/integration-tests/tests/promql_numeric_regressions.rs @@ -1,36 +1,41 @@ -//! Numeric regression fixtures: actual PromQL lowering plus numeric update/readout checks. +//! Numeric regression fixtures: actual PromQL lowering plus numeric update/evaluation checks. //! The count/sum interpreter below verifies planner update semantics, not a deployed backend. -use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; +use asap_aware_mapping::replacement::is_logical_rewrite; +use asap_aware_mapping::{ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, FieldDataType, SummaryExpr, SummaryInputExpr, SummaryNode, - SummaryUpdate, -}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; +use asap_types::post_asap::{ExactKind, FieldDataType, SummaryInputExpr, SummaryUpdate}; use asap_types::pre_asap::{ColumnRef, Reduction}; use asap_types::types::AccuracyTarget; use std::rc::Rc; -fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { - let pre = Rc::new(lower_promql(query, accuracy).unwrap()); - SketchAlgorithmStrategy::default_cost_model() +fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { + let pre = lower_promql(query, accuracy).unwrap(); + ASAPStrategies::default_cost_model() .replacements(&TargetSubDAG::new(&pre)) .into_iter() .find_map(|r| match r.replacement { - Replacement::Summary(n) => Some(n), + // A bound decision: a summary DAG or a kept (exact) sub-DAG. + Replacement::SubDAG(n) if !is_logical_rewrite(&n) => Some(n), _ => None, }) - .unwrap_or_else(|| asap_aware_mapping::replacement::keep_pre_asap(&pre).unwrap()) + .unwrap_or_else(|| asap_aware_mapping::replacement::retain_exact(&pre).unwrap()) } -fn aggregate(node: &SummaryNode) -> (&FieldDataType, &SummaryUpdate, &Reduction) { - match &node.expr { - SummaryExpr::SummaryAgg { +fn aggregate(node: &OperatorNode) -> (&FieldDataType, &SummaryUpdate, &Reduction) { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, .. - } => (family, input, reduction), - SummaryExpr::SummaryEstimate { summary_input, .. } => aggregate(summary_input), - SummaryExpr::ValueOperation { child, .. } => aggregate(child), + }) => (family, input, reduction), + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => aggregate(summary_input), + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => aggregate(child), + // A value operation (Project/Filter/Sort/Limit/...) over the state. + Operator::NonASAP(op) if op.children().len() == 1 && node.contains_asap() => { + aggregate(op.children()[0]) + } other => panic!("not a maintained accumulator: {other:?}"), } } @@ -63,7 +68,7 @@ fn count_up_counts_targets_even_when_values_repeat_or_change_sign() { 3. ); } - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } /// Ten samples give count ten, whereas sum retains the signed sample values. @@ -81,7 +86,7 @@ fn window_counts_and_sums_distinguish_one_zero_three_and_negative_values() { let got: f64 = (0..10).map(|_| contribution(family, update, value)).sum(); assert_eq!(got, if is_count { 10. } else { value * 10. }); } - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -97,7 +102,7 @@ fn sum_rate_and_increase_have_real_exact_accumulator_nodes() { let (family, _, _) = aggregate(&node); assert!(matches!(family, FieldDataType::ExactAggregate(k, _) if *k == kind)); assert!(node.guarantee.as_ref().unwrap().is_exact()); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -109,11 +114,12 @@ fn checked_ratio_must_not_certify_cross_zero_interpolation() { AccuracyTarget::Epsilon(0.01), ); assert!( - matches!(node.expr, SummaryExpr::BinaryOp { .. }), + matches!(node.operator, Operator::NonASAP(NonASAPOp::BinaryOp { .. })) + && node.contains_asap(), "direct quantile ratio should remain an available candidate" ); assert!(node.guarantee.is_none()); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); // Keep the actual signed-sketch counterexample: division guards alone pass // even though the quantile interpolation does not preserve relative error. let alpha = (0.01 - 8.0 * f64::EPSILON) / 2.01; @@ -148,20 +154,20 @@ fn quantile_over_temporal_average_keeps_a_legal_candidate() { ] { let node = plan(query, AccuracyTarget::Epsilon(0.01)); assert!( - !matches!(node.expr, SummaryExpr::KeepPreAsap(_)), + node.contains_asap(), "outer sketch candidate must survive: {query}" ); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { - panic!("outer sketch readout") + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator else { + panic!("outer sketch evaluation") }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("outer sketch state") }; assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + !child.contains_asap(), "guarded expression must retain native maintenance input" ); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -195,7 +201,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { use asap_aware_mapping::accuracy::{DefaultAccuracyModel, EqualSplitAllocator}; use asap_aware_mapping::cost_model::DefaultCostModel; use asap_types::post_asap::{NonNegativeWeightProof, SketchAlgorithm, WeightDomain}; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -207,7 +213,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { } else { "topk(1, sum_over_time(up[5m]))" }; - let pre = Rc::new(lower_promql(query, AccuracyTarget::Epsilon(0.01)).unwrap()); + let pre = lower_promql(query, AccuracyTarget::Epsilon(0.01)).unwrap(); let candidates = strategy.replacements(&TargetSubDAG::new(&pre)); let wanted = if is_count { SketchAlgorithm::CmsWithHeap @@ -217,7 +223,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { let node = candidates .iter() .find_map(|c| { - let Replacement::Summary(node) = &c.replacement else { + let Replacement::SubDAG(node) = &c.replacement else { return None; }; let (family, _, _) = aggregate(node); @@ -240,7 +246,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { SummaryInputExpr::Column(ColumnRef::SampleValue) ); for c in &candidates { - if let Replacement::Summary(n) = &c.replacement { + if let Replacement::SubDAG(n) = &c.replacement { assert!( !matches!(aggregate(n).0, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) ); @@ -271,6 +277,6 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { }; assert_eq!(got, if is_count { 10. } else { 10. * value }); } - compile_post_asap_dag(node).unwrap(); + post_asap_dag(node); } } diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 97a08a0c4..b999720fa 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -1,8 +1,8 @@ //! End-to-end query-string → post-ASAP IR pin (issue #98). //! -//! Drives the full pipeline — PromQL text → pre-ASAP `QueryExpr` -//! (`lower_promql`) → post-ASAP `SummaryExpr` DAG (via -//! `SketchAlgorithmStrategy::replacements`, see [`realize`] below) — and pins +//! Drives the full pipeline — PromQL text → non-ASAP `OperatorNode` +//! (`lower_promql`) → post-ASAP `OperatorNode` DAG (via +//! `ASAPStrategies::replacements`, see [`realize`] below) — and pins //! the summary-bound shape node by node, including the family `(Kind, //! Params)` committed on each edge's schema. @@ -13,67 +13,77 @@ use asap_aware_mapping::accuracy::{ QuantileInputDomain, }; use asap_aware_mapping::cost_model::DefaultCostModel; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{is_logical_rewrite, retain_exact, RealizationError}; use asap_aware_mapping::{ - search_workload, search_workload_with_targets, AccuracyModel, Replacement, ReplacementStrategy, - ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, + search_workload, search_workload_with_targets, ASAPStrategies, AccuracyModel, Replacement, + ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use asap_integration_tests::fixtures::lower_promql; +use asap_integration_tests::post_asap::{post_asap_dag, timed}; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::physical_export::PhysicalASAPOperatorPayload; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; use asap_types::post_asap::{ - compile_post_asap_dag, CompositionOperator, EntityIdentity, ExactKind, ExactParams, - FieldDataType, GroupingStrategy, Schema, SketchAlgorithm, SketchKind, SketchParams, - SketchStatistic, SummaryExpr, SummaryInputExpr, SummaryNode, SummaryUpdate, ValueOperation, + CompositionOperator, EntityIdentity, ExactKind, ExactParams, FieldDataType, GroupingStrategy, + Schema, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, SummaryInputExpr, + SummaryUpdate, }; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; use asap_types::pre_asap::schema::DataType; use asap_types::types::AccuracyTarget; -/// This crate has no "bind me one DAG" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// This crate has no "bind me one tree" public API any more — +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn realize(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default_cost_model() .replacements(&target) .into_iter() .next() { + // A bound decision (summary DAG or kept sub-DAG); a logical rewrite + // is not a binding, so it falls back to keeping the target. Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. - }) => Ok(node), - _ => keep_pre_asap(&root), + }) if !is_logical_rewrite(&node) => Ok(node), + _ => retain_exact(root), } + .inspect(|node| { + node.validate_structure() + .expect("planned dag satisfies the unified IR contract") + }) } #[test] -fn distinct_over_time_offers_hll_cardinality_readout() { +fn distinct_over_time_offers_hll_cardinality_evaluation() { // The real frontend must reach an existing HLL candidate without a // function-specific post-ASAP node or a sample-count rewrite. - let root = Rc::new( - lower_promql( - "distinct_over_time(cpu_usage{job=\"worker\"}[5m])", - AccuracyTarget::Epsilon(0.02), - ) - .unwrap(), - ); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql( + "distinct_over_time(cpu_usage{job=\"worker\"}[5m])", + AccuracyTarget::Epsilon(0.02), + ) + .unwrap(); + let candidates = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); + for candidate in &candidates { + if let Replacement::SubDAG(node) = &candidate.replacement { + node.validate_structure().unwrap(); + } + } assert!(candidates.iter().any(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { return false }; - let SummaryExpr::SummaryEstimate { summary_input, query, .. } = &node.expr else { return false }; + let Replacement::SubDAG(node) = &candidate.replacement else { return false }; + let Some(ASAPOp::SummaryEstimate { summary_input, query, .. }) = node.asap() else { return false }; matches!(query, SketchStatistic::Cardinality) - && matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + && matches!(summary_input.asap(), Some(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::Hll) }), "no HLL cardinality candidate: {candidates:?}"); } -fn lower_search_and_materialize(query: &str) -> Rc { - let pre = Rc::new(lower_promql(query, AccuracyTarget::Exact).expect("lowering failed")); +fn lower_search_and_materialize(query: &str) -> Rc { + let pre = lower_promql(query, AccuracyTarget::Exact).expect("lowering failed"); let space = search_workload(vec![("query", pre)]); let selection = space.global_selection(&DefaultCostModel); selection @@ -88,36 +98,38 @@ fn value_ranked_topk_preserves_summary_children_in_post_asap_dag() { "topk(3, rate(cpu_seconds_total[5m]))", "topk by (job) (2, max_over_time(memory_bytes[6h]))", ] { - let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { n, offset, .. }, + let root = timed(&lower_search_and_materialize(query)); + let Some(NonASAPOp::Limit { + n, + offset, child: sort, .. - } = &root.expr + }) = root.non_asap() else { - panic!("expected query-time Limit for {query}, got {:?}", root.expr); + panic!( + "expected query-time Limit for {query}, got {:?}", + root.operator + ); }; - assert!(*n > 0 && *offset == 0); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Sort { .. }, - child, - .. - } = &sort.expr - else { + assert_eq!( + root.timing, + Some(asap_types::post_asap::ExecutionTiming::QueryTime) + ); + assert!(n.is_some_and(|n| n > 0) && *offset == 0); + let Some(NonASAPOp::Sort { child, .. }) = sort.non_asap() else { panic!("expected query-time Sort under Limit for {query}"); }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - child: state, - .. - } = &child.expr - else { + assert_eq!( + sort.timing, + Some(asap_types::post_asap::ExecutionTiming::QueryTime) + ); + let Some(ASAPOp::FinalizeExactAccumulator { child: state }) = child.asap() else { panic!( "Sort must consume finalized values for {query}: {:?}", - child.expr + child.operator ); }; - assert!(matches!(state.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(state.asap(), Some(ASAPOp::SummaryAgg { .. }))); assert!(child .schema .fields @@ -135,9 +147,9 @@ fn exact_counter_weighted_topk_fails_closed_without_membership_certificate() { ] { let root = lower_search_and_materialize(query); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + !root.contains_asap(), "exact target must not accept an uncertified membership sidecar for {query}: {:?}", - root.expr + root.operator ); } } @@ -146,23 +158,14 @@ fn exact_counter_weighted_topk_fails_closed_without_membership_certificate() { fn instant_topk_and_unsupported_child_remain_local_residuals() { for query in ["topk(3, memory_bytes)", "topk(3, deriv(memory_bytes[5m]))"] { let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { .. }, - child: sort, - .. - } = &root.expr - else { + let Some(NonASAPOp::Limit { child: sort, .. }) = root.non_asap() else { panic!("expected Limit for {query}"); }; - let SummaryExpr::ValueOperation { - child, operation, .. - } = &sort.expr - else { + let Some(NonASAPOp::Sort { child, .. }) = sort.non_asap() else { panic!("expected Sort for {query}"); }; - assert!(matches!(operation, ValueOperation::Sort { .. })); assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + !child.contains_asap(), "only the unsupported child should remain exact for {query}" ); } @@ -177,7 +180,7 @@ fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { .dtype } -fn lower_and_realize(query: &str) -> Rc { +fn lower_and_realize(query: &str) -> Rc { let pre = lower_promql(query, AccuracyTarget::Exact).expect("lowering failed"); realize(&pre).expect("binding failed") } @@ -186,19 +189,17 @@ fn lower_and_realize(query: &str) -> Rc { fn promql_binary_arithmetic_retains_two_summary_leaves() { for op in ["+", "-", "*", "/", "%", "^", "atan2"] { let root = lower_and_realize(&format!("rate(a[1m]) {op} rate(b[1m])")); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp for {op}, got {:?}", root.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp for {op}, got {:?}", root.operator); }; for operand in [lhs, rhs] { - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } = &operand.expr - else { - panic!("expected an explicit exact readout, got {:?}", operand.expr); + let Some(ASAPOp::FinalizeExactAccumulator { child }) = operand.asap() else { + panic!( + "expected an explicit exact evaluation, got {:?}", + operand.operator + ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(child.asap(), Some(ASAPOp::SummaryAgg { .. }))); } } } @@ -207,47 +208,36 @@ fn promql_binary_arithmetic_retains_two_summary_leaves() { fn value_ranked_topk_over_binary_ratio_finalizes_both_summary_operands() { let query = "topk(1, sum by(job)(increase(a[6h])) / sum by(job)(increase(b[6h])))"; let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { - n: 1, offset: 0, .. - }, + let Some(NonASAPOp::Limit { + n: Some(1), + offset: 0, child: sort, .. - } = &root.expr + }) = root.non_asap() else { - panic!("expected Limit root, got {:?}", root.expr); + panic!("expected Limit root, got {:?}", root.operator); }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::Sort { .. }, - child: binary, - .. - } = &sort.expr - else { - panic!("expected Sort below Limit, got {:?}", sort.expr); + let Some(NonASAPOp::Sort { child: binary, .. }) = sort.non_asap() else { + panic!("expected Sort below Limit, got {:?}", sort.operator); }; - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &binary.expr else { - panic!("expected BinaryOp below Sort, got {:?}", binary.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = binary.non_asap() else { + panic!("expected BinaryOp below Sort, got {:?}", binary.operator); }; for operand in [lhs, rhs] { - let SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - child, - .. - } = &operand.expr - else { + let Some(ASAPOp::FinalizeExactAccumulator { child }) = operand.asap() else { panic!( "expected exact accumulator finalization, got {:?}", - operand.expr + operand.operator ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(child.asap(), Some(ASAPOp::SummaryAgg { .. }))); } } struct SeparatedTopK; impl AccuracyEvidenceProvider for SeparatedTopK { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &OperatorNode) -> Option { Some(1000) } @@ -271,14 +261,12 @@ impl AccuracyEvidenceProvider for SeparatedTopK { // Rate-weighted summaries must consume finalized rates, never raw counter deltas. #[test] fn grouped_rate_topk_consumes_finalized_rate_values() { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -288,24 +276,29 @@ fn grouped_rate_topk_consumes_finalized_rate_values() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), + Replacement::SubDAG(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), _ => None, }) .expect("rate-weighted CMS plan"); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); assert!(!dag.nodes.iter().any(|node| matches!( node.payload, - asap_types::post_asap::PostAsapOperatorPayload::RelationalJoin { .. } + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Join { .. }) ))); - let node = dag.nodes.iter().find(|node| matches!(&node.payload, - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } - if kind.algorithm() == &SketchAlgorithm::CmsWithHeap)).unwrap(); + let node = dag + .nodes + .iter() + .find(|node| { + matches!(&node.payload, + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) + if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) + }) + .unwrap(); assert_eq!( node.output_state.timing, asap_types::post_asap::ExecutionTiming::QueryTime ); - let asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { input, .. } = &node.payload - else { + let PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { input, .. }) = &node.payload else { unreachable!() }; assert_eq!( @@ -332,14 +325,12 @@ fn weighted_topk_keeps_candidates_with_missing_population_evidence() { SeparatedTopK.propagation_stats(op, family, query) } } - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -355,27 +346,24 @@ fn weighted_topk_keeps_candidates_with_missing_population_evidence() { // Unknown requirements must survive physical export for deployment to inspect. #[test] fn weighted_topk_exports_symbolic_evidence_requirements() { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let candidates = ASAPStrategies::default_cost_model().replacements(&TargetSubDAG::new(&root)); let candidate = candidates .iter() .find(|candidate| candidate.rationale.contains("CmsWithHeap")) .unwrap(); assert!(candidate.has_missing_accuracy_evidence()); - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { panic!("summary candidate") }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); let exported = serde_json::to_string(&dag).unwrap(); assert!(exported.contains("topk_max_distinct_items")); assert!(exported.contains("topk_membership_margin")); @@ -387,18 +375,16 @@ fn weighted_topk_exports_symbolic_evidence_requirements() { fn weighted_topk_rejects_invalid_population_evidence() { struct InvalidPopulation; impl AccuracyEvidenceProvider for InvalidPopulation { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &OperatorNode) -> Option { Some(0) } } - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -415,17 +401,15 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { "topk by(job)(2, sum by(service, job)(rate(m[1m])))", "topk(2, sum by(job)(increase(m[6h])))", ] { - let root = Rc::new( - lower_promql( - query, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let root = lower_promql( + query, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -435,60 +419,50 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => { + Replacement::SubDAG(node) if candidate.rationale.contains("CmsWithHeap") => { Some(node) } _ => None, }) .expect("weighted summary"); - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Limit { + n: Some(2), + offset: 0, + partition_by, child: sorted, - operation: - ValueOperation::Limit { - n: 2, - offset: 0, - partition_by, - }, - .. - } = &plan.expr + }) = plan.non_asap() else { panic!("grouped limit") }; - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Sort { + partition_by: sort_groups, child: projected, - operation: - ValueOperation::Sort { - partition_by: sort_groups, - .. - }, .. - } = &sorted.expr + }) = sorted.non_asap() else { panic!("grouped sort") }; assert_eq!(partition_by, sort_groups); assert_eq!(partition_by.len(), usize::from(query.contains("topk by"))); - let SummaryExpr::ValueOperation { - child: readout, - operation: ValueOperation::Project { .. }, - .. - } = &projected.expr + let Some(NonASAPOp::Project { + child: evaluation, .. + }) = projected.non_asap() else { panic!("logical output projection") }; - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, query: SketchStatistic::TopK { k }, - } = &readout.expr + }) = evaluation.asap() else { - panic!("heap readout") + panic!("heap evaluation") }; assert!(*k > 2, "candidate capacity is independent of output count"); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child: rates, input, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("weighted summary") }; @@ -497,13 +471,10 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { SummaryInputExpr::Column(ColumnRef::SampleValue) ); assert!(matches!( - rates.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + rates.asap(), + Some(ASAPOp::FinalizeExactAccumulator { .. }) )); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); for phase in [ asap_types::post_asap::ExecutionTiming::IngestionTime, asap_types::post_asap::ExecutionTiming::QueryTime, @@ -525,57 +496,41 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { #[test] fn promql_binary_arithmetic_preserves_both_scalar_operand_orders() { - fn is_exact_readout_or_scalar(node: &SummaryNode) -> bool { - matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - || matches!( - node.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) - } - for query in ["rate(a[1m]) / 2", "2 / rate(a[1m])"] { + for (query, scalar_left) in [("rate(a[1m]) / 2", false), ("2 / rate(a[1m])", true)] { let root = lower_and_realize(query); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp for {query}, got {:?}", root.expr); + let Some(NonASAPOp::Project { cols, .. }) = root.non_asap() else { + panic!("expected Project") }; - assert!(is_exact_readout_or_scalar(lhs)); - assert!(is_exact_readout_or_scalar(rhs)); - assert!( - matches!( - lhs.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) || matches!( - rhs.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) - ); + let ScalarExpr::Arithmetic { left, right, .. } = &cols[1].expr else { + panic!() + }; + let (scalar, sample) = if scalar_left { + (left, right) + } else { + (right, left) + }; + assert_eq!(**scalar, ScalarExpr::literal_f64(2.0)); + assert_eq!(**sample, ScalarExpr::Column(1)); + assert!(root.schema.has_promql_series_identity()); } } #[test] fn promql_binary_arithmetic_falls_back_as_a_whole_for_unsupported_arm() { let root = lower_and_realize("rate(a[1m]) + stddev_over_time(b[1m])"); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] fn promql_binary_arithmetic_preserves_nested_structure_and_rejects_modifiers() { let nested = lower_and_realize("(rate(a[1m]) + rate(b[1m])) / 2"); - let SummaryExpr::BinaryOp { lhs, .. } = &nested.expr else { - panic!("expected outer BinaryOp, got {:?}", nested.expr); + let Some(NonASAPOp::Project { child: lhs, .. }) = nested.non_asap() else { + panic!("expected outer BinaryOp, got {:?}", nested.operator); }; - assert!(matches!(lhs.expr, SummaryExpr::BinaryOp { .. })); + assert!(matches!(lhs.non_asap(), Some(NonASAPOp::BinaryOp { .. }))); let modified = lower_and_realize("rate(a[1m]) + on(job) rate(b[1m])"); - assert!(matches!(modified.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!modified.contains_asap()); } #[test] @@ -586,8 +541,8 @@ fn promql_binary_arithmetic_never_relabels_approximate_children_as_exact() { ) .expect("lowering failed"); let root = realize(&pre).expect("binding failed"); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp, got {:?}", root.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp, got {:?}", root.operator); }; assert!(lhs.guarantee.as_ref().is_some_and(|g| !g.is_exact())); assert!(rhs.guarantee.as_ref().is_some_and(|g| !g.is_exact())); @@ -606,13 +561,11 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { epsilon: 0.01, delta: 0.01, }; - let query = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - target.clone(), - ) - .expect("lowering failed"), - ); + let query = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + target.clone(), + ) + .expect("lowering failed"); let evidence = FixtureQuantileDomain { lower: 1.0, @@ -632,7 +585,7 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { .for_target(root) .and_then(|selection| selection.chosen.as_ref()) .expect("the certified DDSketch ratio should be selectable"); - let Replacement::Summary(node) = &chosen.replacement else { + let Replacement::SubDAG(node) = &chosen.replacement else { panic!("expected a summary candidate") }; let guarantee = node.guarantee.as_ref().expect("ratio guarantee"); @@ -642,23 +595,22 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { "ratio guarantee should satisfy the requested target: {guarantee:?}" ); - let shared = - asap_types::post_asap::share_common_summary_sub_dags(vec![("ratio", node.clone())]); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &shared[0].1.expr else { + let shared = asap_types::ir::cse::share_common_sub_dags(vec![("ratio", node.clone())]); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = shared[0].1.non_asap() else { panic!("expected binary ratio") }; - let producer = |readout: &Rc| match &readout.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => Rc::clone(summary_input), - other => panic!("expected DDSketch readout, got {other:?}"), + let producer = |evaluation: &Rc| match &evaluation.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => Rc::clone(summary_input), + other => panic!("expected DDSketch evaluation, got {other:?}"), }; assert!( Rc::ptr_eq(&producer(lhs), &producer(rhs)), - "the two quantile readouts should share one DDSketch producer" + "the two quantile evaluations should share one DDSketch producer" ); } #[test] -fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() { +fn planner_only_e2e_temporal_topk_preserves_query_update_and_evaluation_contract() { // Self-contained Planner E2E: each case starts from PromQL text and ends // at the post-ASAP summary DAG. No controller/backend types, // fixtures, configuration, or runtime are involved. @@ -677,17 +629,15 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() ), ]; for (source, expected_update, expected_family, excluded_labels) in cases { - let pre = Rc::new( - lower_promql( - source, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .expect("lower temporal Top-K"), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + source, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .expect("lower temporal Top-K"); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -697,29 +647,29 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() .replacements(&TargetSubDAG::new(&pre)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains(expected_family) => { + Replacement::SubDAG(node) if candidate.rationale.contains(expected_family) => { Some(node) } _ => None, }) .expect("heap-backed temporal Top-K candidate"); - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, query: SketchStatistic::TopK { k, .. }, - } = &candidate.expr + }) = candidate.asap() else { - panic!("expected Top-K estimate, got {:?}", candidate.expr) + panic!("expected Top-K estimate, got {:?}", candidate.operator) }; assert_eq!( *k, 5, "the requested Top-K cardinality must survive binding" ); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { input: state_input, family, child, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("expected structured Top-K state input") }; @@ -742,42 +692,41 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() )) ); assert_eq!(state_input.weight, expected_update); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } } /// Execute the ungrouped temporal TopK subset with exact state. This tests /// the emitted update contract, not sketch approximation or backend execution. -fn execute_topk_reference(plan: &SummaryNode) -> Vec<(String, f64)> { +fn execute_topk_reference(plan: &OperatorNode) -> Vec<(String, f64)> { use std::collections::BTreeMap; - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, query: SketchStatistic::TopK { k }, - } = &plan.expr + }) = plan.asap() else { - panic!("expected TopK readout") + panic!("expected TopK evaluation") }; - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { input, child, reduction, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("expected summary updates") }; assert_eq!(reduction, &Reduction::by(vec![])); - let SummaryExpr::KeepPreAsap(raw) = &child.expr else { - panic!("expected fused raw input") - }; - let QueryExpr::TimeRange { range, child } = raw.as_ref() else { + // The fused raw input is the kept non-ASAP sub-DAG itself. + assert!(!child.contains_asap(), "expected fused raw input"); + let Some(NonASAPOp::TimeRange { range, child, .. }) = child.non_asap() else { panic!("expected temporal input") }; - let QueryExpr::Scan { + let Some(NonASAPOp::Scan { source: asap_types::pre_asap::Source::TimeSeries { metric }, predicates, .. - } = child.as_ref() + }) = child.non_asap() else { panic!("expected metric scan") }; @@ -849,17 +798,15 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { vec![("worker", 100.0), ("cron", 30.0)], ), ] { - let pre = Rc::new( - lower_promql( - query, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + query, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -868,10 +815,10 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { // This reference executor consumes keyed heap updates. The inventory // also contains maintained exact values followed by sort/limit; those // have a different execution contract and must not enter this fixture. - let candidates: Vec<_> = strategy.replacements(&TargetSubDAG::new(&pre)).into_iter().filter(|candidate| matches!(&candidate.replacement, Replacement::Summary(plan) if matches!(plan.expr, SummaryExpr::SummaryEstimate { query: SketchStatistic::TopK { .. }, .. }))).collect(); + let candidates: Vec<_> = strategy.replacements(&TargetSubDAG::new(&pre)).into_iter().filter(|candidate| matches!(&candidate.replacement, Replacement::SubDAG(plan) if matches!(plan.asap(), Some(ASAPOp::SummaryEstimate { query: SketchStatistic::TopK { .. }, .. })))).collect(); assert!(!candidates.is_empty(), "no heap candidate for {query}"); for candidate in candidates { - let Replacement::Summary(plan) = candidate.replacement else { + let Replacement::SubDAG(plan) = candidate.replacement else { panic!("expected summary plan for {query}") }; let expected: Vec<_> = expected @@ -889,11 +836,11 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { /// SummaryEstimate { query: Quantile{0.99} } → {quantile_0_99: Float64} /// └─ SummaryAgg { Kll{k:269}, input: SampleValue } → {value: Sketch(Kll, {k:269})} /// └─ SummaryAgg { Rate, input: SampleValue } → {ts, value: ExactAggregate(Rate), …} -/// └─ KeepPreAsap(TimeRange{5m} → Scan) → {ts, value} +/// └─ TimeRange{5m} → Scan → {ts, value} /// ``` /// -/// The nested DAG exercises both realizations: the approximate quantile -/// binds a KLL sketch + readout; the per-series `rate` binds the exact +/// The nested tree exercises both realizations: the approximate quantile +/// binds a KLL sketch + evaluation; the per-series `rate` binds the exact /// counter-reset-aware accumulator (no estimate — its state is the value). #[test] fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { @@ -904,13 +851,13 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { .expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); - // Root: the sketch readout, back to a plain row shape. - let SummaryExpr::SummaryEstimate { + // Root: the sketch evaluation, back to a plain row shape. + let Some(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = root.asap() else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, SketchStatistic::Quantile { q } if *q == 0.99)); assert_eq!( @@ -924,15 +871,15 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { // reduction, one output row — not to be confused with the inner rate's // per-entity grouping below, even though both once collapsed to the // same empty `by: []` (issue #163). - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = summary_input.asap() else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -955,26 +902,24 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { ) ); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - } = &child.expr - else { - panic!("rate needs a maintenance readout"); + let Some(ASAPOp::FinalizeExactAccumulator { child }) = child.asap() else { + panic!("rate needs a maintenance evaluation"); }; // The rate: exact counter-reset-aware accumulator, per-series (labels // and time axis preserved), no estimate wrapper. `rate(...)` has no // grouping concept at all — every entity stays its own summary. - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child: leaf, family, reduction, .. - } = &child.expr + }) = child.asap() else { - panic!("expected inner SummaryAgg for rate, got {:?}", child.expr); + panic!( + "expected inner SummaryAgg for rate, got {:?}", + child.operator + ); }; assert_eq!( family, @@ -992,14 +937,20 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { ); // The leaf: unrewritten pass-through — TimeRange marker over the Scan. - let SummaryExpr::KeepPreAsap(kept_leaf) = &leaf.expr else { - panic!("expected KeepPreAsap leaf, got {:?}", leaf.expr); - }; - let QueryExpr::TimeRange { range, child: scan } = kept_leaf.as_ref() else { - panic!("expected TimeRange leaf, got {kept_leaf:?}"); + // The kept leaf is the non-ASAP sub-DAG itself. + assert!( + !leaf.contains_asap(), + "expected kept leaf, got {:?}", + leaf.operator + ); + let Some(NonASAPOp::TimeRange { + range, child: scan, .. + }) = leaf.non_asap() + else { + panic!("expected TimeRange leaf, got {:?}", leaf.operator); }; assert_eq!(range.as_secs(), 300); - assert!(matches!(scan.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(scan.non_asap(), Some(NonASAPOp::Scan { .. }))); assert!( leaf.schema .fields @@ -1017,11 +968,11 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { let pre_asap = lower_promql("sum by (job) (http_requests_total)", AccuracyTarget::Exact) .expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { family, reduction, .. - } = &root.expr + }) = root.asap() else { - panic!("expected SummaryAgg, got {:?}", root.expr); + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, @@ -1042,21 +993,19 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { lower_promql("avg(http_requests_total)", AccuracyTarget::Exact).expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + !root.contains_asap(), "avg has no mergeable accumulator — stays logical" ); } #[test] fn promql_sum_of_count_over_time_is_composed_by_default_search() { - let original = Rc::new( - lower_promql( - "sum by (service) (count_over_time(metrics[5m]))", - AccuracyTarget::Exact, - ) - .expect("lowering failed"), - ); - let original_schema = original.output_schema().unwrap(); + let original = lower_promql( + "sum by (service) (count_over_time(metrics[5m]))", + AccuracyTarget::Exact, + ) + .expect("lowering failed"); + let original_schema = original.schema.clone(); let space = search_workload(vec![("query", original)]); let root = &space.roots[0].1; let group = space.candidates_for_target(root).expect("root memo group"); @@ -1065,20 +1014,21 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { .iter() .find(|candidate| candidate.strategy == "SemanticEquivalentRewriteStrategy") .expect("default search should compose the lowered PromQL query"); - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDAG(rewritten) = &candidate.replacement else { panic!("expected logical rewrite") }; + assert!(is_logical_rewrite(rewritten), "expected logical rewrite"); - assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Project { child, .. } = rewritten.as_ref() else { + assert_eq!(rewritten.schema, original_schema); + let Some(NonASAPOp::Project { child, .. }) = rewritten.non_asap() else { panic!("sum(count_over_time) needs a Float64 cast Project") }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, child, .. - } = child.as_ref() + }) = child.non_asap() else { panic!("expected one composed aggregate") }; @@ -1090,9 +1040,9 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { }] )); assert!(matches!( - child.as_ref(), - QueryExpr::TimeRange { range, child } - if range.as_secs() == 300 && matches!(child.as_ref(), QueryExpr::Scan { .. }) + child.non_asap(), + Some(NonASAPOp::TimeRange { range, child, .. }) + if range.as_secs() == 300 && matches!(child.non_asap(), Some(NonASAPOp::Scan { .. })) )); } @@ -1100,50 +1050,41 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { // Real workload selection must expose the state-to-value edge; an outer // sketch must not interpret exact accumulator bytes as input samples. - let pre = Rc::new( - lower_promql( - "quantile(0.9, sum_over_time(m[1m]))", - AccuracyTarget::Epsilon(0.05), - ) - .unwrap(), - ); + let pre = lower_promql( + "quantile(0.9, sum_over_time(m[1m]))", + AccuracyTarget::Epsilon(0.05), + ) + .unwrap(); let space = search_workload(vec![("query", pre)]); let selected = space.global_selection(&DefaultCostModel); let plan = selected .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &plan.expr else { + // Stored timings are gone: time the plan and read the timed copy. + let timed_plan = timed(&plan); + let Some(ASAPOp::SummaryEstimate { summary_input, .. }) = timed_plan.asap() else { panic!("expected selected quantile summary"); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Some(ASAPOp::SummaryAgg { child, .. }) = summary_input.asap() else { panic!("expected maintained outer summary"); }; - let SummaryExpr::ValueOperation { - child: source, - operation, - timing, - } = &child.expr - else { + let Some(ASAPOp::FinalizeExactAccumulator { child: source }) = child.asap() else { panic!( "missing explicit accumulator finalization: {:?}", - child.expr + child.operator ); }; - assert!(matches!( - operation, - ValueOperation::FinalizeExactAccumulator - )); assert_eq!( - *timing, - asap_types::post_asap::ExecutionTiming::IngestionTime + child.timing, + Some(asap_types::post_asap::ExecutionTiming::IngestionTime) ); assert!(matches!( - source.expr, - SummaryExpr::SummaryAgg { + source.asap(), + Some(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. - } + }) )); assert!(child .schema @@ -1155,12 +1096,13 @@ fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { .fields .iter() .any(|field| matches!(field.dtype, FieldDataType::Plain(DataType::Float64)))); - compile_post_asap_dag(&plan).expect("explicit boundary is a valid post-ASAP DAG"); + // Explicit boundary is a valid post-ASAP DAG. + post_asap_dag(&plan); } #[test] fn physical_node_owns_phase_independently_of_binary_payload() { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; + use asap_types::post_asap::ExecutionTiming; for (query, expected) in [ ( // One selector: both operands cover the same series. @@ -1173,27 +1115,30 @@ fn physical_node_owns_phase_independently_of_binary_payload() { ), ] { let input = lower_promql(query, AccuracyTarget::Epsilon(0.05)).unwrap(); - // Backend lowering carries opaque series identity before candidate export. - let input = asap_types::pre_asap::schema::with_promql_series_identity(&input).unwrap(); - let search = search_workload(vec![("q", Rc::new(input))]); + let search = search_workload(vec![("q", input)]); let choice = search.global_selection(&DefaultCostModel); let plan = choice .assemble_selected_dag(&search.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); let node = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Binary { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::BinaryOp { .. }) + ) + }) .unwrap(); assert_eq!(node.output_state.timing, expected); let wire = serde_json::to_value(&node.payload).unwrap(); assert!(wire.get("timing").is_none()); let mut obsolete = wire.clone(); obsolete["timing"] = serde_json::json!(expected.as_str()); - assert!(serde_json::from_value::(obsolete).is_err()); - let restored: PostAsapOperatorPayload = serde_json::from_value(wire).unwrap(); + assert!(serde_json::from_value::(obsolete).is_err()); + let restored: PhysicalASAPOperatorPayload = serde_json::from_value(wire).unwrap(); assert_eq!(restored, node.payload); } } @@ -1207,14 +1152,10 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { ) .unwrap(); let root = realize(&pre).unwrap(); - assert!(matches!(root.expr, SummaryExpr::BinaryOp { .. })); + assert!(matches!(root.non_asap(), Some(NonASAPOp::BinaryOp { .. }))); assert!(root.guarantee.is_none()); let space = search_workload_with_targets( - vec![( - "unproven", - Rc::new(pre), - Some(AccuracyTarget::Epsilon(0.01)), - )], + vec![("unproven", pre, Some(AccuracyTarget::Epsilon(0.01)))], &asap_aware_mapping::default_strategies(), &DefaultAccuracyModel, ); @@ -1226,8 +1167,8 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { root_group.candidates.iter().any(|candidate| { matches!( &candidate.replacement, - Replacement::Summary(node) - if matches!(node.expr, SummaryExpr::BinaryOp { .. }) + Replacement::SubDAG(node) + if matches!(node.non_asap(), Some(NonASAPOp::BinaryOp { .. })) && node.guarantee.is_none() ) }), @@ -1247,7 +1188,7 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { .assemble_selected_dag(&space.roots[0].1) .unwrap() .expect("materialized root"); - assert!(matches!(materialized.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!materialized.contains_asap()); } struct FixtureQuantileDomain { @@ -1255,7 +1196,7 @@ struct FixtureQuantileDomain { upper: f64, } impl AccuracyEvidenceProvider for FixtureQuantileDomain { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &OperatorNode) -> Option { Some(QuantileInputDomain { lower: self.lower, upper: self.upper, @@ -1278,14 +1219,12 @@ fn ddsketch_ratio_rejects_unsafe_domains() { (f64::MIN_POSITIVE / 2., f64::MIN_POSITIVE / 2.), ] { let evidence = FixtureQuantileDomain { lower, upper }; - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -1304,8 +1243,8 @@ fn ddsketch_ratio_rejects_unsafe_domains() { fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { struct PartialUnsafeDomain; impl AccuracyEvidenceProvider for PartialUnsafeDomain { - fn quantile_input_domain(&self, operand: &QueryExpr) -> Option { - let QueryExpr::Aggregate { measures, .. } = operand else { + fn quantile_input_domain(&self, operand: &OperatorNode) -> Option { + let Some(NonASAPOp::Aggregate { measures, .. }) = operand.non_asap() else { return None; }; matches!( @@ -1321,14 +1260,12 @@ fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { } } - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -1339,40 +1276,38 @@ fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { /// The committed planner alpha is exercised against the pinned sketch implementation. #[test] -fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { +fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_evaluations() { for sign in [-1., 1.] { let evidence = FixtureQuantileDomain { lower: if sign < 0. { -100. } else { 1. }, upper: if sign < 0. { -1. } else { 100. }, }; - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, ); let candidates = strategy.replacements(&TargetSubDAG::new(&pre)); - let Replacement::Summary(node) = &candidates[0].replacement else { + let Replacement::SubDAG(node) = &candidates[0].replacement else { panic!("summary") }; - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &node.expr else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = node.non_asap() else { panic!("ratio") }; - let alpha = |node: &SummaryNode| { - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { - panic!("readout") + let alpha = |node: &OperatorNode| { + let Some(ASAPOp::SummaryEstimate { summary_input, .. }) = node.asap() else { + panic!("evaluation") }; - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("sketch") }; @@ -1407,12 +1342,12 @@ fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { } } -/// Empty or overlarge population contracts cannot promise a supported readout. +/// Empty or overlarge population contracts cannot promise a supported evaluation. #[test] fn ddsketch_ratio_requires_a_supported_population_size() { struct PopulationEvidence(u64); impl AccuracyEvidenceProvider for PopulationEvidence { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &OperatorNode) -> Option { Some(QuantileInputDomain { lower: 1., upper: 10., @@ -1421,16 +1356,14 @@ fn ddsketch_ratio_requires_a_supported_population_size() { }) } } - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); for count in [0, (1u64 << 53) + 1] { let evidence = PopulationEvidence(count); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, @@ -1441,7 +1374,7 @@ fn ddsketch_ratio_requires_a_supported_population_size() { } // Every `without` aggregation candidate exports a valid DAG: its summary state -// column carries the family instead of the readout's Float64 value. +// column carries the family instead of the evaluation's Float64 value. #[test] fn without_aggregation_candidates_export_valid_dags() { for accuracy in [ @@ -1452,7 +1385,7 @@ fn without_aggregation_candidates_export_valid_dags() { }, ] { for query in ["sum without (pod) (m)", "quantile without (pod) (0.5, m)"] { - let root = Rc::new(lower_promql(query, accuracy.clone()).unwrap()); + let root = lower_promql(query, accuracy.clone()).unwrap(); let space = search_workload_with_targets( vec![(0, root, Some(accuracy.clone()))], &asap_aware_mapping::default_strategies(), @@ -1461,7 +1394,7 @@ fn without_aggregation_candidates_export_valid_dags() { let inventory = space.enumerate_candidate_dags_for_root(&0, 65_536).unwrap(); assert!(!inventory.candidates.is_empty(), "{query}"); for (_, node) in inventory.candidates.iter().flatten() { - compile_post_asap_dag(node).unwrap_or_else(|e| panic!("{query}: {e}")); + post_asap_dag(node); } } } diff --git a/crates/integration-tests/tests/scan.rs b/crates/integration-tests/tests/scan.rs index bb7988d7a..e43e27a6c 100644 --- a/crates/integration-tests/tests/scan.rs +++ b/crates/integration-tests/tests/scan.rs @@ -1,4 +1,4 @@ -//! `QueryExpr::Scan` — label matcher / predicate tests. +//! `NonASAPOp::Scan` — label matcher / predicate tests. //! //! The Scan schema is always [ts(0), value(1), label_a(2), label_b(3), …] //! where labels are appended alphabetically after dedup by the SchemaResolver. @@ -11,60 +11,62 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{CompareOpKind, Predicate, QueryExpr, ScalarValue, Source}; +use asap_types::ir::{ + ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, TimeRangeKind, +}; +use asap_types::pre_asap::{CompareOpKind, ScalarValue, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn bare_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn bare_scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn instant(child: QueryExpr) -> QueryExpr { - QueryExpr::TimeRange { +fn instant(child: Rc) -> Rc { + node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(child), - } + kind: TimeRangeKind::Instant, + child, + }) +} + +fn label_pred(col_id: usize, op: CompareOpKind, value: &str) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col_id)), + op, + right: Box::new(ScalarExpr::Literal(ScalarValue::Utf8(value.into()))), + semantics: ExprSemantics::Promql, + }) } fn eq_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), - })) + label_pred(col_id, CompareOpKind::Eq, value) } fn ne_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Ne, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), - })) + label_pred(col_id, CompareOpKind::Ne, value) } fn regex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Regex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), - })) + label_pred(col_id, CompareOpKind::Regex, pattern) } fn notregex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::NotRegex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), - })) + label_pred(col_id, CompareOpKind::NotRegex, pattern) } // #1 — bare metric name, no matchers @@ -80,13 +82,13 @@ fn q01_bare_scan() { // schema: [ts(0), value(1), job(2)] #[test] fn q02_equality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![eq_pred(2, "api-server")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job="api-server"}"#), expected); } @@ -94,13 +96,13 @@ fn q02_equality_predicate() { // schema: [ts(0), value(1), status(2)] #[test] fn q03_inequality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![ne_pred(2, "500")], schema: metric_schema(&["status"]), - }); + })); assert_eq!(lower(r#"http_requests_total{status!="500"}"#), expected); } @@ -108,13 +110,13 @@ fn q03_inequality_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q04_regex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![regex_pred(2, "api.*")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job=~"api.*"}"#), expected); } @@ -122,13 +124,13 @@ fn q04_regex_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q_notregex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![notregex_pred(2, "internal.*")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job!~"internal.*"}"#), expected); } @@ -137,13 +139,13 @@ fn q_notregex_predicate() { // predicates in same alphabetical order: job first, then status #[test] fn q_multi_two_predicates() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![eq_pred(2, "api-server"), ne_pred(3, "500")], schema: metric_schema(&["job", "status"]), - }); + })); assert_eq!( lower(r#"http_requests_total{job="api-server",status!="500"}"#), expected, diff --git a/crates/integration-tests/tests/schema.rs b/crates/integration-tests/tests/schema.rs index 08b925dd1..c616c0d5f 100644 --- a/crates/integration-tests/tests/schema.rs +++ b/crates/integration-tests/tests/schema.rs @@ -1,7 +1,7 @@ //! `Schema::closed` propagation — open/closed invariant tests. //! -//! Verifies that `QueryExpr::output_schema()` propagates the open/closed -//! completeness flag correctly through a lowered query DAG. +//! Verifies that the derived `OperatorNode::schema` propagates the open/closed +//! completeness flag correctly through a lowered query tree. //! //! Key invariant: a PromQL scan is always `closed: false` (open) because its //! label set is runtime-only. The schema freezes to `closed: true` exactly at @@ -12,14 +12,14 @@ use asap_integration_tests::fixtures::lower_promql; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> asap_types::pre_asap::QueryExpr { +fn lower(q: &str) -> std::rc::Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } // bare scan is open — the metric's full label set is unknown at plan time #[test] fn schema_bare_scan_is_open() { - let s = lower("http_requests_total").output_schema().unwrap(); + let s = lower("http_requests_total").schema.clone(); assert!(!s.closed, "PromQL scan must be open"); } @@ -27,8 +27,8 @@ fn schema_bare_scan_is_open() { #[test] fn schema_filtered_scan_is_open() { let s = lower(r#"http_requests_total{job="api-server"}"#) - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "PromQL scan with predicates must remain open"); assert_eq!(s.fields.len(), 3, "[ts, value, job]"); } @@ -36,9 +36,7 @@ fn schema_filtered_scan_is_open() { // per-series rate is label-preserving → output stays open #[test] fn schema_rate_stays_open() { - let s = lower("rate(http_requests_total[5m])") - .output_schema() - .unwrap(); + let s = lower("rate(http_requests_total[5m])").schema.clone(); assert!(!s.closed, "per-series rate is label-preserving; stays open"); } @@ -46,24 +44,22 @@ fn schema_rate_stays_open() { #[test] fn schema_count_over_time_stays_open() { let s = lower("count_over_time(http_requests_total[5m])") - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "per-series count_over_time stays open"); } // cross-series sum with no group keys freezes to closed #[test] fn schema_sum_freezes_to_closed() { - let s = lower("sum(http_requests_total)").output_schema().unwrap(); + let s = lower("sum(http_requests_total)").schema.clone(); assert!(s.closed, "cross-series aggregate must freeze to closed"); } // cross-series sum grouped by job also freezes to closed #[test] fn schema_sum_by_job_freezes_to_closed() { - let s = lower("sum by (job) (http_requests_total)") - .output_schema() - .unwrap(); + let s = lower("sum by (job) (http_requests_total)").schema.clone(); assert!( s.closed, "grouped cross-series aggregate must freeze to closed" @@ -74,8 +70,8 @@ fn schema_sum_by_job_freezes_to_closed() { #[test] fn schema_sum_over_rate_freezes_to_closed() { let s = lower("sum by (job) (rate(http_requests_total[5m]))") - .output_schema() - .unwrap(); + .schema + .clone(); assert!( s.closed, "cross-series aggregate over rate must freeze to closed" @@ -86,8 +82,8 @@ fn schema_sum_over_rate_freezes_to_closed() { #[test] fn schema_binary_op_two_open_stays_open() { let s = lower("http_requests_total / http_errors_total") - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "binary op over two open scans must stay open"); } @@ -95,8 +91,8 @@ fn schema_binary_op_two_open_stays_open() { #[test] fn schema_binary_op_two_closed_is_closed() { let s = lower("sum by (job) (http_requests_total) / sum by (job) (http_errors_total)") - .output_schema() - .unwrap(); + .schema + .clone(); assert!( s.closed, "binary op over two closed aggregates must be closed" diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs index dd4426c22..ed9c52fe9 100644 --- a/crates/integration-tests/tests/sql_to_physical.rs +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -1,4 +1,5 @@ //! SQL frontend, candidate selection, physical compilation and fresh-run execution. +mod physical_common; use asap_aware_mapping::{search_workload, DefaultCostModel}; use asap_frontend_sql::{lower_sql, SqlCatalog}; use asap_physical_operators::{ @@ -7,13 +8,15 @@ use asap_physical_operators::{ sources::{DataSources, MemorySource}, values::{Batch, Value}, }; +use asap_types::ir::physical_export::PhysicalASAPOperatorPayload; use asap_types::{ - post_asap::{compile_post_asap_dag, FieldDataType, PostAsapOperatorPayload}, - pre_asap::{DataType, Field, QueryExpr, Schema}, + post_asap::FieldDataType, + pre_asap::{DataType, Field, Schema}, types::AccuracyTarget, }; use futures::StreamExt; -use std::{collections::BTreeMap, rc::Rc, sync::Arc}; +use physical_common::compile_physical_asap_dag; +use std::{collections::BTreeMap, sync::Arc}; /// SQL filtering and grouped aggregation survive logical/physical lowering; /// rebinding the compiled DAG runs against new data rather than cached results. @@ -30,27 +33,23 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { "SELECT service, SUM(value) AS total FROM metrics WHERE value > 1 GROUP BY service", "SELECT service, SUM(value) AS total FROM metrics GROUP BY service", ] { - let logical = Rc::new( - lower_sql(query, &catalog, AccuracyTarget::Exact) - .await - .unwrap(), - ); + let logical = lower_sql(query, &catalog, AccuracyTarget::Exact) + .await + .unwrap(); let space = search_workload(vec![("sql", logical)]); let selected = space .global_selection(&DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let dag = compile_physical_asap_dag(&selected).unwrap(); let scan = dag .nodes .iter() .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Scan { .. } - } + PhysicalASAPOperatorPayload::NonASAP(asap_types::ir::NonASAPOp::Scan { .. }) ) }) .expect("raw SQL scan"); @@ -61,8 +60,8 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); let plan = compile( &dag, - BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], + BTreeMap::from([(scan.id as u64, InputContract::bounded(schema.clone()))]), + &[dag.roots[0] as u64], ) .unwrap(); for multiplier in [1., 2.] { @@ -88,12 +87,15 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .collect() }) .collect(); - let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { - unreachable!() - }; - let QueryExpr::Scan { source, .. } = expression else { + let PhysicalASAPOperatorPayload::NonASAP(asap_types::ir::NonASAPOp::Scan { + source, + predicates: _, + schema: _scan_schema, + }) = &scan.payload + else { unreachable!() }; + let expression = asap_types::ir::OperatorNode::reachable(&selected).into_iter().find(|n| matches!(n.non_asap(), Some(asap_types::ir::NonASAPOp::Scan { source: s, .. }) if s == source)).unwrap(); let mut sources = DataSources::default(); sources .register( @@ -109,8 +111,8 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .unwrap(); let bound = plan .instantiate(BTreeMap::from([( - u64::from(scan.id.0), - Box::new(sources.bind(expression).unwrap()) as Source<'_>, + scan.id as u64, + Box::new(sources.bind(&expression).unwrap()) as Source<'_>, )])) .unwrap(); let mut stream = bound diff --git a/crates/integration-tests/tests/sql_to_post_asap.rs b/crates/integration-tests/tests/sql_to_post_asap.rs index 4cecd6952..3121e99cb 100644 --- a/crates/integration-tests/tests/sql_to_post_asap.rs +++ b/crates/integration-tests/tests/sql_to_post_asap.rs @@ -1,61 +1,90 @@ //! End-to-end SQL query-string → post-ASAP IR pin (issue #191). //! //! The SQL counterpart of `promql_to_post_asap.rs`: drives SQL text — -//! `lower_sql` (text → pre-ASAP `QueryExpr`) → -//! `SketchAlgorithmStrategy::replacements` (pre-ASAP → post-ASAP -//! `SummaryExpr`, see [`realize`] below) — and pins the resulting -//! sketch-vs-exact-accumulator shape node by node, the way -//! `promql_to_post_asap.rs` does for PromQL. +//! `lower_sql` (text → non-ASAP `OperatorNode` tree) → +//! `ASAPStrategies::replacements` (→ a tree with ASAP operators, +//! see [`realize`] below) — and pins the resulting sketch-vs-exact-accumulator +//! shape node by node, the way `promql_to_post_asap.rs` does for PromQL. //! //! ## A structural wrinkle PromQL doesn't have //! -//! `lower_promql` returns a *bare* `QueryExpr::Aggregate` for a top-level +//! `lower_promql` returns a *bare* `NonASAPOp::Aggregate` for a top-level //! aggregation (`sum by (job) (m)`, `quantile(0.99, …)`), so [`realize`] can //! bind it directly at the DAG root. `lower_sql` never does: DataFusion's //! planner always wraps even a single, unaliased aggregate in an identity //! `Project` (confirmed below), so a SQL DAG's *root* is normally `Project { //! child: Aggregate { .. } }`. Final materialization retains that projection -//! as a query-time value operation and independently plans its child, keeping +//! as a query-time non-ASAP node and independently plans its child, keeping //! both SELECT-list semantics and the summary-bound aggregate visible. use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; +use asap_aware_mapping::replacement::{retain_exact, RealizationError}; use asap_aware_mapping::{ - search_workload, DefaultCostModel, Replacement, ReplacementStrategy, ReplacementSubDAG, - SketchAlgorithmStrategy, TargetSubDAG, + search_workload, ASAPStrategies, DefaultCostModel, Replacement, ReplacementStrategy, + ReplacementSubDAG, TargetSubDAG, }; use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_types::ir::operator_properties::Reduction; +use asap_types::ir::physical_export::{EdgeRole, PhysicalASAPNodeId, PhysicalASAPOperatorPayload}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, Predicate, ScalarExpr}; use asap_types::post_asap::{ - compile_post_asap_dag, EdgeRole, ExactKind, ExactParams, FieldDataType, GroupingStrategy, - PostAsapOperatorPayload, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, - SummaryExpr, SummaryNode, SummaryUpdate, ValueOperation, + ExactKind, ExactParams, FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, + SketchParams, SketchStatistic, SummaryUpdate, }; use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -/// This crate has no "bind me one DAG" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// This crate has no "bind me one tree" public API any more — +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the -/// take-the-first-(`cost_model`-preferred)-candidate pattern so the +/// take-the-first-(`cost_model`-preferred)-summary-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() - .replacements(&target) +fn realize(target: &Rc) -> Result, RealizationError> { + let target_dag = TargetSubDAG::new(target); + match ASAPStrategies::default_cost_model() + .replacements(&target_dag) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. - }) => Ok(node), - _ => keep_pre_asap(&root), + }) if node.contains_asap() => Ok(node), + _ => retain_exact(target), } + .inspect(|node| { + node.validate_structure() + .expect("planned dag satisfies the unified IR contract") + }) +} + +/// The single input of a unary non-ASAP node (Project, Filter, Sort, ...) or +/// of a `FinalizeExactAccumulator`; `None` for anything else. +fn unary_child(node: &OperatorNode) -> Option<&Rc> { + match &node.operator { + Operator::NonASAP(op) => match op.children().as_slice() { + [child] => Some(*child), + _ => None, + }, + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => Some(child), + Operator::ASAP(_) => None, + } +} + +/// A sub-DAG kept as plain (non-ASAP) work: no ASAP operator anywhere below. +fn is_kept_non_asap(node: &OperatorNode) -> bool { + node.non_asap().is_some() && !node.contains_asap() +} + +/// Mirror a scalar-only predicate (no operator references) to its wire form. +fn wire_pred(pred: &Predicate) -> Predicate { + Predicate(pred.0.map_operator_refs(&mut |_| -> PhysicalASAPNodeId { + panic!("fixture predicate references no operator") + })) } fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { @@ -89,7 +118,7 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str, accuracy: AccuracyTarget) -> QueryExpr { +async fn lower(sql: &str, accuracy: AccuracyTarget) -> Rc { lower_sql(sql, &catalog(), accuracy) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) @@ -121,26 +150,27 @@ async fn clickhouse_temporal_sql_reuses_rate_and_increase_physical_summaries() { .expect("explicit temporal SQL must lower"); let physical = realize(inner_aggregate(&pre_asap)).expect("temporal reducer must be planned"); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, reduction, child, .. - } = &physical.expr + }) = &physical.operator else { - panic!("expected a shared SummaryAgg, got {:?}", physical.expr); + panic!("expected a shared SummaryAgg, got {:?}", physical.operator); }; assert_eq!(family, &expected); assert_eq!(reduction, &Reduction::PerEntity); - let SummaryExpr::KeepPreAsap(raw) = &child.expr else { - panic!( - "expected a retained temporal SQL input, got {:?}", - child.expr - ); - }; - assert!(matches!(raw.as_ref(), QueryExpr::TimeRange { range, child } + assert!( + is_kept_non_asap(child), + "expected a retained temporal SQL input, got {:?}", + child.operator + ); + assert!( + matches!(child.non_asap(), Some(NonASAPOp::TimeRange { range, child, .. }) if *range == std::time::Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Project { .. }))); + && matches!(child.non_asap(), Some(NonASAPOp::Project { .. }))) + ); } } @@ -156,16 +186,14 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { SELECT service, {function}(latency, ts, {window_ms}) AS v \ FROM metrics GROUP BY service)" ); - let pre_asap = Rc::new( - lower_sql_dialect( - &sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect("nested temporal SQL must lower"), - ); + let pre_asap = lower_sql_dialect( + &sql, + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .expect("nested temporal SQL must lower"); let space = search_workload(vec![("nested", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection @@ -173,42 +201,40 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { .expect("materialization failed") .expect("root must be discovered"); - fn has_temporal_summary(node: &SummaryNode) -> bool { - match &node.expr { - SummaryExpr::SummaryAgg { + fn has_temporal_summary(node: &OperatorNode) -> bool { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate | ExactKind::Increase, _), .. - } => true, - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryEstimate { - summary_input: child, - .. - } => has_temporal_summary(child), - _ => false, + }) => true, + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + has_temporal_summary(summary_input) + } + _ => unary_child(node).is_some_and(|child| has_temporal_summary(child)), } } assert!( has_temporal_summary(&root), "inner {function} was hidden: {root:?}" ); - let dag = compile_post_asap_dag(&root).expect("nested SQL DAG must compile"); + let dag = post_asap_dag(&root); assert!(dag.nodes.iter().any(|node| matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Exact(_), - .. - } + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Aggregate { .. }) ))); } } /// The `Aggregate` node beneath the identity `Project` DataFusion's planner /// always wraps a top-level aggregate in — see the module docs above. -fn inner_aggregate(qe: &QueryExpr) -> &QueryExpr { - match qe { - QueryExpr::Project { child, .. } => inner_aggregate(child), - QueryExpr::Aggregate { .. } => qe, - other => panic!("expected a Project{{Aggregate}} shape, got {other:?}"), +fn inner_aggregate(node: &Rc) -> &Rc { + match node.non_asap() { + Some(NonASAPOp::Project { child, .. }) => inner_aggregate(child), + Some(NonASAPOp::Aggregate { .. }) => node, + _ => panic!( + "expected a Project{{Aggregate}} shape, got {:?}", + node.operator + ), } } @@ -221,32 +247,27 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { AccuracyTarget::Epsilon(0.01), ) .await; - assert!( - matches!(pre_asap, QueryExpr::Project { .. }), - "sanity: a SQL root is a Project, unlike lower_promql's bare Aggregate" - ); - let pre_asap = Rc::new(pre_asap); + let Some(NonASAPOp::Project { + cols: expected_cols, + qualifier: expected_qualifier, + .. + }) = pre_asap.non_asap() + else { + panic!("sanity: a SQL root is a Project, unlike lower_promql's bare Aggregate"); + }; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let QueryExpr::Project { - cols: expected_cols, - qualifier: expected_qualifier, - .. - } = pre_asap.as_ref() - else { - unreachable!() - }; - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Project { child, - operation: asap_types::post_asap::ValueOperation::Project { cols, qualifier }, - .. - } = &root.expr + cols, + qualifier, + }) = root.non_asap() else { - panic!("expected retained Project root, got {:?}", root.expr); + panic!("expected retained Project root, got {:?}", root.operator); }; assert_eq!(cols, expected_cols, "projection expressions and aliases"); assert_eq!(qualifier, expected_qualifier, "projection qualifier"); @@ -256,7 +277,10 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { FieldDataType::Plain(DataType::Float64) ); assert!( - matches!(child.expr, SummaryExpr::SummaryEstimate { .. }), + matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ), "the Aggregate under Project must be summary-bound" ); } @@ -265,63 +289,62 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { /// aggregates are independently selected as physical summaries. #[tokio::test] async fn sql_join_recursively_binds_both_temporal_aggregate_children() { - let pre_asap = Rc::new( - lower_sql_dialect( - "SELECT a.service, a.v / b.v AS ratio FROM \ - (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='errors' GROUP BY service) a \ - INNER JOIN \ - (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='requests' GROUP BY service) b \ - ON b.service=a.service", - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect("two-subquery rate ratio must lower"), - ); + let pre_asap = lower_sql_dialect( + "SELECT a.service, a.v / b.v AS ratio FROM \ + (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='errors' GROUP BY service) a \ + INNER JOIN \ + (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='requests' GROUP BY service) b \ + ON b.service=a.service", + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .expect("two-subquery rate ratio must lower"); let space = search_workload(vec![("ratio", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let SummaryExpr::ValueOperation { - child: join, - operation: ValueOperation::Project { cols, .. }, - .. - } = &root.expr + let Some(NonASAPOp::Project { + child: join, cols, .. + }) = root.non_asap() else { panic!( "expected Project above relational join, got {:?}", - root.expr + root.operator ); }; assert!(matches!( &cols[1].expr, - QueryExpr::Arithmetic { + ScalarExpr::Arithmetic { op: asap_types::pre_asap::ArithmeticOpKind::Div, .. } )); - let SummaryExpr::RelationalJoin { + let Some(NonASAPOp::Join { left, right, kind, pred, - pruning: None, - } = &join.expr + }) = join.non_asap() else { - panic!("expected read-time relational join, got {:?}", join.expr); + panic!( + "expected read-time relational join, got {:?}", + join.operator + ); }; assert_eq!(kind, &asap_types::pre_asap::JoinKind::Inner); assert!(matches!( - pred.0.as_ref(), - QueryExpr::Compare { + &pred.0, + ScalarExpr::Compare { left, op: asap_types::pre_asap::CompareOpKind::Eq, right, - } if matches!(left.as_ref(), QueryExpr::Column(0)) - && matches!(right.as_ref(), QueryExpr::Column(2)) + .. + } if matches!(left.as_ref(), ScalarExpr::Column(0)) + && matches!(right.as_ref(), ScalarExpr::Column(2)) )); assert_eq!( join.schema @@ -332,39 +355,42 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { vec!["service", "v", "service", "v"] ); for child in [left, right] { - let SummaryExpr::ValueOperation { - child: aggregate, - operation: ValueOperation::Project { .. }, - .. - } = &child.expr + let Some(NonASAPOp::Project { + child: aggregate, .. + }) = child.non_asap() else { - panic!("derived table Project was not retained: {:?}", child.expr); + panic!( + "derived table Project was not retained: {:?}", + child.operator + ); }; - let SummaryExpr::ValueOperation { - child: aggregate, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } = &aggregate.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: aggregate }) = + &aggregate.operator else { panic!("derived table Project must consume finalized exact values"); }; assert!(matches!( - aggregate.expr, - SummaryExpr::SummaryAgg { + aggregate.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), .. - } + }) )); } assert!(join .guarantee .as_ref() .is_some_and(|value| value.is_exact())); - let dag = compile_post_asap_dag(&root).expect("join DAG must compile"); + let dag = post_asap_dag(&root); let join_id = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::RelationalJoin { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Join { .. }) + ) + }) .expect("relational join node") .id; let roles = dag @@ -383,29 +409,27 @@ async fn unsupported_sql_join_shapes_remain_fail_closed() { "SELECT a.service FROM (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) a INNER JOIN (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) b ON a.v>b.v", "SELECT a.service FROM (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) a INNER JOIN (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) b ON a.service=a.service", ] { - let pre_asap = Rc::new( - lower_sql_dialect( - sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap_or_else(|error| panic!("join must lower before fail-closed mapping: {error}")), - ); + let pre_asap = lower_sql_dialect( + sql, + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .unwrap_or_else(|error| panic!("join must lower before fail-closed mapping: {error}")); let space = search_workload(vec![("unsupported-join", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let SummaryExpr::ValueOperation { child, .. } = &root.expr else { - panic!("SQL projection must remain explicit: {:?}", root.expr); + let Some(NonASAPOp::Project { child, .. }) = root.non_asap() else { + panic!("SQL projection must remain explicit: {:?}", root.operator); }; assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + is_kept_non_asap(child), "unsupported join was partially accelerated: {:?}", - child.expr + child.operator ); } } @@ -414,16 +438,14 @@ async fn unsupported_sql_join_shapes_remain_fail_closed() { /// explicit read-time nodes while the aggregate is summary-bound. #[tokio::test] async fn sql_relational_parents_retain_summary_bound_aggregate() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.p FROM \ - (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ - FROM metrics GROUP BY service) t \ - WHERE t.p > 100 ORDER BY t.p DESC LIMIT 5", - AccuracyTarget::Epsilon(0.01), - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.p FROM \ + (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ + FROM metrics GROUP BY service) t \ + WHERE t.p > 100 ORDER BY t.p DESC LIMIT 5", + AccuracyTarget::Epsilon(0.01), + ) + .await; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection @@ -437,28 +459,29 @@ async fn sql_relational_parents_retain_summary_bound_aggregate() { let mut saw_sort = false; let mut saw_limit = false; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, operation, .. - } => { - match operation { - asap_types::post_asap::ValueOperation::Project { .. } => saw_project = true, - asap_types::post_asap::ValueOperation::Filter { .. } => saw_filter = true, - asap_types::post_asap::ValueOperation::Sort { .. } => saw_sort = true, - asap_types::post_asap::ValueOperation::Limit { n, offset, .. } => { - assert_eq!((*n, *offset), (5, 0)); - saw_limit = true; - } - _ => {} - } - node = child; - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - assert!(matches!(summary_input.expr, SummaryExpr::SummaryAgg { .. })); - break; + if let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator { + assert!(matches!( + summary_input.operator, + Operator::ASAP(ASAPOp::SummaryAgg { .. }) + )); + break; + } + match node.non_asap() { + Some(NonASAPOp::Project { .. }) => saw_project = true, + Some(NonASAPOp::Filter { .. }) => saw_filter = true, + Some(NonASAPOp::Sort { .. }) => saw_sort = true, + Some(NonASAPOp::Limit { n, offset, .. }) => { + assert_eq!((*n, *offset), (Some(5), 0)); + saw_limit = true; } - other => panic!("expected relational parents over SummaryEstimate, got {other:?}"), + _ => {} } + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected relational parents over SummaryEstimate, got {:?}", + node.operator + ) + }); } assert!(saw_project && saw_filter && saw_sort && saw_limit); } @@ -468,39 +491,47 @@ async fn sql_relational_parents_retain_summary_bound_aggregate() { /// may be dropped or moved across the aggregation boundary. #[tokio::test] async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.p FROM \ - (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ - FROM metrics WHERE service = 'api' GROUP BY service) t \ - WHERE t.p > 100", - AccuracyTarget::Epsilon(0.01), - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.p FROM \ + (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ + FROM metrics WHERE service = 'api' GROUP BY service) t \ + WHERE t.p > 100", + AccuracyTarget::Epsilon(0.01), + ) + .await; let expected_read_predicate = { - let mut node = pre_asap.as_ref(); + let mut node = &pre_asap; loop { - match node { - QueryExpr::Filter { pred, .. } => break pred.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, - other => panic!("expected a Filter above the aggregate, got {other:?}"), + match node.non_asap() { + Some(NonASAPOp::Filter { pred, .. }) => break pred.clone(), + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => node = child, + _ => panic!( + "expected a Filter above the aggregate, got {:?}", + node.operator + ), } } }; let expected_source_predicates = { - let mut node = pre_asap.as_ref(); + let mut node = &pre_asap; loop { - match node { - QueryExpr::Scan { predicates, .. } => break predicates.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, - other => panic!("expected a unary SQL plan over Scan, got {other:?}"), + match node.non_asap() { + Some(NonASAPOp::Scan { predicates, .. }) => break predicates.clone(), + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => node = child, + _ => panic!( + "expected a unary SQL plan over Scan, got {:?}", + node.operator + ), } } }; @@ -516,41 +547,37 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { let mut node = root.as_ref(); let mut retained_read_predicate = None; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Filter { pred }, - .. - } => { - retained_read_predicate = Some(pred.clone()); - node = child; - } - SummaryExpr::ValueOperation { child, .. } => node = child, - SummaryExpr::SummaryEstimate { summary_input, .. } => { - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { - panic!("expected SummaryAgg below SummaryEstimate"); - }; - let SummaryExpr::KeepPreAsap(raw_input) = &child.expr else { - panic!("expected raw summary population below SummaryAgg"); - }; - let QueryExpr::Scan { predicates, .. } = raw_input.as_ref() else { - panic!("expected source selection to remain a Scan"); - }; - assert_eq!(predicates, &expected_source_predicates); - break; - } - other => panic!("expected read-time operations over a summary, got {other:?}"), + if let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg below SummaryEstimate"); + }; + assert!( + is_kept_non_asap(child), + "expected raw summary population below SummaryAgg" + ); + let Some(NonASAPOp::Scan { predicates, .. }) = child.non_asap() else { + panic!("expected source selection to remain a Scan"); + }; + assert_eq!(predicates, &expected_source_predicates); + break; + } + if let Some(NonASAPOp::Filter { pred, .. }) = node.non_asap() { + retained_read_predicate = Some(pred.clone()); } + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected read-time operations over a summary, got {:?}", + node.operator + ) + }); } assert_eq!(retained_read_predicate, Some(expected_read_predicate)); - let dag = compile_post_asap_dag(&root).expect("typed DAG compilation failed"); + let dag = post_asap_dag(&root); + let expected_wire = wire_pred(retained_read_predicate.as_ref().unwrap()); assert!(dag.nodes.iter().any(|node| matches!( &node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Filter { pred }, - .. - } if pred == retained_read_predicate.as_ref().unwrap() + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Filter { pred, .. }) if *pred == expected_wire ))); } @@ -559,15 +586,13 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { /// read-time operation. #[tokio::test] async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.avg_bytes FROM \ - (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t \ - WHERE t.avg_bytes > 100", - AccuracyTarget::Exact, - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.avg_bytes FROM \ + (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t \ + WHERE t.avg_bytes > 100", + AccuracyTarget::Exact, + ) + .await; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); let selection = space.global_selection(&DefaultCostModel); let root = selection @@ -578,22 +603,20 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { let mut node = root.as_ref(); let mut saw_filter = false; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, operation, .. - } => { - saw_filter |= matches!(operation, ValueOperation::Filter { .. }); - node = child; - } - SummaryExpr::KeepPreAsap(fallback) => { - assert!( - matches!(fallback.as_ref(), QueryExpr::BinaryOp { .. }), - "AVG's unsupported rewritten child should be opaque, got {fallback:?}" - ); - break; - } - other => panic!("expected local value operations over fallback child, got {other:?}"), + if let Some(NonASAPOp::BinaryOp { .. }) = node.non_asap() { + assert!( + is_kept_non_asap(node), + "AVG's unsupported rewritten child should be kept whole, got {node:?}" + ); + break; } + saw_filter |= matches!(node.non_asap(), Some(NonASAPOp::Filter { .. })); + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected local value operations over fallback child, got {:?}", + node.operator + ) + }); } assert!(saw_filter, "supported Filter must remain explicit"); } @@ -604,7 +627,7 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { /// ```text /// SummaryEstimate { query: Quantile{0.99} } → {…: Float64} /// └─ SummaryAgg { Kll{k:269}, input: metrics.latency } → {…: Sketch(Kll, {k:269})} -/// └─ KeepPreAsap(Scan) → {ts, service, latency, bytes} +/// └─ Scan (kept non-ASAP) → {ts, service, latency, bytes} /// ``` /// /// The SQL counterpart of `promql_to_post_asap.rs`'s @@ -622,12 +645,12 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, SketchStatistic::Quantile { q } if *q == 0.99)); assert_eq!( @@ -641,15 +664,15 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { "the summary-state type must not propagate past the estimate" ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -679,10 +702,12 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { ) ); - let SummaryExpr::KeepPreAsap(kept_leaf) = &child.expr else { - panic!("expected KeepPreAsap leaf, got {:?}", child.expr); - }; - assert!(matches!(kept_leaf.as_ref(), QueryExpr::Scan { .. })); + assert!( + is_kept_non_asap(child), + "expected a kept non-ASAP leaf, got {:?}", + child.operator + ); + assert!(matches!(child.non_asap(), Some(NonASAPOp::Scan { .. }))); assert!( child .schema @@ -709,12 +734,12 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, SketchStatistic::Cardinality)); assert_eq!( @@ -723,14 +748,14 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { "COUNT(DISTINCT …) reads back out as an integer count" ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -763,11 +788,11 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { .await; let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, reduction, .. - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryAgg, got {:?}", root.expr); + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, @@ -788,37 +813,42 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + is_kept_non_asap(&root), "avg has no mergeable accumulator — stays logical" ); + assert!( + root.guarantee.as_ref().is_some_and(|g| g.is_exact()), + "a kept logical sub_dag is exact" + ); } #[tokio::test] async fn map_projection_export_preserves_unsupported_child_boundary() { - let pre = Rc::new(lower_sql_dialect( + let pre = lower_sql_dialect( "SELECT map('job', t.service) AS labels, t.avg_bytes FROM (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t WHERE t.avg_bytes > 100", &catalog(), SqlDialect::ClickhouseSQL, AccuracyTarget::Exact, - ).await.unwrap()); + ).await.unwrap(); let space = search_workload(vec![("map_query", pre)]); let root = space .global_selection(&DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&root).unwrap(); + let dag = post_asap_dag(&root); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::Value { operation: ValueOperation::Project { cols, .. }, .. } - if cols.iter().any(|item| matches!(&item.expr, QueryExpr::FunctionCall { name, .. } if name == "map")) + PhysicalASAPOperatorPayload::NonASAP(NonASAPOp::Project { cols, .. }) + if cols.iter().any(|item| matches!(&item.expr, ScalarExpr::FunctionCall { name, .. } if name == "map")) ))); let mut node = root.as_ref(); loop { - match &node.expr { - SummaryExpr::ValueOperation { child, .. } => node = child, - SummaryExpr::KeepPreAsap(child) => { - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); - break; - } - other => panic!("unexpected map/fallback composition: {other:?}"), + if let Some(NonASAPOp::BinaryOp { .. }) = node.non_asap() { + assert!( + is_kept_non_asap(node), + "fallback child must stay whole: {node:?}" + ); + break; } + node = unary_child(node) + .unwrap_or_else(|| panic!("unexpected map/fallback composition: {:?}", node.operator)); } } diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index eeade3ae5..3edc21d7f 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -2,6 +2,9 @@ //! source workload -> PromQL lowering -> candidate search -> //! summary-maintenance lifecycle selection -> materialized deployment guarantees. +use asap_types::ir::physical_export::PhysicalASAPOperatorPayload; +use asap_types::ir::ASAPOp; +use physical_common::compile_physical_asap_dag; use std::rc::Rc; use asap_aware_mapping::cost_model::Cost; @@ -13,8 +16,9 @@ use asap_aware_mapping::{ SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecycleRejection, WorkloadDemand, }; use asap_frontend_promql::lower_promql_workload; +use asap_types::ir::OperatorNode; use asap_types::post_asap::{ - EvaluationSchedule, SummaryMaintenanceLifecycle, SummaryMaintenanceMode, SummaryNode, + EvaluationSchedule, SummaryMaintenanceLifecycle, SummaryMaintenanceMode, }; use asap_types::pre_asap::agg_intent::AggIntent; use asap_types::types::AccuracyTarget; @@ -32,7 +36,7 @@ struct FullyCostedRuntime; impl CostModel for FullyCostedRuntime { fn raw_query_recompute_total_cost( &self, - _target: &asap_types::pre_asap::QueryExpr, + _target: &OperatorNode, _expected_reads: f64, ) -> Option { Some(Cost(1_000.0)) @@ -48,7 +52,7 @@ impl CostModel for FullyCostedRuntime { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(10.0)), @@ -61,7 +65,7 @@ impl CostModel for FullyCostedRuntime { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -179,8 +183,8 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() .as_array() .unwrap() .iter() - .find(|node| node["kind"] == "SummaryAgg") - .expect("exported SummaryAgg node"); + .find(|node| node["kind"] == "summary_agg") + .expect("exported summary_agg node"); assert_eq!( summary_node["detail"]["summary_maintenance"]["selected"]["lifecycle"]["kind"], "continuously_maintained" @@ -217,11 +221,11 @@ fn selected_plan_with_horizon( fn selected_plan_for_lowered( workload: &PlanningWorkload, - lowered: asap_types::pre_asap::QueryExpr, + lowered: Rc, model: &dyn CostModel, horizon: Horizon, ) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { - let root = Rc::new(lowered); + let root = lowered; let strategies = asap_aware_mapping::default_strategies_with(model); let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); let target = Rc::clone(&space.roots[0].1); @@ -273,10 +277,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { runtime::Scope, values::{Batch, Value}, }; - use asap_types::{ - post_asap::{compile_post_asap_dag, FieldDataType, PostAsapOperatorPayload}, - pre_asap::DataType, - }; + use asap_types::{post_asap::FieldDataType, pre_asap::DataType}; use std::{collections::BTreeMap, sync::Arc}; let mut workload = dashboard_workload(); workload.query_workload.query_batch.as_mut().unwrap()[0].query = @@ -292,11 +293,16 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { .summary_maintenance_lifecycle, SummaryMaintenanceLifecycle::ContinuouslyMaintained ); - let dag = compile_post_asap_dag(&selected.root).unwrap(); + let dag = compile_physical_asap_dag(&selected.root).unwrap(); let build = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { .. }) + ) + }) .unwrap(); let input = dag .edges @@ -308,9 +314,9 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { let schema = Arc::new(raw.output_schema.clone()); let candidate = compile_candidate( &dag, - BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], - &[u64::from(build.id.0)], + BTreeMap::from([(input as u64, InputContract::bounded(schema.clone()))]), + &[dag.roots[0] as u64], + &[build.id as u64], ) .unwrap(); @@ -321,15 +327,15 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { unbounded.properties.boundedness = asap_physical_operators::plan::Boundedness::Unbounded; let rejected = compile_candidate( &dag, - BTreeMap::from([(u64::from(input.0), unbounded)]), - &[u64::from(dag.root.0)], - &[u64::from(build.id.0)], + BTreeMap::from([(input as u64, unbounded)]), + &[dag.roots[0] as u64], + &[build.id as u64], ); assert!(rejected.is_err()); let request = compile_candidate( &dag, - BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], + BTreeMap::from([(input as u64, InputContract::bounded(schema.clone()))]), + &[dag.roots[0] as u64], &[], ) .unwrap(); @@ -375,7 +381,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); let direct = physical_common::execute( &feedback.candidate.1.query, - BTreeMap::from([(u64::from(input.0), raw_batch.clone())]), + BTreeMap::from([(input as u64, raw_batch.clone())]), Scope::Query { evaluation_time_ms: 300_000, revision, @@ -383,7 +389,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { ); let state = physical_common::execute( candidate.precompute.as_ref().unwrap(), - BTreeMap::from([(u64::from(input.0), raw_batch)]), + BTreeMap::from([(input as u64, raw_batch)]), Scope::Ingestion { window_start_ms: 0, window_end_ms: 300_000, @@ -392,7 +398,7 @@ fn continuous_lifecycle_compiles_and_executes_spatial_kll() { ); let result = physical_common::execute( &candidate.query, - BTreeMap::from([(u64::from(build.id.0), state[0][0].clone())]), + BTreeMap::from([(build.id as u64, state[0][0].clone())]), Scope::Query { evaluation_time_ms: 300_000, revision, @@ -446,7 +452,7 @@ fn quantile_workload(query: &str) -> PlanningWorkload { fn lifecycle_timed_dag( query: &str, lifecycle: &SummaryMaintenanceLifecycle, -) -> (asap_types::post_asap::PostAsapDAG, Vec) { +) -> (asap_types::ir::physical_export::PhysicalASAPDAG, Vec) { use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; let workload = quantile_workload(query); let mut lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); @@ -475,7 +481,7 @@ fn lifecycle_timed_dag( .iter() .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) .collect(); - let mut states: Vec<_> = choices.iter().map(|(id, _)| u64::from(id.0)).collect(); + let mut states: Vec<_> = choices.iter().map(|(id, _)| *id as u64).collect(); states.sort_unstable(); let dag = candidates .select(&choices) @@ -487,7 +493,7 @@ fn lifecycle_timed_dag( /// Compile inputs for a timed DAG: its raw source, available at either phase. fn raw_inputs( - dag: &asap_types::post_asap::PostAsapDAG, + dag: &asap_types::ir::physical_export::PhysicalASAPDAG, ) -> std::collections::BTreeMap { let raw = dag .nodes @@ -495,12 +501,14 @@ fn raw_inputs( .find(|node| { matches!( node.payload, - asap_types::post_asap::PostAsapOperatorPayload::Fallback { .. } + asap_types::ir::physical_export::PhysicalASAPOperatorPayload::NonASAP( + asap_types::ir::NonASAPOp::TimeRange { .. } + ) ) }) .unwrap(); std::collections::BTreeMap::from([( - u64::from(raw.id.0), + raw.id as u64, asap_physical_operators::physical_planner::InputContract::bounded(std::sync::Arc::new( raw.output_schema.clone(), )), @@ -527,7 +535,7 @@ fn planner_lifecycle_selection_reproduces_strategy_timing() { != SummaryMaintenanceLifecycle::Ephemeral }) })); - let strategy = asap_types::post_asap::compile_post_asap_dag(&plan.root).unwrap(); + let strategy = compile_physical_asap_dag(&plan.root).unwrap(); assert_eq!(plan.execution_timed_dag().unwrap(), strategy, "{query}"); } } @@ -558,8 +566,7 @@ fn chosen_lifecycle_timing_decides_precompute_contents() { let (&raw_id, contract) = inputs.iter().next().unwrap(); let schema = contract.schema.clone(); let frontier = frontier_from_timing(&dag).unwrap(); - let candidate = - compile_candidate(&dag, inputs, &[u64::from(dag.root.0)], &frontier).unwrap(); + let candidate = compile_candidate(&dag, inputs, &[dag.roots[0] as u64], &frontier).unwrap(); let rows = (1..=100) .map(|value| { schema @@ -640,7 +647,7 @@ fn lifecycle_timing_cuts_one_compilation() { let ephemeral = SummaryMaintenanceLifecycle::Ephemeral; let (compiled_dag, _) = lifecycle_timed_dag(query, &ephemeral); let inputs = raw_inputs(&compiled_dag); - let roots = [u64::from(compiled_dag.root.0)]; + let roots = [compiled_dag.roots[0] as u64]; let compiled = compile(&compiled_dag, inputs.clone(), &roots).unwrap(); for lifecycle in [ SummaryMaintenanceLifecycle::ContinuouslyMaintained, @@ -651,7 +658,7 @@ fn lifecycle_timing_cuts_one_compilation() { // Retained states read by a query-time consumer, or the root itself. let query_time = |id: u64| { dag.nodes.iter().any(|node| { - u64::from(node.id.0) == id + node.id as u64 == id && node.output_state.timing == asap_types::post_asap::ExecutionTiming::QueryTime }) @@ -663,10 +670,9 @@ fn lifecycle_timing_cuts_one_compilation() { .iter() .copied() .filter(|state| { - *state == u64::from(dag.root.0) + *state == dag.roots[0] as u64 || dag.edges.iter().any(|edge| { - u64::from(edge.producer.0) == *state - && query_time(u64::from(edge.consumer.0)) + edge.producer as u64 == *state && query_time(edge.consumer as u64) }) }) .collect() @@ -701,15 +707,12 @@ fn chosen_population_lifecycle_decides_precompute_contents() { runtime::Scope, values::{Batch, Value}, }; - use asap_types::post_asap::{ - maintained_population::PopulationInput, PostAsapOperatorPayload, ValueOperation, - }; + use asap_types::post_asap::maintained_population::PopulationInput; use std::{collections::BTreeMap, sync::Arc}; let workload = quantile_workload("topk by(job)(1, m)"); - let root = Rc::new( - with_series_identity(&lower_promql_workload(&workload, 0).unwrap().remove(0)).unwrap(), - ); + let root = + with_series_identity(&lower_promql_workload(&workload, 0).unwrap().remove(0)).unwrap(); let root = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); @@ -741,9 +744,8 @@ fn chosen_population_lifecycle_decides_precompute_contents() { .execution_timed_dag() .unwrap(); let population = dag.nodes.iter().find(|node| node.id == id).unwrap(); - let PostAsapOperatorPayload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &population.payload + let PhysicalASAPOperatorPayload::ASAP(ASAPOp::MaintainPopulation { population, .. }) = + &population.payload else { panic!("the deployment is the maintained population"); }; @@ -754,15 +756,22 @@ fn chosen_population_lifecycle_decides_precompute_contents() { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP( + asap_types::ir::NonASAPOp::TimeRange { .. } + ) + ) + }) .unwrap(); - let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); + let (raw_id, schema) = (raw.id as u64, Arc::new(raw.output_schema.clone())); let frontier = asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); let candidate = compile_candidate( &dag, BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], + &[dag.roots[0] as u64], &frontier, ) .unwrap(); @@ -796,7 +805,7 @@ fn chosen_population_lifecycle_decides_precompute_contents() { query_scope, ) } else { - let state = u64::from(id.0); + let state = id as u64; assert_eq!(frontier, [state]); let stored = physical_common::execute( candidate.precompute.as_ref().unwrap(), @@ -836,20 +845,18 @@ fn chosen_population_lifecycle_decides_precompute_contents() { fn grouped_rate_sum_placement_is_a_lifecycle_choice() { use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; use asap_physical_operators::physical_planner::{compile_candidate, InputContract}; - use asap_types::post_asap::{ExactKind, FieldDataType, PostAsapOperatorPayload, SummaryExpr}; + use asap_types::post_asap::{ExactKind, FieldDataType}; use std::{collections::BTreeMap, sync::Arc}; let workload = quantile_workload("sum by(job)(rate(m[1m]))"); - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity( - &lower_promql_workload(&workload, 0).unwrap().remove(0), - ) - .unwrap(), - ); - let is_exact = |node: &SummaryNode, kind: ExactKind| { - matches!(&node.expr, SummaryExpr::SummaryAgg { + let root = asap_physical_operators::physical_planner::promql_rows::with_series_identity( + &lower_promql_workload(&workload, 0).unwrap().remove(0), + ) + .unwrap(); + let is_exact = |node: &OperatorNode, kind: ExactKind| { + matches!(&node.operator, asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. - } if *k == kind) + }) if *k == kind) }; let inventory = asap_aware_mapping::search_workload(vec![("q", root)]) .enumerate_candidate_dags(4096) @@ -859,7 +866,7 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { .into_iter() .map(|mut forest| forest.remove(0).1) .filter(|candidate| { - matches!(&candidate.expr, SummaryExpr::ValueOperation { child, .. } + matches!(&candidate.operator, asap_types::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_exact(child, ExactKind::Sum)) }) .collect::>(); @@ -905,7 +912,14 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP( + asap_types::ir::NonASAPOp::TimeRange { .. } + ) + ) + }) .unwrap(); let frontier = asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); @@ -915,15 +929,15 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { let boundary = dag .nodes .iter() - .find(|node| u64::from(node.id.0) == *boundary) + .find(|node| node.id as u64 == *boundary) .unwrap(); let physical = compile_candidate( &dag, BTreeMap::from([( - u64::from(raw.id.0), + raw.id as u64, InputContract::bounded(Arc::new(raw.output_schema.clone())), )]), - &[u64::from(dag.root.0)], + &[dag.roots[0] as u64], &frontier, ) .unwrap(); @@ -944,10 +958,10 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { else { unreachable!() }; - let state = |payload: &PostAsapOperatorPayload, kind: ExactKind| { - matches!(payload, PostAsapOperatorPayload::SummaryAgg { + let state = |payload: &PhysicalASAPOperatorPayload, kind: ExactKind| { + matches!(payload, PhysicalASAPOperatorPayload::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. - } if *k == kind) + }) if *k == kind) }; assert!(state(retained, ExactKind::Sum)); assert!(builds(retained_pre, "Rate") && builds(retained_pre, "Sum")); @@ -959,10 +973,10 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { /// The lifecycle-timed DAG Planner selects for `query` with upfront series /// typing, and whether it keeps an ingestion-time Binary. -fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDAG, bool) { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; +fn typed_selection(query: &str) -> (asap_types::ir::physical_export::PhysicalASAPDAG, bool) { + use asap_types::post_asap::ExecutionTiming; let workload = quantile_workload(query); - let lowered = asap_types::pre_asap::schema::with_promql_series_identity( + let lowered = asap_types::ir::schema_support::with_promql_series_identity( &lower_promql_workload(&workload, 0).unwrap().remove(0), ) .unwrap(); @@ -970,8 +984,10 @@ fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDAG, bool) { .execution_timed_dag() .unwrap(); let ingestion_binary = dag.nodes.iter().any(|node| { - matches!(node.payload, PostAsapOperatorPayload::Binary { .. }) - && node.output_state.timing == ExecutionTiming::IngestionTime + matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP(asap_types::ir::NonASAPOp::BinaryOp { .. }) + ) && node.output_state.timing == ExecutionTiming::IngestionTime }); (dag, ingestion_binary) } @@ -979,58 +995,49 @@ fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDAG, bool) { /// Execute a timed DAG's precompute and query DAGs over `samples` /// (`(metric, job, seconds, value)`) at 300s; returns the root's values. fn execute_timed( - dag: &asap_types::post_asap::PostAsapDAG, + dag: &asap_types::ir::physical_export::PhysicalASAPDAG, samples: &[(&str, &str, i64, f64)], ) -> Vec { use asap_physical_operators::{ - physical_planner::{ - compile_candidate, frontier_from_timing, promql_fallback, promql_rows, InputContract, - }, + physical_planner::{compile_candidate, frontier_from_timing, promql_rows, InputContract}, runtime::Scope, values::{Batch, Value}, }; - use asap_types::{ - post_asap::PostAsapOperatorPayload, - pre_asap::{QueryExpr, Source}, - }; + use asap_types::{ir::physical_export::PhysicalASAPOperatorPayload, pre_asap::Source}; use std::{collections::BTreeMap, sync::Arc}; // Raw inputs: a selector Fallback is itself the input; a retained // expression reads each of its selectors through its raw-series slots. let mut raw = BTreeMap::new(); for node in &dag.nodes { - let PostAsapOperatorPayload::Fallback { expression } = &node.payload else { + if !matches!( + node.payload, + PhysicalASAPOperatorPayload::NonASAP( + asap_types::ir::NonASAPOp::TimeRange { .. } + | asap_types::ir::NonASAPOp::Scan { .. } + ) + ) { continue; - }; - let metric = |selector: &QueryExpr| match selector { - QueryExpr::TimeRange { child, .. } => match child.as_ref() { - QueryExpr::Scan { - source: Source::TimeSeries { metric }, - .. - } => Some(metric.clone()), - _ => None, - }, - QueryExpr::Scan { + } + let mut id = node.id; + loop { + let n = dag.nodes.iter().find(|n| n.id == id).unwrap(); + if let PhysicalASAPOperatorPayload::NonASAP(asap_types::ir::NonASAPOp::Scan { source: Source::TimeSeries { metric }, .. - } => Some(metric.clone()), - _ => None, - }; - if let Some(name) = metric(expression) { - raw.insert( - u64::from(node.id.0), - (Arc::new(node.output_schema.clone()), name), - ); - } else { - for (i, (selector, schema)) in promql_fallback::raw_series(expression) - .unwrap() - .into_iter() - .enumerate() + }) = &n.payload { raw.insert( - promql_fallback::raw_series_input(u64::from(node.id.0), i), - (schema, metric(&selector).unwrap()), + node.id as u64, + (Arc::new(node.output_schema.clone()), metric.clone()), ); + break; } + id = dag + .edges + .iter() + .find(|e| e.consumer == id) + .unwrap() + .producer; } } let batch = |schema: &asap_physical_operators::values::SchemaRef, name: &str| { @@ -1053,7 +1060,7 @@ fn execute_timed( raw.iter() .map(|(id, (schema, _))| (*id, InputContract::bounded(schema.clone()))) .collect(), - &[u64::from(dag.root.0)], + &[dag.roots[0] as u64], &frontier, ) .unwrap(); @@ -1126,7 +1133,7 @@ fn maintained_arithmetic_over_different_selectors_matches_prometheus() { } /// Arithmetic over one selector keeps its maintained layout and adds each -/// series' two readouts before the quantile. +/// series' two evaluations before the quantile. #[test] fn maintained_arithmetic_over_one_selector_executes() { let (dag, ingestion_binary) = diff --git a/crates/integration-tests/tests/time_range.rs b/crates/integration-tests/tests/time_range.rs index d3ab732fe..95212ca99 100644 --- a/crates/integration-tests/tests/time_range.rs +++ b/crates/integration-tests/tests/time_range.rs @@ -1,46 +1,53 @@ -//! `QueryExpr::TimeRange` — range / streaming function tests. +//! `NonASAPOp::TimeRange` — range / streaming function tests. //! -//! All range functions lower to `Aggregate { child: TimeRange { range, child: Scan } }`. +//! All range functions lower to `Aggregate { child: TimeRange { range, kind: Range, child: Scan } }`. //! The temporal range lives on the `TimeRange` node, not in the `AggIntent`. //! `rate` / `increase` use `AggIntent::Rate` / `AggIntent::Increase` (no window field). //! `*_over_time` functions reuse the corresponding cross-series intents -//! (`Count`, `Sum`, `Quantile`, …) — the `TimeRange` child is what marks them -//! as per-series reductions. +//! (`Count`, `Sum`, `Quantile`, …) — the `Range` selector child is what marks +//! them as per-series reductions. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; +use asap_types::pre_asap::{AggIntent, Reduction, Source}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(&[]), - } + }) } -fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(range_secs), - child: Rc::new(scan(metric)), + kind: TimeRangeKind::Range, + child: scan(metric), }), - } + }) } // #13 — rate: counter-reset-aware per-second rate; range on TimeRange node diff --git a/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs b/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs index 3ac1170e8..6befd206d 100644 --- a/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs +++ b/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs @@ -12,7 +12,7 @@ use crate::functions::{BuiltinFunction, TransformFunction}; use crate::parser::{parse_number, ParseError, ParseResult}; #[allow(rustdoc::private_intra_doc_links)] -/// Partially evaluate `Expr`s so constant subtrees are evaluated at plan time. +/// Partially evaluate `Expr`s so constant sub-DAGs are evaluated at plan time. /// /// Note it does not handle algebraic rewrites such as `(a or false)` /// --> `a`, which is handled by [`Simplifier`] diff --git a/crates/planner/src/lib.rs b/crates/planner/src/lib.rs index 88a085bf8..4f527f148 100644 --- a/crates/planner/src/lib.rs +++ b/crates/planner/src/lib.rs @@ -11,15 +11,13 @@ //! and a catalog — skips this crate and calls //! [`asap_aware_mapping::optimize`] directly. -use std::rc::Rc; - use asap_types::parsed_workload::{ParsedWorkload, ParsedWorkloadError}; -use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::workload::{PlanningWorkload, QueryLanguage, SqlDialect, WorkloadError}; -use asap_frontend_metricsql::{lower_metricsql, MetricsqlError}; +use asap_frontend_metricsql::{lower_metricsql_query, MetricsqlError}; use asap_frontend_promql::{ - lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, + lower_promql_query_workload, lower_promql_query_workload_with_histograms, HistogramCatalog, + PromqlError, }; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError}; @@ -191,7 +189,7 @@ pub async fn e2e_plan(input: UserInput<'_>) -> Result { input.validate()?; let exprs = lower(&input).await?; - let parsed = ParsedWorkload::new(input.workload.clone(), exprs)?; + let parsed = ParsedWorkload::from_roots(input.workload.clone(), exprs)?; let fallback = MajorPass; let pass: &dyn OptimizationPass = input.pass.unwrap_or(&fallback); @@ -206,7 +204,7 @@ pub async fn e2e_plan(input: UserInput<'_>) -> Result { /// through `lower_sql_batch`, which walks `query_batch` alone and would drop /// every repeating query — exactly the entries whose recurrence the lifecycle /// stage needs. -async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { +async fn lower(input: &UserInput<'_>) -> Result, PlanError> { let entries = || input.workload.query_workload.entries(); match &input.frontend_specific { @@ -229,7 +227,7 @@ async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { entry_index: Some(index), source: LoweringError::Sql(source), })?; - lowered.push(Rc::new(expr)); + lowered.push(expr.into()); } Ok(lowered) } @@ -237,28 +235,29 @@ async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { now_ms, histograms, .. } => { let lowered = match histograms { - Some(histograms) => lower_promql_workload_with_histograms( + Some(histograms) => lower_promql_query_workload_with_histograms( input.workload, histograms.clone(), *now_ms, ), - None => lower_promql_workload(input.workload, *now_ms), + None => lower_promql_query_workload(input.workload, *now_ms), } .map_err(|source| PlanError::Lowering { entry_index: None, source: LoweringError::Promql(source), })?; - Ok(lowered.into_iter().map(Rc::new).collect()) + Ok(lowered) } FrontendInput::Metricsql => { let mut lowered = Vec::new(); for (index, entry) in entries().enumerate() { - let expr = lower_metricsql(&entry.query.0, entry.requirements.accuracy.target()) - .map_err(|source| PlanError::Lowering { - entry_index: Some(index), - source: LoweringError::Metricsql(source), - })?; - lowered.push(Rc::new(expr)); + let expr = + lower_metricsql_query(&entry.query.0, entry.requirements.accuracy.target()) + .map_err(|source| PlanError::Lowering { + entry_index: Some(index), + source: LoweringError::Metricsql(source), + })?; + lowered.push(expr); } Ok(lowered) } diff --git a/crates/planner/tests/e2e_plan.rs b/crates/planner/tests/e2e_plan.rs index ff8a8dde0..3ab55b101 100644 --- a/crates/planner/tests/e2e_plan.rs +++ b/crates/planner/tests/e2e_plan.rs @@ -12,7 +12,6 @@ use asap_aware_mapping::{ }; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; use asap_planner::{e2e_plan, FrontendInput, PlanError, UserInput, UserInputError}; -use asap_types::post_asap::SummaryExpr; use asap_types::pre_asap::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ @@ -137,7 +136,7 @@ async fn builtin_cost_model_cannot_price_lifecycles_and_falls_back_to_raw_recomp ) .await .expect("lowers"); - roots.push((index, Rc::new(expr), Some(accuracy))); + roots.push((index, expr, Some(accuracy))); } let strategies = default_strategies_with_evidence(models.cost, models.evidence); let space = search_workload_with_targets(roots, &strategies, models.accuracy); @@ -150,12 +149,12 @@ async fn builtin_cost_model_cannot_price_lifecycles_and_falls_back_to_raw_recomp .expect("assembles") .expect("root has a group"); assert!( - !matches!(cost_only.expr, SummaryExpr::KeepPreAsap(_)), + cost_only.contains_asap(), "entry {}: cost-only selection was expected to pick a summary", plan.entry_index ); assert!( - matches!(plan.plan.root.expr, SummaryExpr::KeepPreAsap(_)) + !plan.plan.root.contains_asap() && plan.plan.selected_raw_recompute && plan.plan.deployments.is_empty() && plan.plan.summary_total_cost.is_none() @@ -390,7 +389,7 @@ async fn lifecycle_decisions_ride_inside_each_plan() { assert_eq!(output.plans.len(), 1); assert_eq!(output.plans[0].entry_index, 0); let _: &Rc<_> = &output.plans[0].plan.root; - assert_eq!(output.dags().len(), 1); + assert_eq!(output.operator_roots().len(), 1); } /// Each root's lifecycle is planned against the entries that read it: a @@ -437,3 +436,54 @@ async fn each_plan_counts_only_its_own_entries_reads() { let reads: Vec<_> = output.plans.iter().map(|p| p.plan.expected_reads).collect(); assert_eq!(reads, vec![Some(60.0), Some(6.0)]); } + +/// Scalar-only and mixed workloads preserve entry bindings without wrapper nodes. +#[tokio::test] +async fn scalar_roots_survive_planning_in_workload_order() { + for queries in [ + vec!["2", "time()"], + vec!["2", "up * 2", "scalar(sum(up)) + 1"], + ] { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(queries.iter().map(|q| batch(q)).collect()), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let output = e2e_plan(UserInput::new( + &workload, + FrontendInput::Promql { + now_ms: NOW_MS, + histograms: None, + }, + PlanningModels::builtin(), + lifecycle(), + )) + .await + .unwrap(); + assert_eq!( + output.entry_indices(), + (0..queries.len()).collect::>() + ); + assert!(matches!( + output.roots()[0], + asap_types::ir::QueryRoot::Scalar(_) + )); + assert_eq!(output.roots().len(), queries.len()); + if queries.len() == 3 { + assert_eq!(output.plans[0].entry_index, 1); + let asap_types::ir::QueryRoot::Scalar(expr) = &output.roots()[2] else { + panic!() + }; + assert_eq!(expr.operator_refs().len(), 1); + } + } +} diff --git a/crates/planner/tests/summary_sharing.rs b/crates/planner/tests/summary_sharing.rs index 41b489e9c..7c198493b 100644 --- a/crates/planner/tests/summary_sharing.rs +++ b/crates/planner/tests/summary_sharing.rs @@ -1,6 +1,8 @@ //! Structurally identical summary producers chosen by different queries are -//! shared after Pass 1: one `Rc` across their plans, costed once. +//! shared after Pass 1: one `Rc` across their plans, costed once. +use asap_types::ir::cse::share_common_sub_dags; +use asap_types::ir::{ASAPOp, OperatorNode}; use std::rc::Rc; use asap_aware_mapping::accuracy::{ @@ -11,7 +13,7 @@ use asap_aware_mapping::pass::{PlanOutput, PlanningModels}; use asap_aware_mapping::replacement::{default_size_params, DEFAULT_DELTA}; use asap_aware_mapping::{ global_selection_with_summary_maintenance_lifecycles, search_workload_with_targets, - ReplacementStrategy, SketchAlgorithmStrategy, WorkloadDemand, + ASAPStrategies, ReplacementStrategy, WorkloadDemand, }; use asap_aware_mapping::{ CostModel, CostRate, DefaultCostModel, Horizon, LifecycleInput, SummaryMaintenanceCapabilities, @@ -21,15 +23,13 @@ use asap_frontend_promql::lower_promql_workload; use asap_frontend_sql::SqlCatalog; use asap_planner::{e2e_plan, FrontendInput, UserInput}; use asap_types::post_asap::{ - share_common_summary_sub_dags, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, - ProbabilityExpr, ResultGuarantee, SketchStatistic, -}; -use asap_types::post_asap::{ - FieldDataType, SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, + SketchStatistic, }; +use asap_types::post_asap::{FieldDataType, SketchAlgorithm, SketchParams}; use asap_types::pre_asap::agg_intent::default_quantile; use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::pre_asap::AggIntent; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, DataArrival, DataWorkload, DurationMs, Evidence, LatencyRequirement, @@ -58,7 +58,7 @@ impl CostModel for FixedCosts { fn summary_maintenance_lifecycle_cost_inputs( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceLifecycleCostInputs { SummaryMaintenanceLifecycleCostInputs { build_cost: Some(Cost(self.build)), @@ -71,7 +71,7 @@ impl CostModel for FixedCosts { fn summary_maintenance_capabilities( &self, - _summary: &SummaryNode, + _summary: &OperatorNode, ) -> SummaryMaintenanceCapabilities { SummaryMaintenanceCapabilities { incremental_update: true, @@ -80,7 +80,7 @@ impl CostModel for FixedCosts { } } - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { + fn raw_query_recompute_cost(&self, _target: &OperatorNode) -> Option { Some(Cost(self.raw_per_read)) } } @@ -185,7 +185,7 @@ async fn plan_sql(queries: &[&str], costs: &FixedCosts) -> PlanOutput { } /// Every summary state each plan deploys. -fn states(output: &PlanOutput) -> Vec>> { +fn states(output: &PlanOutput) -> Vec>> { output .plans .iter() @@ -202,7 +202,7 @@ fn states(output: &PlanOutput) -> Vec>> { } /// Whether the two plans deploy exactly the same states, by pointer. -fn same_states(states: &[Vec>]) -> bool { +fn same_states(states: &[Vec>]) -> bool { states[0].len() == states[1].len() && states[0] .iter() @@ -212,7 +212,7 @@ fn same_states(states: &[Vec>]) -> bool { /// The deployments a consumer would run, deduplicated by pointer. fn unique_deployments(output: &PlanOutput) -> usize { - let mut seen: Vec<*const SummaryNode> = Vec::new(); + let mut seen: Vec<*const OperatorNode> = Vec::new(); for plan in &output.plans { for deployment in &plan.plan.deployments { let ptr = Rc::as_ptr(&deployment.summary); @@ -297,12 +297,12 @@ fn kll_k(plan: &asap_aware_mapping::pass::QueryLifecyclePlan) -> u32 { let [deployment] = plan.plan.deployments.as_slice() else { panic!("one state: {:?}", plan.plan.deployments.len()); }; - let SummaryExpr::SummaryAgg { + let asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &deployment.summary.expr + }) = &deployment.summary.operator else { - panic!("sketch state: {:?}", deployment.summary.expr); + panic!("sketch state: {:?}", deployment.summary.operator); }; let SketchParams::Kll { k } = kind.params() else { panic!("KLL state: {kind:?}"); @@ -382,7 +382,7 @@ async fn identical_sql_percentiles_share_one_producer() { assert_eq!(unique_deployments(&output), 1); } -/// The quantile is a readout parameter: SQL p50 and p99 over one filtered +/// The quantile is a evaluation parameter: SQL p50 and p99 over one filtered /// column build one KLL, named after its input, while each query keeps its /// own output column. #[tokio::test] @@ -451,7 +451,7 @@ async fn shared_amortization_alone_can_beat_raw_recompute() { assert_eq!(unique_deployments(&output), 1); } -/// Synthetic evidence certifying UnivMon readouts; it exercises sharing, never +/// Synthetic evidence certifying UnivMon evaluations; it exercises sharing, never /// runtime accuracy. struct UnivMonEvidence; @@ -493,7 +493,7 @@ impl AccuracyModel for UnivMonEvidence { /// when the states are identical. `MajorPass` builds candidates with the /// built-in accuracy model, so this runs its pipeline with the test model. #[test] -fn certified_frequency_readouts_share_one_univmon_state() { +fn certified_frequency_evaluations_share_one_univmon_state() { let queries = [ ("distinct_over_time(m[5m])", 0.02), ("entropy_over_time(m[5m])", 0.02), @@ -505,12 +505,10 @@ fn certified_frequency_readouts_share_one_univmon_state() { .into_iter() .zip(queries) .enumerate() - .map(|(index, (expr, (_, epsilon)))| { - (index, Rc::new(expr), Some(AccuracyTarget::Epsilon(epsilon))) - }) + .map(|(index, (expr, (_, epsilon)))| (index, expr, Some(AccuracyTarget::Epsilon(epsilon)))) .collect(); let strategies: Vec> = - vec![Box::new(SketchAlgorithmStrategy::new_with_planning_inputs( + vec![Box::new(ASAPStrategies::new_with_planning_inputs( &CHEAP_SUMMARY, &UnivMonEvidence, &EqualSplitAllocator, @@ -541,15 +539,17 @@ fn certified_frequency_readouts_share_one_univmon_state() { (*index, dag) }) .collect(); - let mut states: Vec> = Vec::new(); - for (_, root) in share_common_summary_sub_dags(assembled) { - assert!(root.guarantee.is_some(), "{:?}", root.expr); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("summary readout: {:?}", root.expr); + let mut states: Vec> = Vec::new(); + for (_, root) in share_common_sub_dags(assembled) { + assert!(root.guarantee.is_some(), "{:?}", root.operator); + let asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = + &root.operator + else { + panic!("summary evaluation: {:?}", root.operator); }; assert!(matches!( - &summary_input.expr, - SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + &summary_input.operator, + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::UnivMon )); states.push(Rc::clone(summary_input)); diff --git a/crates/sql-function-catalog/src/lib.rs b/crates/sql-function-catalog/src/lib.rs index 0ae4ee5ab..f2aedbaa9 100644 --- a/crates/sql-function-catalog/src/lib.rs +++ b/crates/sql-function-catalog/src/lib.rs @@ -277,7 +277,7 @@ pub struct ClickHouseBuiltin { pub const CLICKHOUSE_BUILTINS: &[ClickHouseBuiltin] = &[ // Explicit time-series reducers. These deliberately survive under their // own names: the SQL frontend validates (value, timestamp, window_ms) and - // lowers the window to QueryExpr::TimeRange rather than pretending these + // lowers the window to NonASAPOp::TimeRange rather than pretending these // are ordinary tabular aggregates. ClickHouseBuiltin { name: "asap_rate", diff --git a/crates/types/src/dag_export.rs b/crates/types/src/dag_export.rs index f26673ea7..9efbd66c5 100644 --- a/crates/types/src/dag_export.rs +++ b/crates/types/src/dag_export.rs @@ -1,31 +1,46 @@ -//! Export the pre-ASAP [`QueryExpr`] DAG as a generic node/edge DAG, for tools -//! that need to render or diff the IR (the `dag_export` example + the -//! `tools/dag-viewer` viewer — see issue #133) rather than walk it in Rust. +//! Export an [`OperatorNode`] DAG as a generic node/edge dag, for tools +//! that need to render or diff the IR (the `dag_export` devtools binary + +//! the `tools/dag-viewer` viewer — see issue #133) rather than walk it in +//! Rust. //! -//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged DAG -//! (`Rc` children nested inside each variant's own field). This module -//! flattens that into an explicit node list + child-id edges — the shape a -//! generic DAG renderer wants — and additionally tags each node with -//! [`structural_hash`](crate::pre_asap::cse::structural_hash), so a caller -//! with several exported queries can spot identical sub-DAGs (a -//! shared `Scan`, a repeated `Aggregate` shape, …) by comparing hashes -//! rather than re-implementing `QueryExpr: PartialEq` structural comparison -//! client-side. +//! `OperatorNode` already derives `Serialize`, but as a Rust-shaped tagged +//! tree (`Rc` children nested inside each variant's own field, repeated once +//! per reference). This module flattens that into an explicit node list + +//! child-id edges — one entry per unique node, deduplicated by `Rc` pointer +//! identity, so a shared sub-DAG stays one node with several parents — and +//! additionally tags each node with +//! [`structural_hash`](crate::ir::cse::structural_hash), so a caller with +//! several exported queries can spot identical sub-DAGs (a shared `Scan`, a +//! repeated `Aggregate` shape, …) by comparing hashes rather than +//! re-implementing structural comparison client-side. //! //! This is literally the same hashing -//! [`share_common_sub_dags`](crate::pre_asap::cse::share_common_sub_dags) -//! uses to bucket candidates in its `InternTable` (issue #223 stage 3) — not -//! a parallel reimplementation. `tools/dag-viewer`'s "shared sub-DAG" +//! [`share_common_sub_dags`](crate::ir::cse::share_common_sub_dags) uses to +//! bucket candidates in its `InternTable` (issue #223 stage 3) — not a +//! parallel reimplementation. `tools/dag-viewer`'s "shared sub-DAG" //! highlighting is still a *proxy* for real CSE, though: a hash match here //! only means two nodes are legal `InternTable` bucket-mates (same coarse -//! hash), the same candidate-narrowing step `structural_hash` performs -//! inside `InternTable::intern` — it does not mean `share_common_sub_dags` -//! actually ran on this data and merged them onto one `Rc` (that also -//! requires the `PartialEq` check `InternTable::intern` performs, and the +//! hash) — it does not mean `share_common_sub_dags` actually ran on this +//! data and merged them onto one `Rc` (that also requires the structural +//! equality check `InternTable::intern` performs, and the //! `Schema::has_unique_key` legality gate, neither of which this export //! step evaluates). See `tools/dag-viewer/README.md` for the up-to-date //! caveat. //! +//! There is one IR before and after ASAP optimization, so there is one +//! exporter: an ordinary operator and an ASAP summary operator are both +//! rendered by the same per-variant [`shape`] match, whichever entry point +//! ([`export`], [`export_summary`], [`export_post_asap`]) reached them. +//! +//! ## Scalar expressions +//! +//! A [`ScalarExpr`] is owned by value by an operator field (`Filter.pred`, +//! `Project.cols`, …) and is rendered into that operator's `detail`, not as +//! a node of its own. The operator nodes a scalar expression reads +//! (`scalar(v)`, `EXISTS (subquery)`, …) *are* nodes of the dag — they are +//! in [`OperatorNode::children`] — so inside `detail` each such reference is +//! rendered as `{"scalar_ref": }` rather than inlined. +//! //! ## `DAGNode::notes` — a layering seam, not a feature this module implements //! //! [`DAGNode`] also carries `notes: Vec<`[`DAGNote`]`>`, always empty coming @@ -33,13 +48,12 @@ //! `asap_types`, never the reverse — can annotate an already-exported DAG //! after the fact without this module needing to know anything about that //! layer's concepts. Concretely: `asap-aware-mapping`'s `explanation` module -//! (issue #257) computes `structural_hash` over the same `QueryExpr` -//! sub-DAGs this module does (via the identical function). The devtools -//! exporter uses that hash to narrow candidates, then compares -//! `ReplacementExplanation::target` with [`DAGNode::source_expr`] for a -//! collision-safe match before pushing a [`DAGNote`] onto the node. -//! `asap_types` itself never constructs a `DAGNote` — see [`DAGNode::notes`] -//! for the layering rule this keeps. +//! (issue #257) computes `structural_hash` over the same nodes this module +//! does (via the identical function). The devtools exporter uses that hash +//! to narrow candidates, then compares its target with +//! [`DAGNode::source_node`] for a collision-safe match before pushing a +//! [`DAGNote`] onto the node. `asap_types` itself never constructs a +//! `DAGNote` — see [`DAGNode::notes`] for the layering rule this keeps. use std::collections::HashMap; use std::rc::Rc; @@ -47,9 +61,11 @@ use std::rc::Rc; use serde::Serialize; use crate::cost::CostAnnotation; -use crate::post_asap::{AccuracyError, ResultGuarantee, SummaryExpr, SummaryNode}; -use crate::pre_asap::cse::{structural_hash, HashCache}; -use crate::pre_asap::query_expr::{QueryExpr, Source}; +use crate::ir::cse::{structural_hash, HashCache}; +use crate::ir::operator_properties::Source; +use crate::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; +use crate::post_asap::{AccuracyError, ResultGuarantee}; +use crate::pre_asap::schema::FieldDataType; /// One flattened IR node. `detail` holds this node's own scalar fields /// (predicates, aggregate funcs, schema, sort keys, …) — everything except @@ -57,56 +73,48 @@ use crate::pre_asap::query_expr::{QueryExpr, Source}; #[derive(Debug, Clone, Serialize)] pub struct DAGNode { pub id: u32, - /// The `QueryExpr` variant name (e.g. `"Aggregate"`). + /// The operator variant name — [`Operator::kind_name`] (e.g. + /// `"Aggregate"`, `"SummaryAgg"`). pub kind: &'static str, /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, - /// Output schema carried by every exported node. Edge renderers use the - /// child node's schema as the schema flowing along child → consumer. + /// Output schema carried by every exported node ([`OperatorNode::schema`] + /// as JSON). Edge renderers use the child node's schema as the schema + /// flowing along child → consumer. #[serde(skip_serializing_if = "Option::is_none")] pub schema: Option, - /// Child node ids, in the variant's field order (e.g. `Join` is - /// `[left, right]`). + /// Child node ids in [`OperatorNode::children`] order: the operator's + /// own inputs in field order (e.g. `Join` is `[left, right]`), then the + /// nodes referenced from its scalar expressions. pub children: Vec, /// Explicit workload-wide identity assigned by a higher-level exporter. /// Viewers use this field to union nodes and must not reconstruct a /// structural signature client-side. #[serde(skip_serializing_if = "Option::is_none")] pub workload_node_id: Option, - /// [`structural_hash`](crate::pre_asap::cse::structural_hash) of the - /// sub-DAG rooted at this node — the exact same function `cse`'s - /// `InternTable` uses to bucket CSE candidates, so two nodes hash - /// equally here iff they would land in the same `InternTable` bucket. - /// See the module doc for what a hash match here does and doesn't - /// guarantee. - /// - /// `None` for the same reason `source_expr` is `None` — a post-ASAP- - /// originated node in an [`export_post_asap`] merged DAG has no - /// `QueryExpr` to hash. Omitted from JSON entirely (rather than, say, - /// serialized as `0`) so a consumer's shared-sub-DAG-by-hash pass can - /// tell "no hash" apart from a real hash that happens to collide with a - /// placeholder — `0` is a legal `structural_hash` output, not a safe - /// sentinel. + /// [`structural_hash`](crate::ir::cse::structural_hash) of the sub-DAG + /// rooted at this node — the exact same function `cse`'s `InternTable` + /// uses to bucket CSE candidates, so two nodes hash equally here iff they + /// would land in the same `InternTable` bucket. See the module doc for + /// what a hash match here does and doesn't guarantee. Always `Some` + /// for a node this module produces; the `Option` is retained for the + /// JSON shape (`None` is omitted rather than serialized as a sentinel, + /// since `0` is a legal hash). #[serde(skip_serializing_if = "Option::is_none")] pub hash: Option, - /// Exact source expression for in-process annotation matching. It is not - /// part of the JSON format: callers first narrow by `hash`, then compare - /// this value structurally to avoid treating a hash collision as node - /// identity. - /// - /// `None` for a node with no corresponding pre-ASAP `QueryExpr` at all — - /// only possible for a post-ASAP-originated node inside a merged - /// [`export_post_asap`] DAG (a `SummaryAgg`/`SummaryJoin`/… node has no - /// single `QueryExpr` it corresponds to). Every node [`export`] itself - /// produces is pre-ASAP by construction and always carries `Some`. + /// The exported node itself, for in-process annotation matching. Not + /// part of the JSON format: callers first narrow by `hash`, then + /// compare this value (by pointer or structurally) to avoid treating a + /// hash collision as node identity. Always `Some` for a node this + /// module produces. #[serde(skip)] - pub source_expr: Option, - /// In-process identity of the source `QueryExpr`. Unlike `source_expr`'s - /// structural value, this preserves an `Rc` child reached from multiple - /// parents so post-ASAP flattening can retain true DAG sharing. + pub source_node: Option>, + /// In-process identity of `source_node` (`Rc::as_ptr` as an address): + /// the key the builder deduplicates on, so a node reached from several + /// parents is exported once. Not part of the JSON format. #[serde(skip)] - source_ptr: Option, + pub source_ptr: Option, /// Arbitrary reporting-layer annotations for this node — e.g. why a /// replacement exists here. `asap_types` never populates this itself /// (it has no notion of a "replacement" at all — see the module doc's @@ -187,7 +195,7 @@ pub struct EdgeCostAnnotation { pub cost: CostAnnotation, } -/// One query's exported DAG. `nodes[root as usize]` is the DAG's root. +/// One query's exported dag. `nodes[root as usize]` is the DAG's root. #[derive(Debug, Clone, Serialize)] pub struct ExportDAG { pub nodes: Vec, @@ -195,8 +203,7 @@ pub struct ExportDAG { /// See [`EdgeCostAnnotation`]. Always empty unless a higher layer /// explicitly populated it (same layering rule as [`DAGNode::notes`]); /// omitted from JSON entirely when empty, so every existing producer of - /// [`ExportDAG`] (every call to [`export`]/[`export_summary`]) is - /// unaffected. + /// [`ExportDAG`] is unaffected. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub edge_annotations: Vec, } @@ -205,10 +212,10 @@ pub struct ExportDAG { #[derive(Debug, Clone, Serialize)] pub struct NamedDAG { pub name: String, - /// The original query text (SQL or PromQL) this DAG was lowered from, - /// for display alongside the DAG — not used by `export` itself, since - /// that only sees the already-lowered `QueryExpr`. Optional because not - /// every producer of a `NamedDAG` has the source text on hand. + /// The original query text (SQL or PromQL) this dag was lowered from, + /// for display alongside the dag — not used by `export` itself, since + /// that only sees the already-lowered DAG. Optional because not every + /// producer of a `NamedDAG` has the source text on hand. #[serde(skip_serializing_if = "Option::is_none")] pub source: Option, pub dag: ExportDAG, @@ -228,15 +235,13 @@ pub struct NamedDAG { /// [`TargetReplacement::before`]/`::after` (small, self-contained /// before/after pairs, one per independently-discovered replacement /// site), this is a single flattened [`ExportDAG`] spanning the whole - /// query: every node that has no winning replacement renders as an - /// ordinary pre-ASAP [`DAGNode`] (same shape [`export`] itself - /// produces), and every node that does splices in its winning - /// candidate's shape instead — a rewritten [`QueryExpr`] sub-DAG, or a - /// bound `SummaryNode` sub-DAG, rendered inline in the very same node - /// list. `None` unless a higher layer explicitly built one (e.g. the - /// `dag_export` devtools binary's `--post-asap` flag); omitted from the - /// JSON entirely when absent, so every existing producer/consumer of - /// `NamedDAG` is unaffected. + /// query: every node that has no winning replacement renders as it does + /// in [`export`], and every node that does splices in its winning + /// candidate's sub-DAG instead, in the very same node list. `None` + /// unless a higher layer explicitly built one (e.g. the `dag_export` + /// devtools binary's `--post-asap` flag); omitted from the JSON entirely + /// when absent, so every existing producer/consumer of `NamedDAG` is + /// unaffected. #[serde(default, skip_serializing_if = "Option::is_none")] pub post_dag: Option, /// This query's own selected-workload cost/benefit — one of issue @@ -282,81 +287,51 @@ pub struct WorkloadDAG { pub workload_cost: Option, } -// ── Post-ASAP replacement export — a second, layering-seam-shaped feature ── +// ── Post-ASAP replacement export — a layering-seam-shaped feature ────────── // -// Everything below this point is the post-ASAP counterpart of the pre-ASAP -// flattening above: [`export_summary`] flattens a `SummaryNode` the same way -// [`export`] flattens a `QueryExpr`, and [`TargetReplacement`] is the -// generic, crate-agnostic "one replacement site, before and after" shape a -// higher layer (`asap-aware-mapping`, via the `dag_export` devtools binary's -// `--post-asap` flag) populates after running its own search — the exact -// same layering rule [`DAGNode::notes`]'s doc above already states: this -// module never runs `asap_aware_mapping::replacement::search_workload_with` -// itself, never picks a "winning" candidate, and has no opinion on what a +// [`TargetReplacement`] is the generic, crate-agnostic "one replacement +// site, before and after" shape a higher layer (`asap-aware-mapping`, via +// the `dag_export` devtools binary's `--post-asap` flag) populates after +// running its own search — the exact same layering rule [`DAGNode::notes`]'s +// doc above already states: this module never runs +// `asap_aware_mapping::replacement::search_workload_with` itself, never +// picks a "winning" candidate, and has no opinion on what a // `ReplacementProvenance` or a cost model even is. It only defines shapes // concrete and serializable enough for a higher layer to fill in, and for // `tools/dag-viewer` to render without needing to know anything about // `asap-aware-mapping`'s own vocabulary. -// -// A single whole-query "post-ASAP DAG" isn't attempted here, and isn't -// representable in the current type system either: `SummaryExpr` has no -// variant letting a `SummaryNode` be embedded back inside a plain -// `QueryExpr`'s child slot (`QueryExpr`'s own children are always -// `Rc`, never `Rc`), so there is no way to splice a -// post-ASAP binding back into its original pre-ASAP DAG in place. Inventing -// a bridge type for that is a real `asap_types`/`asap-aware-mapping` IR -// design decision, well beyond what a devtools visualization export should -// decide unilaterally. Instead, each independently-discovered replacement -// target gets its own small, self-contained `before`/`after` pair — the -// target's own pre-ASAP sub-DAG, and either the winning `SummaryNode` or the -// winning rewritten `QueryExpr`, both of which *are* fully representable -// today via [`export`]/[`export_summary`] as-is. - -/// One flattened post-ASAP node — the [`SummaryExpr`] analogue of -/// [`DAGNode`]. `detail` holds this node's own scalar fields (the summarized -/// column, the summary family, grouping strategy, sketch-query kind, …) — -/// everything except its `SummaryNode` children, which live in `children` -/// instead. -/// -/// Unlike [`DAGNode`], this carries no `hash`/`source_expr` pair: nothing in -/// this module ever needs to re-identify a particular `SummaryDAGNode` the -/// way `DAGNode::hash` lets a higher layer re-identify a pre-ASAP node (a -/// `SummaryNode` is always freshly exported for exactly one -/// [`TargetReplacementAfter::Summary`] site, never matched back against a -/// separately-exported DAG the way pre-ASAP notes are). -/// -/// Several of `SummaryExpr`'s own fields (`FieldDataType`, -/// `GroupingStrategy`, `SketchStatistic`) derive neither `Serialize` nor -/// `Deserialize` in `asap_types::post_asap` — they carry no reporting -/// obligation there, since nothing before this module ever needed to -/// serialize a post-ASAP node. Rather than adding `Serialize` impls to -/// `post_asap`'s own core types purely for this devtools-facing export (a -/// change to that module's own public API contract, out of scope for a -/// reporting concern), this module renders those particular fields into -/// `detail` via their `Debug` formatting instead — human-readable, and -/// sufficient for the display purpose `detail` exists for on every other -/// node in this file (see [`DAGNode::detail`]'s own doc), at the cost of -/// those particular fields being opaque strings rather than structured JSON -/// on the `SummaryDAGNode` side of the export. + +/// One flattened node of a [`SummaryDAG`] — the same node as a +/// [`DAGNode`], in the shape the summary-maintenance consumers read: +/// snake_case `kind`, the accuracy guarantee as its own field, no +/// hash/annotation seams. #[derive(Debug, Clone, Serialize)] pub struct SummaryDAGNode { pub id: u32, - /// The `SummaryExpr` variant name (e.g. `"SummaryAgg"`). + /// The operator variant name in snake_case (e.g. `"summary_agg"`, + /// `"scan"`) — see [`snake_case_kind`]. pub kind: &'static str, /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, - /// Child node ids, in the variant's field order (e.g. `SummaryJoin` is - /// `[outer, inner]`). + /// [`OperatorNode::schema`] as JSON. + #[serde(skip_serializing_if = "Option::is_none")] + pub schema: Option, + /// Child node ids in [`OperatorNode::children`] order. pub children: Vec, /// The value's machine-readable accuracy guarantee (issue #172) — - /// [`SummaryNode::guarantee`] serialized structurally (metric, symbolic + /// [`OperatorNode::guarantee`] serialized structurally (metric, symbolic /// bound, failure probability, provenance including any budget /// allocation), not as prose. Omitted when the node carries none (raw /// summary state, or a family with no error model), so every consumer /// predating this field parses the same shape it always has. #[serde(default, skip_serializing_if = "Option::is_none")] pub guarantee: Option, + /// The exported node itself, so a caller annotating the dag can find + /// a node by `Rc` pointer identity rather than by walk order. Not part + /// of the JSON format. Always `Some`. + #[serde(skip)] + pub source_node: Option>, } /// One accuracy-illegal candidate a higher layer's search refused for a @@ -378,239 +353,14 @@ pub struct TargetRejection { pub error: AccuracyError, } -/// One post-ASAP `SummaryNode` DAG, flattened the same way [`ExportDAG`] -/// flattens a pre-ASAP `QueryExpr` DAG. +/// A DAG flattened into [`SummaryDAGNode`]s — the same dag [`ExportDAG`] +/// holds, in the summary-maintenance consumers' node shape. #[derive(Debug, Clone, Serialize)] pub struct SummaryDAG { pub nodes: Vec, pub root: u32, } -/// Flatten a [`SummaryNode`] the same way [`export`] flattens a `QueryExpr` -/// — post-order, one [`SummaryDAGNode`] per [`SummaryExpr`] variant, no -/// memoization of repeated `Rc` references (a shared -/// sub-expression reachable through two parents is flattened twice, into two -/// separate node entries — the same "this is a flattened DAG view, not a -/// pointer-identity-preserving DAG" behavior [`build`] already has for -/// `QueryExpr`). -/// -/// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP sub-DAG beneath it -/// as a nested [`ExportDAG`] (via [`export(inner)`](export)) inside its own -/// `detail` field (`{"pre_asap_sub_dag": }`) rather than trying to -/// flatten it into this same node list — [`DAGNode`] and [`SummaryDAGNode`] -/// are different types with different id spaces, so mixing them into one -/// `Vec` isn't type-safe; nesting is. `label` for a `KeepPreAsap` node is -/// `format!("KeepPreAsap({kind})")`, where `kind` is the inner sub-DAG's own -/// top-level `DAGNode::kind`. -pub fn export_summary(node: &SummaryNode) -> SummaryDAG { - let mut nodes = Vec::new(); - let root = build_summary(node, &mut nodes); - SummaryDAG { nodes, root } -} - -fn push_summary_node( - nodes: &mut Vec, - kind: &'static str, - label: String, - detail: serde_json::Value, - children: Vec, - guarantee: Option, -) -> u32 { - let id = nodes.len() as u32; - nodes.push(SummaryDAGNode { - id, - kind, - label, - detail, - children, - guarantee, - }); - id -} - -/// A short, human-readable label for a [`crate::post_asap::FieldDataType`] -/// (e.g. `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for -/// [`SummaryDAGNode::label`] text on a `SummaryAgg`/`SummaryJoin` node. Not -/// exhaustive prose (mirrors `asap_aware_mapping::replacement::describe_intent`'s -/// own "this is a label, not a decision" stance) — every variant is covered, -/// but via `Debug` for the inner kind rather than hand-written prose per -/// algorithm. -fn family_label(family: &crate::post_asap::FieldDataType) -> String { - use crate::post_asap::FieldDataType; - match family { - FieldDataType::Plain(dtype) => format!("Plain({dtype:?})"), - FieldDataType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), - FieldDataType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), - FieldDataType::Sample(kind, _) => format!("Sample({kind:?})"), - FieldDataType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), - FieldDataType::StatModel(kind, _) => format!("StatModel({kind:?})"), - } -} - -/// `(kind, label, detail)` for every [`SummaryExpr`] variant *except* -/// [`SummaryExpr::KeepPreAsap`] — that variant has no `SummaryDAGNode`/ -/// `DAGNode` of its own (see [`build_summary`]/[`build_summary_hybrid`], its -/// only two callers, both of which special-case it before ever reaching -/// this function). Factored out so [`build_summary`] (nests a `KeepPreAsap` -/// leaf's pre-ASAP sub-DAG as its own [`SummaryDAG`]) and -/// [`build_summary_hybrid`] (splices that same sub-DAG directly into a -/// shared [`ExportDAG`] node list — see [`export_post_asap`]) can't drift -/// apart on how every *other* variant's own shape is described, since -/// nothing about that description differs between the two. -macro_rules! define_summary_kind_tags { - ($($pattern:pat => $tag:literal),+ $(,)?) => { - #[cfg(test)] - const SUMMARY_KIND_TAGS: &[&str] = &[$($tag),+]; - - fn summary_kind_tag(expr: &SummaryExpr) -> &'static str { - match expr { - SummaryExpr::KeepPreAsap(_) => unreachable!( - "summary_kind_tag's callers special-case KeepPreAsap" - ), - $($pattern => $tag),+ - } - } - }; -} - -define_summary_kind_tags! { - SummaryExpr::BinaryOp { .. } => "SummaryBinaryOp", - - SummaryExpr::ValueOperation { .. } => "ValueOperation", - SummaryExpr::RelationalJoin { .. } => "RelationalJoin", - SummaryExpr::SummaryAgg { .. } => "SummaryAgg", - SummaryExpr::SummaryJoin { .. } => "SummaryJoin", - SummaryExpr::SummarySubtract { .. } => "SummarySubtract", - SummaryExpr::SummaryDelete { .. } => "SummaryDelete", - SummaryExpr::SummaryEstimate { .. } => "SummaryEstimate", - SummaryExpr::SummaryMerge { .. } => "SummaryMerge", -} - -fn summary_shape(expr: &SummaryExpr) -> (&'static str, String, serde_json::Value) { - let kind = summary_kind_tag(expr); - match expr { - SummaryExpr::KeepPreAsap(_) => { - unreachable!("summary_shape's callers special-case KeepPreAsap before calling it") - } - SummaryExpr::BinaryOp { operator, .. } => { - let label = format!("BinaryOp({:?})", operator.kind); - let detail = serde_json::json!({ - "kind": format!("{:?}", operator.kind), - "vector_match": operator.vector_match, - }); - (kind, label, detail) - } - - SummaryExpr::ValueOperation { - operation, timing, .. - } => ( - kind, - format!("ValueOperation({operation:?})"), - serde_json::json!({ - "operation": format!("{operation:?}"), - "timing": timing.as_str(), - }), - ), - SummaryExpr::RelationalJoin { - kind: join_kind, - pred, - .. - } => ( - kind, - format!("RelationalJoin({join_kind:?})"), - serde_json::json!({ "join_kind": join_kind, "predicate": pred }), - ), - SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - .. - } => { - let label = format!("SummaryAgg({})", family_label(family)); - let detail = serde_json::json!({ - "family": format!("{family:?}"), - "input": input, - "reduction": reduction, - "grouping": format!("{grouping:?}"), - }); - (kind, label, detail) - } - SummaryExpr::SummaryJoin { key, family, .. } => { - let label = format!("SummaryJoin({})", family_label(family)); - let detail = serde_json::json!({ - "key": key, - "family": format!("{family:?}"), - }); - (kind, label, detail) - } - SummaryExpr::SummarySubtract { .. } => { - (kind, "SummarySubtract".into(), serde_json::json!({})) - } - SummaryExpr::SummaryDelete { key, .. } => { - let detail = serde_json::json!({ "key": key }); - (kind, "SummaryDelete".into(), detail) - } - SummaryExpr::SummaryEstimate { query, .. } => { - let label = format!("SummaryEstimate({query:?})"); - let detail = serde_json::json!({ "query": format!("{query:?}") }); - (kind, label, detail) - } - SummaryExpr::SummaryMerge { children, .. } => { - let label = format!("SummaryMerge({} children)", children.len()); - (kind, label, serde_json::json!({})) - } - } -} - -/// `expr`'s own `Rc` children, in the variant's field order -/// (e.g. `SummaryJoin` is `[outer, inner]`) — empty for -/// [`SummaryExpr::KeepPreAsap`], which has no `SummaryNode` children at all -/// (only a boxed pre-ASAP `QueryExpr`). Shared by [`build_summary`] and -/// [`build_summary_hybrid`] for the same reason [`summary_shape`] is. -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { - match expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], - - SummaryExpr::ValueOperation { child, .. } => vec![child], - SummaryExpr::RelationalJoin { left, right, .. } => vec![left, right], - SummaryExpr::SummaryAgg { child, .. } => vec![child], - SummaryExpr::SummaryJoin { outer, inner, .. } => vec![outer, inner], - SummaryExpr::SummarySubtract { left, right } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryMerge { children, .. } => children.iter().collect(), - } -} - -/// Recursively flatten `node`, appending [`SummaryDAGNode`]s to `nodes` in -/// post-order (children pushed before their parent), and return the pushed -/// root's id. Exhaustive over every [`SummaryExpr`] variant, matching this -/// file's own exhaustive style for `QueryExpr` in [`build`]. -fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { - if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - let pre_asap_sub_dag = export(inner); - let inner_kind = pre_asap_sub_dag.nodes[pre_asap_sub_dag.root as usize].kind; - let label = format!("KeepPreAsap({inner_kind})"); - let detail = serde_json::json!({ "pre_asap_sub_dag": pre_asap_sub_dag }); - return push_summary_node( - nodes, - "KeepPreAsap", - label, - detail, - vec![], - node.guarantee.clone(), - ); - } - let children: Vec = summary_children(&node.expr) - .into_iter() - .map(|child| build_summary(child, nodes)) - .collect(); - let (kind, label, detail) = summary_shape(&node.expr); - push_summary_node(nodes, kind, label, detail, children, node.guarantee.clone()) -} - /// One replacement site a higher layer (the `dag_export` binary) found by /// running `asap_aware_mapping::replacement::search_workload_with` + /// `CandidateLogicalASAPDAGs::cost_sorted` and picking the best-ranked candidate for one @@ -624,7 +374,7 @@ pub struct TargetReplacement { /// so renderers can explain a clicked post-ASAP node without guessing by /// label, hash, or DAG shape. pub decision_id: u32, - /// Id of the [`DAGNode`] (in this query's own `DAG.nodes`, i.e. the + /// Id of the [`DAGNode`] (in this query's own `dag.nodes`, i.e. the /// [`NamedDAG`] this `TargetReplacement` is attached to) this /// replacement's `before` sub-DAG is rooted at. pub target_pre_id: u32, @@ -648,7 +398,7 @@ pub struct TargetReplacement { /// doesn't estimate a numeric cost for this candidate shape (see that /// field's own doc upstream). pub cost: f64, - /// The target's own pre-ASAP sub-DAG, before replacement — literally + /// The target's own sub-DAG, before replacement — literally /// `export(target)` for the `TargetSubDAGCandidates`'s own `target`, reused as-is. pub before: ExportDAG, pub after: TargetReplacementAfter, @@ -668,8 +418,10 @@ pub struct TargetReplacement { } /// What a [`TargetReplacement`] became — either a genuine post-ASAP binding -/// or a still-pre-ASAP-shaped structural rewrite, mirroring -/// `asap_aware_mapping::replacement::Replacement`'s own two variants. +/// or a still-relational structural rewrite, mirroring +/// `asap_aware_mapping::replacement::Replacement`'s own two variants. Both +/// carry an ordinary [`ExportDAG`]: the unified IR renders a summary sub-DAG +/// and a rewritten relational sub-DAG through the same [`export`]. /// /// Serializes as `{"kind": "Summary"|"Rewrite", "DAG": {...}}` (serde's /// adjacently-tagged representation for a `#[serde(tag = "kind", content = @@ -680,73 +432,83 @@ pub struct TargetReplacement { #[serde(tag = "kind", content = "dag")] pub enum TargetReplacementAfter { /// A `Replacement::Summary` candidate — a genuine post-ASAP binding. - Summary(SummaryDAG), - /// A `Replacement::Rewrite` candidate — still pre-ASAP shaped (CSE + Summary(ExportDAG), + /// A `Replacement::Rewrite` candidate — still relational (CSE /// share/recompute, `AvgToSumOverCountStrategy`, and `RollupStrategy` - /// all produce this kind), so this reuses [`ExportDAG`]/[`export`] too, - /// not a new type. + /// all produce this kind). Rewrite(ExportDAG), } -/// Flatten `expr` into a [`ExportDAG`]. -pub fn export(expr: &QueryExpr) -> ExportDAG { - let mut nodes = Vec::new(); - // One cache for the whole export — persisted across every `build`/ - // `push_node` call, not reset per node, so `structural_hash` memoizes - // real work across this pass instead of re-walking an already-hashed - // shared descendant once per node that references it. - let mut cache = HashCache::new(); - // No substitution: an ordinary pre-ASAP export never splices anything - // in — see `build`'s own doc for why it always takes a `find_winner` - // callback regardless (so `export_post_asap` can share this exact - // per-variant traversal instead of duplicating it). - let root = build(expr, &mut nodes, &mut cache, &mut |_| None); - ExportDAG { - nodes, - root, - edge_annotations: Vec::new(), - } -} - -/// What a higher layer found for one specific pre-ASAP node when building a -/// merged post-ASAP DAG via [`export_post_asap`] — see that function's own -/// doc for the full design. `asap_types` has no opinion on *how* this is +/// What a higher layer found for one specific node when building a merged +/// post-ASAP dag via [`export_post_asap`] — see that function's own doc +/// for the full design. `asap_types` has no opinion on *how* this is /// decided (that's `asap_aware_mapping::replacement::search_workload_with` + /// `CandidateLogicalASAPDAGs::cost_sorted`'s job, a higher layer, exactly the layering rule /// [`DAGNode::notes`] already states); it only defines the shape a decision -/// comes back in. +/// comes back in. Both variants render identically (one IR, one builder); +/// they are kept apart so the caller's `Replacement` maps one-to-one. #[derive(Debug, Clone)] pub enum PostAsapSubstitution { /// This exact node has a winning `Replacement::Rewrite` — keep building - /// from `.0` instead of the original node. Still pre-ASAP shaped, so - /// [`build`] renders it via the same ordinary `DAGNode` path — see - /// [`build`]'s own doc for why `.0`'s own top level is rendered without - /// re-querying `find_winner` on it (its descendants still are). + /// from `replacement` instead of the original node. Rewrite { - replacement: Rc, + replacement: Rc, decision: DAGDecision, }, - /// This exact node has a winning `Replacement::Summary` — switch to - /// rendering `.0`'s bound `SummaryNode` shape from here down, via - /// [`build_summary_hybrid`]. + /// This exact node has a winning `Replacement::Summary` — keep building + /// from `replacement` (a summary-bound sub-DAG) instead of the original + /// node. Summary { - replacement: Rc, + replacement: Rc, decision: DAGDecision, }, } +/// Flatten the DAG rooted at `root` into a [`ExportDAG`]: one [`DAGNode`] +/// per unique reachable node, children pushed before their parents. +pub fn export(root: &Rc) -> ExportDAG { + let mut no_substitution = |_: &Rc| None; + let mut builder = Builder::new(&mut no_substitution); + let root = builder.build(root); + builder.finish(root) +} + +/// Flatten the DAG rooted at `node` into a [`SummaryDAG`] — the same +/// nodes [`export`] produces, in the [`SummaryDAGNode`] shape (snake_case +/// `kind`, `guarantee` as its own field). +pub fn export_summary(node: &Rc) -> SummaryDAG { + let dag = export(node); + let nodes = dag + .nodes + .into_iter() + .map(|node| { + let source = node + .source_node + .expect("every exported node carries its source"); + SummaryDAGNode { + id: node.id, + kind: snake_case_kind(&source.operator), + label: node.label, + detail: node.detail, + schema: node.schema, + children: node.children, + guarantee: source.guarantee.clone(), + source_node: Some(source), + } + }) + .collect(); + SummaryDAG { + nodes, + root: dag.root, + } +} + /// Build one merged "whole query, but post-ASAP" [`ExportDAG`] by walking -/// `root`'s ordinary pre-ASAP shape and, at every node, asking `find_winner` -/// whether *that exact node* has a winning replacement — if so, splicing -/// the replacement's own shape in at that position instead, in the very -/// same flattened node list (not a nested sub-DAG the way -/// [`TargetReplacement::before`]/`::after` — small, independent, per-site -/// before/after pairs — already do; see this file's "Post-ASAP replacement -/// export" section doc for why *that* design doesn't attempt a single -/// whole-query composite, and why this one can: this is a synthetic -/// id/edge list, the same kind of thing [`ExportDAG`] already is for the -/// pre-ASAP side, not a real `QueryExpr`/`SummaryNode` value with a type -/// system to satisfy). +/// `root` and, at every node, asking `find_winner` whether *that exact +/// node* has a winning replacement — if so, splicing the replacement's own +/// sub-DAG in at that position instead, in the very same flattened node +/// list (not a nested sub-dag the way [`TargetReplacement::before`]/ +/// `::after` — small, independent, per-site before/after pairs — do). /// /// `find_winner` is the whole layering seam: `asap_types` never runs /// `asap_aware_mapping::replacement::search_workload_with` or @@ -758,7 +520,7 @@ pub enum PostAsapSubstitution { /// for [`TargetReplacement`] discovery, and passes it in here unchanged. /// /// `find_winner` is deliberately consulted only once per node, at the -/// moment [`build`] first reaches it — **not** re-consulted on a +/// moment the builder first reaches it — **not** re-consulted on a /// substitution's own immediate top level (only on that substitution's /// *descendants*, which get an ordinary fresh call same as any other node). /// This matters for correctness, not just efficiency: @@ -770,225 +532,189 @@ pub enum PostAsapSubstitution { /// re-query at exactly that one level is what makes this termination-safe /// for every registered strategy, not just the ones that happen not to /// return the target itself as a candidate. +/// +/// Every node a substitution introduced carries the substitution's +/// [`DAGDecision`] (`role = "replacement_root"` on the spliced-in root, +/// `"replacement_region"` on its newly exported descendants); a descendant +/// that was already exported before the splice (a shared input the +/// replacement reuses) keeps whatever it already had. pub fn export_post_asap( - root: &QueryExpr, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, + root: &Rc, + find_winner: &mut dyn FnMut(&Rc) -> Option, ) -> ExportDAG { - let mut nodes = Vec::new(); - let mut cache = HashCache::new(); - let root_id = build(root, &mut nodes, &mut cache, find_winner); - deduplicate_pointer_shared_nodes(nodes, root_id) + let mut builder = Builder::new(find_winner); + let root = builder.build(root); + builder.finish(root) } -fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> ExportDAG { - let mut by_source_ptr = HashMap::::new(); - let mut old_to_new = vec![0_u32; nodes.len()]; - let mut deduplicated = Vec::with_capacity(nodes.len()); - for mut node in nodes { - node.children = node - .children - .into_iter() - .map(|child| old_to_new[child as usize]) - .collect(); - if let Some(existing) = node - .source_ptr - .and_then(|source_ptr| by_source_ptr.get(&source_ptr).copied()) - { - old_to_new[node.id as usize] = existing; - continue; - } - let old_id = node.id; - let new_id = deduplicated.len() as u32; - node.id = new_id; - if let Some(source_ptr) = node.source_ptr { - by_source_ptr.insert(source_ptr, new_id); +/// The one flattening pass behind every entry point. Nodes are memoized by +/// `Rc` pointer identity: a node reached from several parents (an operator +/// input shared with a scalar reference, say) is exported once. +struct Builder<'a> { + nodes: Vec, + /// `Rc::as_ptr` of every node already exported (or substituted) → its id. + ids: HashMap<*const OperatorNode, u32>, + /// One cache for the whole export — persisted across every node, not + /// reset per node, so `structural_hash` memoizes real work across this + /// pass instead of re-walking an already-hashed shared descendant once + /// per node that references it. + cache: HashCache, + find_winner: &'a mut dyn FnMut(&Rc) -> Option, +} + +impl<'a> Builder<'a> { + fn new( + find_winner: &'a mut dyn FnMut(&Rc) -> Option, + ) -> Self { + Self { + nodes: Vec::new(), + ids: HashMap::new(), + cache: HashCache::new(), + find_winner, } - old_to_new[old_id as usize] = new_id; - deduplicated.push(node); } - ExportDAG { - nodes: deduplicated, - root: old_to_new[root as usize], - edge_annotations: Vec::new(), + fn finish(self, root: u32) -> ExportDAG { + ExportDAG { + nodes: self.nodes, + root, + edge_annotations: Vec::new(), + } } -} -macro_rules! define_query_kind_tags { - ($($pattern:pat => $tag:literal),+ $(,)?) => { - #[cfg(test)] - const QUERY_KIND_TAGS: &[&str] = &[$($tag),+]; - - fn kind_tag(expr: &QueryExpr) -> &'static str { - match expr { - $($pattern => $tag),+, - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => unreachable!( - "kind_tag reached a scalar QueryExpr variant directly: {other:?}" - ), + /// Export `node` (or, when `find_winner` has a substitution for it, the + /// substitution's sub-DAG in its place) and return its id. + fn build(&mut self, node: &Rc) -> u32 { + let ptr = Rc::as_ptr(node); + if let Some(&id) = self.ids.get(&ptr) { + return id; + } + let (replacement, decision) = match (self.find_winner)(node) { + None => return self.build_node(node), + Some(PostAsapSubstitution::Rewrite { + replacement, + decision, + }) + | Some(PostAsapSubstitution::Summary { + replacement, + decision, + }) => (replacement, decision), + }; + let first = self.nodes.len(); + let root = self.build_node(&replacement); + for exported in &mut self.nodes[first..] { + if exported.decision.is_none() { + let mut node_decision = decision.clone(); + node_decision.role = if exported.id == root { + "replacement_root" + } else { + "replacement_region" + }; + exported.decision = Some(node_decision); } } - }; -} - -define_query_kind_tags! { - QueryExpr::Scan { .. } => "Scan", - QueryExpr::PromqlScalarBridge(_) => "PromqlScalarBridge", - QueryExpr::EvalTimestamp => "EvalTimestamp", - QueryExpr::CurrentTimestamp => "CurrentTimestamp", - QueryExpr::PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", - QueryExpr::PromqlScalarFromVector(_) => "PromqlScalarFromVector", - QueryExpr::PromqlRelabel { .. } => "PromqlRelabel", - QueryExpr::PromqlInfoEnrich { .. } => "PromqlInfoEnrich", - QueryExpr::PromqlSeriesSample { .. } => "PromqlSeriesSample", - QueryExpr::Filter { .. } => "Filter", - QueryExpr::Project { .. } => "Project", - QueryExpr::Aggregate { .. } => "Aggregate", - QueryExpr::Dedup { .. } => "Dedup", - QueryExpr::Concat { .. } => "Concat", - QueryExpr::Join { .. } => "Join", - QueryExpr::SetOp { .. } => "SetOp", - QueryExpr::Sort { .. } => "Sort", - QueryExpr::Limit { .. } => "Limit", - QueryExpr::PromqlSubquery { .. } => "PromqlSubquery", - QueryExpr::TimeRange { .. } => "TimeRange", - QueryExpr::TimeShift { .. } => "TimeShift", - QueryExpr::SQLWindowFunc { .. } => "SQLWindowFunc", - QueryExpr::BinaryOp { .. } => "BinaryOp", -} - -/// Push one flattened node for `expr`. `expr` is the *whole* sub-DAG this -/// node represents (not just its own fields) — `hash` is -/// [`structural_hash(expr)`](structural_hash), the identical function and -/// the identical input `InternTable::intern` would hash for this same -/// sub-DAG, so this node's `hash` matches what `cse::share_common_sub_dags` -/// would bucket it under. `kind` is [`kind_tag(expr)`](kind_tag), not a -/// caller-supplied argument — see that function's doc for why. -fn push_node( - nodes: &mut Vec, - expr: &QueryExpr, - cache: &mut HashCache, - label: String, - detail: serde_json::Value, - children: Vec, -) -> u32 { - let id = nodes.len() as u32; - let hash = Some(structural_hash(expr, cache)); - nodes.push(DAGNode { - id, - kind: kind_tag(expr), - label, - detail, - schema: expr - .output_schema() - .ok() - .and_then(|schema| serde_json::to_value(schema).ok()), - children, - workload_node_id: None, - hash, - source_expr: Some(expr.clone()), - source_ptr: Some(expr as *const QueryExpr as usize), - notes: Vec::new(), - decision: None, - }); - id -} + // The original node now resolves to the substitution: another + // parent of the same `Rc` reuses the spliced-in sub-DAG. + self.ids.insert(ptr, root); + root + } -/// Push one flattened node with no corresponding pre-ASAP `QueryExpr` at -/// all — a post-ASAP-originated node inside [`export_post_asap`]'s merged -/// DAG (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). -/// `hash`/`source_expr`-based re-identification (see [`DAGNode::hash`]'s own -/// doc) has no meaning for a node with no `QueryExpr` behind it, so this -/// pushes a fixed placeholder hash (`0`) and `source_expr: None` rather than -/// inventing a hash over `SummaryExpr` (which, unlike `QueryExpr`, has no -/// [`structural_hash`]-equivalent function at all — see [`SummaryDAGNode`]'s -/// own doc on why `SummaryExpr`'s fields don't even derive `Hash`/`PartialEq` -/// consistently enough to build one). -fn push_summary_originated_node( - nodes: &mut Vec, - kind: &'static str, - label: String, - detail: serde_json::Value, - children: Vec, -) -> u32 { - let id = nodes.len() as u32; - nodes.push(DAGNode { - id, - kind, - label, - detail, - schema: None, - children, - workload_node_id: None, - hash: None, - source_expr: None, - source_ptr: None, - notes: Vec::new(), - decision: None, - }); - id + /// Export `node` itself (no substitution check at this level; children + /// still go through [`Self::build`]) and return its id. + fn build_node(&mut self, node: &Rc) -> u32 { + let ptr = Rc::as_ptr(node); + if let Some(&id) = self.ids.get(&ptr) { + return id; + } + let children: Vec = node.children().into_iter().map(|c| self.build(c)).collect(); + let (label, mut detail) = shape(node, &self.ids); + if let serde_json::Value::Object(map) = &mut detail { + if let Some(timing) = node.timing { + map.insert("timing".into(), serde_json::json!(timing.as_str())); + } + if let Some(guarantee) = &node.guarantee { + if let Ok(value) = serde_json::to_value(guarantee) { + map.insert("guarantee".into(), value); + } + } + } + let hash = structural_hash(node, &mut self.cache); + self.cache.insert(ptr, hash); + let id = self.nodes.len() as u32; + self.nodes.push(DAGNode { + id, + kind: node.operator.kind_name(), + label, + detail, + schema: serde_json::to_value(&node.schema).ok(), + children, + workload_node_id: None, + hash: Some(hash), + source_node: Some(Rc::clone(node)), + source_ptr: Some(ptr as usize), + notes: Vec::new(), + decision: None, + }); + self.ids.insert(ptr, id); + id + } } -/// The [`build_summary`]/[`build_summary_hybrid`] counterpart of [`build`] -/// for a bound [`SummaryNode`] reached while building -/// [`export_post_asap`]'s merged DAG: appends into the *same* `nodes: -/// Vec` list `build` itself is filling, instead of a separate -/// [`SummaryDAG`]. A `KeepPreAsap(inner)` leaf recurses back into -/// [`build`] on `inner` (the general pre-ASAP entry, `find_winner` included) -/// rather than nesting a `{"pre_asap_sub_dag": ...}` blob the way -/// [`build_summary`] does — so the merged DAG reads as one seamless DAG -/// with no dead ends, and so a target reachable underneath a `KeepPreAsap` -/// wrapper (a nested aggregate a strategy independently found a -/// replacement for, say) still gets spliced in correctly. -fn build_summary_hybrid( - node: &SummaryNode, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { - if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - return build(inner, nodes, cache, find_winner); - } - let children: Vec = summary_children(&node.expr) - .into_iter() - .map(|child| build_summary_hybrid(child, nodes, cache, find_winner)) - .collect(); - let (kind, label, mut detail) = summary_shape(&node.expr); - // The merged DAG's `DAGNode` has no dedicated guarantee field (it is - // the pre-ASAP node shape); the guarantee rides in `detail` under the - // same key/shape `SummaryDAGNode::guarantee` uses, additively. - if let Some(guarantee) = &node.guarantee { - if let (serde_json::Value::Object(map), Ok(value)) = - (&mut detail, serde_json::to_value(guarantee)) - { - map.insert("guarantee".into(), value); - } +/// [`Operator::kind_name`] in snake_case, for [`SummaryDAGNode::kind`]. +/// Exhaustive so a new operator variant fails to compile here until it is +/// named. +fn snake_case_kind(operator: &Operator) -> &'static str { + match operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Scan { .. } => "scan", + NonASAPOp::Values { .. } => "values", + NonASAPOp::Filter { .. } => "filter", + NonASAPOp::Project { .. } => "project", + NonASAPOp::Aggregate { .. } => "aggregate", + NonASAPOp::Join { .. } => "join", + NonASAPOp::SetOp { .. } => "set_op", + NonASAPOp::Concat { .. } => "concat", + NonASAPOp::Dedup { .. } => "dedup", + NonASAPOp::Sort { .. } => "sort", + NonASAPOp::Limit { .. } => "limit", + NonASAPOp::BinaryOp { .. } => "binary_op", + NonASAPOp::SQLWindowFunc { .. } => "sql_window_func", + NonASAPOp::TimeRange { .. } => "time_range", + NonASAPOp::TimeShift { .. } => "time_shift", + NonASAPOp::PromqlVectorFromScalar(_) => "promql_vector_from_scalar", + NonASAPOp::PromqlRelabel { .. } => "promql_relabel", + NonASAPOp::PromqlInfoEnrich { .. } => "promql_info_enrich", + NonASAPOp::PromqlSeriesSample { .. } => "promql_series_sample", + NonASAPOp::PromqlSubquery { .. } => "promql_subquery", + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { .. } => "summary_agg", + ASAPOp::SummaryEstimate { .. } => "summary_estimate", + ASAPOp::FinalizeExactAccumulator { .. } => "finalize_exact_accumulator", + ASAPOp::MaintainPopulation { .. } => "maintain_population", + ASAPOp::EvaluatePopulation { .. } => "read_population", + ASAPOp::SummaryMerge { .. } => "summary_merge", + ASAPOp::SummarySubtract { .. } => "summary_subtract", + ASAPOp::SummaryDelete { .. } => "summary_delete", + ASAPOp::SummaryJoin { .. } => "summary_join", + ASAPOp::Extension { .. } => "extension", + }, } - let id = push_summary_originated_node(nodes, kind, label, detail, children); - nodes[id as usize].schema = Some(summary_schema_json(&node.schema)); - id } -fn summary_schema_json(schema: &crate::post_asap::Schema) -> serde_json::Value { - serde_json::json!({ - "fields": schema.fields.iter().map(|field| serde_json::json!({ - "name": field.name, - "dtype": format!("{:?}", field.dtype), - "nullable": field.nullable, - })).collect::>(), - "time_index": schema.time_index, - }) +/// A short, human-readable label for a [`FieldDataType`] (e.g. +/// `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for the label text on a +/// `SummaryAgg`/`SummaryJoin` node. Every variant is covered, via `Debug` +/// for the inner kind rather than hand-written prose per algorithm. +fn family_label(family: &FieldDataType) -> String { + match family { + FieldDataType::Plain(dtype) => format!("Plain({dtype:?})"), + FieldDataType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), + FieldDataType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), + FieldDataType::Sample(kind, _) => format!("Sample({kind:?})"), + FieldDataType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), + FieldDataType::StatModel(kind, _) => format!("StatModel({kind:?})"), + } } fn source_label(source: &Source) -> String { @@ -998,407 +724,336 @@ fn source_label(source: &Source) -> String { } } -/// Recursively flatten `expr`, appending nodes to `nodes` in post-order -/// (children pushed before their parent), and return the id of the pushed -/// root node. Exhaustive over every **operator** `QueryExpr` variant — a new -/// one fails to compile here until this match is extended, matching the rest -/// of the IR's exhaustive-match style (e.g. `output_schema`). The scalar -/// variants (issue #205) are never passed to `build` directly: every operator -/// arm that carries one (`Filter.pred`, `Project.cols`, `Aggregate.having`, …) -/// serializes it as opaque `detail` JSON via `Predicate`/`ProjectItem`/ -/// `AggIntent`'s own `Serialize` impl, same as before the merge — a scalar -/// sub-DAG was never a separate DAG node, so this doesn't change that. -/// -/// `find_winner` is [`export_post_asap`]'s substitution seam, threaded -/// through every recursive call (including [`export`]'s own, which always -/// passes a closure that returns `None`) so both entry points share this -/// exact traversal instead of maintaining two copies of it. `build` itself -/// only ever calls `find_winner` once, right here at the top, before -/// dispatching into the ordinary per-variant match below — see -/// [`export_post_asap`]'s own doc for why a substitution's own immediate -/// result is rendered via that match directly (recursing into its children -/// through `build` again, so *they* still get a fresh `find_winner` call) -/// rather than by looping back through this check a second time. -fn build( - expr: &QueryExpr, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { - match find_winner(expr) { - Some(PostAsapSubstitution::Rewrite { - replacement, - decision, - }) => { - let first = nodes.len(); - let root = build_no_recheck(&replacement, nodes, cache, find_winner); - for node in &mut nodes[first..] { - if node.decision.is_none() { - let mut node_decision = decision.clone(); - node_decision.role = if node.id == root { - "replacement_root" - } else { - "replacement_region" - }; - node.decision = Some(node_decision); - } +/// `(label, detail)` for one node: its own fields, never its children. +/// Exhaustive over every operator variant — a new one fails to compile +/// here until this match is extended, matching the rest of the IR's +/// exhaustive-match style. Scalar expressions are rendered through +/// [`scalar_json`] with `ids` resolving their operator references. +fn shape( + node: &OperatorNode, + ids: &HashMap<*const OperatorNode, u32>, +) -> (String, serde_json::Value) { + let scalar = |expr: &ScalarExpr| scalar_json(expr, ids); + let scalars = + |exprs: &[ScalarExpr]| -> Vec { exprs.iter().map(scalar).collect() }; + let predicate = |pred: &crate::ir::Predicate| scalar(&pred.0); + let sort_keys = |keys: &[crate::ir::SortKey]| -> Vec { + keys.iter() + .map(|key| { + serde_json::json!({ + "expr": scalar(&key.expr), + "ascending": key.ascending, + "nulls_first": key.nulls_first, + }) + }) + .collect() + }; + match &node.operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Scan { + source, + predicates, + schema, + } => ( + format!("Scan({})", source_label(source)), + serde_json::json!({ + "source": source, + "predicates": predicates.iter().map(predicate).collect::>(), + "schema": schema, + }), + ), + NonASAPOp::Values { rows, schema } => ( + format!("Values({} rows)", rows.len()), + serde_json::json!({ + "rows": rows.iter().map(|row| scalars(row)).collect::>(), + "schema": schema, + }), + ), + NonASAPOp::Filter { pred, .. } => ( + "Filter".into(), + serde_json::json!({ "pred": predicate(pred) }), + ), + NonASAPOp::Project { + cols, qualifier, .. + } => ( + format!("Project({} cols)", cols.len()), + serde_json::json!({ + "cols": cols.iter().map(|item| serde_json::json!({ + "alias": item.alias, + "expr": scalar(&item.expr), + })).collect::>(), + "qualifier": qualifier, + }), + ), + NonASAPOp::Aggregate { + reduction, + measures, + output_names, + having, + .. + } => ( + format!("Aggregate({} measures)", measures.len()), + serde_json::json!({ + "reduction": reduction, + "measures": measures, + "output_names": output_names, + "having": having.as_ref().map(predicate), + }), + ), + NonASAPOp::Join { kind, pred, .. } => ( + format!("Join({kind:?})"), + serde_json::json!({ "kind": kind, "pred": predicate(pred) }), + ), + NonASAPOp::SetOp { kind, all, .. } => ( + format!("SetOp({kind:?})"), + serde_json::json!({ "kind": kind, "all": all }), + ), + NonASAPOp::Concat { + children, + discriminator_unique_key, + } => ( + format!("Concat({} branches)", children.len()), + serde_json::json!({ "discriminator_unique_key": discriminator_unique_key }), + ), + NonASAPOp::Dedup { cols, .. } => ( + format!("Dedup({} cols)", cols.len()), + serde_json::json!({ "cols": cols }), + ), + NonASAPOp::Sort { + keys, partition_by, .. + } => ( + format!("Sort({} keys)", keys.len()), + serde_json::json!({ "keys": sort_keys(keys), "partition_by": partition_by }), + ), + NonASAPOp::Limit { + n, + offset, + partition_by, + .. + } => ( + match n { + Some(n) => format!("Limit({n})"), + None => format!("Limit(offset {offset})"), + }, + serde_json::json!({ "n": n, "offset": offset, "partition_by": partition_by }), + ), + NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } => ( + format!("BinaryOp({})", operator.kind), + serde_json::json!({ + "op": operator.kind.to_string(), + "vector_match": operator.vector_match, + "checked_relative_division": operator.checked_relative_division, + "checked_finite_division": operator.checked_finite_division, + "return_bool": return_bool, + }), + ), + NonASAPOp::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + .. + } => ( + format!("SQLWindowFunc({func:?})"), + serde_json::json!({ + "func": func, + "args": scalars(args), + "partition_by": partition_by, + "order_by": sort_keys(order_by), + "frame": frame, + "output_name": output_name, + }), + ), + NonASAPOp::TimeRange { range, kind, .. } => ( + format!("TimeRange({kind:?}, {range:?})"), + serde_json::json!({ "range": range, "kind": kind }), + ), + NonASAPOp::TimeShift { shift, .. } => { + ("TimeShift".into(), serde_json::json!({ "shift": shift })) } - return root; - } - Some(PostAsapSubstitution::Summary { - replacement, - decision, - }) => { - let first = nodes.len(); - let root = build_summary_hybrid(&replacement, nodes, cache, find_winner); - for node in &mut nodes[first..] { - if node.decision.is_none() { - let mut node_decision = decision.clone(); - node_decision.role = if node.id == root { - "replacement_root" - } else { - "replacement_region" - }; - node.decision = Some(node_decision); - } + NonASAPOp::PromqlVectorFromScalar(value) => ( + "vector()".into(), + serde_json::json!({ "value": scalar(value) }), + ), + NonASAPOp::PromqlRelabel { dst, value, .. } => ( + format!("PromqlRelabel(dst={dst})"), + serde_json::json!({ "dst": dst, "value": scalar(value) }), + ), + NonASAPOp::PromqlInfoEnrich { selector, .. } => ( + "PromqlInfoEnrich".into(), + serde_json::json!({ "selector": selector }), + ), + NonASAPOp::PromqlSeriesSample { by, kind, .. } => ( + format!("PromqlSeriesSample({kind:?})"), + serde_json::json!({ "by": by, "kind": kind }), + ), + NonASAPOp::PromqlSubquery { + range, resolution, .. + } => ( + "PromqlSubquery".into(), + serde_json::json!({ "range": range, "resolution": resolution }), + ), + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } => ( + format!("SummaryAgg({})", family_label(family)), + serde_json::json!({ + "family": format!("{family:?}"), + "input": input, + "reduction": reduction, + "grouping": format!("{grouping:?}"), + }), + ), + ASAPOp::SummaryEstimate { query, .. } => ( + format!("SummaryEstimate({query:?})"), + serde_json::json!({ "query": format!("{query:?}") }), + ), + ASAPOp::FinalizeExactAccumulator { .. } => { + ("FinalizeExactAccumulator".into(), serde_json::json!({})) } - return root; - } - None => {} + ASAPOp::MaintainPopulation { population, .. } => ( + format!("MaintainPopulation(max_k={})", population.max_k), + serde_json::json!({ "population": population }), + ), + ASAPOp::EvaluatePopulation { evaluation, .. } => ( + format!("EvaluatePopulation({evaluation:?})"), + serde_json::json!({ "evaluation": evaluation }), + ), + ASAPOp::SummaryMerge { children } => ( + format!("SummaryMerge({} children)", children.len()), + serde_json::json!({}), + ), + ASAPOp::SummarySubtract { .. } => ("SummarySubtract".into(), serde_json::json!({})), + ASAPOp::SummaryDelete { key, .. } => { + ("SummaryDelete".into(), serde_json::json!({ "key": key })) + } + ASAPOp::SummaryJoin { key, family, .. } => ( + format!("SummaryJoin({})", family_label(family)), + serde_json::json!({ "key": key, "family": format!("{family:?}") }), + ), + ASAPOp::Extension { name, .. } => ( + format!("Extension({name})"), + serde_json::json!({ "name": name }), + ), + }, } - build_no_recheck(expr, nodes, cache, find_winner) } -/// The actual per-variant match [`build`] dispatches to once it has decided -/// (by consulting `find_winner` exactly once) which `QueryExpr` value to -/// render at this position — either `expr` itself (unchanged), or a winning -/// `Replacement::Rewrite`'s own target. Every recursive call here goes back -/// through [`build`] (not this function), so every child gets its own fresh -/// `find_winner` query. -fn build_no_recheck( - expr: &QueryExpr, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { +/// `{"scalar_ref": }` for an operator node a scalar expression reads. +/// The node is one of the owning operator's children, so it has already +/// been exported by the time its parent's `detail` is built. +fn scalar_ref( + node: &Rc, + ids: &HashMap<*const OperatorNode, u32>, +) -> serde_json::Value { + serde_json::json!({ "scalar_ref": ids.get(&Rc::as_ptr(node)).copied() }) +} + +/// `expr` as JSON in `ScalarExpr`'s own serde shape (externally tagged +/// variants), except that every operator reference is rendered via +/// [`scalar_ref`] instead of inlining the referenced sub-DAG. Exhaustive so +/// a new variant fails to compile here until it is rendered. +fn scalar_json(expr: &ScalarExpr, ids: &HashMap<*const OperatorNode, u32>) -> serde_json::Value { + let sub = |e: &ScalarExpr| scalar_json(e, ids); + let list = |es: &[ScalarExpr]| -> Vec { es.iter().map(sub).collect() }; match expr { - QueryExpr::Scan { - source, - predicates, - schema, - } => { - let label = format!("Scan({})", source_label(source)); - let detail = serde_json::json!({ - "source": source, - "predicates": predicates, - "schema": schema, - }); - push_node(nodes, expr, cache, label, detail, vec![]) - } - // The bridged child is a scalar-sub-language node (issue #220), not - // an operator node `build` can recurse into — serialize it as opaque - // `detail` JSON, same as every other scalar-typed field - // (`Filter.pred`, `Project.cols`, …) rather than pushing it as a - // separate DAG node. - QueryExpr::PromqlScalarBridge(inner) => { - let detail = serde_json::json!({ "value": inner }); - push_node( - nodes, - expr, - cache, - format!("PromqlScalarBridge({inner:?})"), - detail, - vec![], - ) - } - QueryExpr::EvalTimestamp => push_node( - nodes, - expr, - cache, - "EvalTimestamp".into(), - serde_json::json!({}), - vec![], - ), - QueryExpr::CurrentTimestamp => push_node( - nodes, - expr, - cache, - "CurrentTimestamp".into(), - serde_json::json!({}), - vec![], - ), - QueryExpr::PromqlVectorFromScalar(child) => { - let c = build(child, nodes, cache, find_winner); - push_node( - nodes, - expr, - cache, - "vector()".into(), - serde_json::json!({}), - vec![c], - ) - } - QueryExpr::PromqlScalarFromVector(child) => { - let c = build(child, nodes, cache, find_winner); - push_node( - nodes, - expr, - cache, - "scalar()".into(), - serde_json::json!({}), - vec![c], - ) - } - QueryExpr::PromqlRelabel { dst, value, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "dst": dst, "value": value }); - push_node( - nodes, - expr, - cache, - format!("PromqlRelabel(dst={dst})"), - detail, - vec![c], - ) - } - QueryExpr::PromqlInfoEnrich { selector, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "selector": selector }); - push_node( - nodes, - expr, - cache, - "PromqlInfoEnrich".into(), - detail, - vec![c], - ) - } - QueryExpr::PromqlSeriesSample { by, kind, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "by": by, "kind": kind }); - push_node( - nodes, - expr, - cache, - format!("PromqlSeriesSample({kind:?})"), - detail, - vec![c], - ) - } - QueryExpr::Filter { pred, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "pred": pred }); - push_node(nodes, expr, cache, "Filter".into(), detail, vec![c]) - } - QueryExpr::Project { - cols, - qualifier, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "cols": cols, "qualifier": qualifier }); - push_node( - nodes, - expr, - cache, - format!("Project({} cols)", cols.len()), - detail, - vec![c], - ) - } - QueryExpr::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ - "reduction": reduction, - "measures": measures, - "output_names": output_names, - "filters": filters, - "having": having, - }); - push_node( - nodes, - expr, - cache, - format!("Aggregate({} measures)", measures.len()), - detail, - vec![c], - ) - } - QueryExpr::Dedup { cols, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "cols": cols }); - push_node( - nodes, - expr, - cache, - format!("Dedup({} cols)", cols.len()), - detail, - vec![c], - ) - } - QueryExpr::Concat { - children, - discriminator_unique_key, - } => { - let ids: Vec = children - .iter() - .map(|c| build(c, nodes, cache, find_winner)) - .collect(); - let label = format!("Concat({} branches)", ids.len()); - let detail = - serde_json::json!({ "discriminator_unique_key": discriminator_unique_key }); - push_node(nodes, expr, cache, label, detail, ids) - } - QueryExpr::Join { - kind, - pred, + ScalarExpr::Column(id) => serde_json::json!({ "Column": id }), + ScalarExpr::Literal(value) => serde_json::json!({ "Literal": value }), + ScalarExpr::Negative { expr, semantics } => serde_json::json!({ + "Negative": { "expr": sub(expr), "semantics": semantics } + }), + ScalarExpr::Compare { left, + op, right, - } => { - let l = build(left, nodes, cache, find_winner); - let r = build(right, nodes, cache, find_winner); - let detail = serde_json::json!({ "kind": kind, "pred": pred }); - push_node( - nodes, - expr, - cache, - format!("Join({kind:?})"), - detail, - vec![l, r], - ) - } - QueryExpr::SetOp { - kind, - all, + semantics, + } => serde_json::json!({ + "Compare": { + "left": sub(left), + "op": op, + "right": sub(right), + "semantics": semantics, + } + }), + ScalarExpr::BoolAnd(parts) => serde_json::json!({ "BoolAnd": list(parts) }), + ScalarExpr::BoolOr(parts) => serde_json::json!({ "BoolOr": list(parts) }), + ScalarExpr::Not(e) => serde_json::json!({ "Not": sub(e) }), + ScalarExpr::IsNull(e) => serde_json::json!({ "IsNull": sub(e) }), + ScalarExpr::IsNotNull(e) => serde_json::json!({ "IsNotNull": sub(e) }), + ScalarExpr::Cast { expr, to, try_cast } => serde_json::json!({ + "Cast": { "expr": sub(expr), "to": to, "try_cast": try_cast } + }), + ScalarExpr::InList { + expr, + list: items, + negated, + } => serde_json::json!({ + "InList": { "expr": sub(expr), "list": list(items), "negated": negated } + }), + ScalarExpr::FunctionCall { name, args } => serde_json::json!({ + "FunctionCall": { "name": name, "args": list(args) } + }), + ScalarExpr::Arithmetic { + op, left, right, - } => { - let l = build(left, nodes, cache, find_winner); - let r = build(right, nodes, cache, find_winner); - let detail = serde_json::json!({ "kind": kind, "all": all }); - push_node( - nodes, - expr, - cache, - format!("SetOp({kind:?})"), - detail, - vec![l, r], - ) - } - QueryExpr::Sort { - keys, - partition_by, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "keys": keys, "partition_by": partition_by }); - push_node( - nodes, - expr, - cache, - format!("Sort({} keys)", keys.len()), - detail, - vec![c], - ) - } - QueryExpr::Limit { n, offset, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "n": n, "offset": offset }); - push_node(nodes, expr, cache, format!("Limit({n})"), detail, vec![c]) - } - QueryExpr::PromqlSubquery { - range, - resolution, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "range": range, "resolution": resolution }); - push_node(nodes, expr, cache, "PromqlSubquery".into(), detail, vec![c]) - } - QueryExpr::TimeRange { range, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "range": range }); - push_node( - nodes, - expr, - cache, - format!("TimeRange({range:?})"), - detail, - vec![c], - ) - } - QueryExpr::TimeShift { shift, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "shift": shift }); - push_node(nodes, expr, cache, "TimeShift".into(), detail, vec![c]) - } - QueryExpr::SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ - "func": func, - "args": args, - "partition_by": partition_by, - "order_by": order_by, - "frame": frame, - "output_name": output_name, - }); - push_node( - nodes, - expr, - cache, - format!("SQLWindowFunc({func:?})"), - detail, - vec![c], - ) - } - QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, - } => { - let l = build(lhs, nodes, cache, find_winner); - let r = build(rhs, nodes, cache, find_winner); - let detail = serde_json::json!({ "op": op.to_string(), "vector_match": vector_match }); - push_node( - nodes, - expr, - cache, - format!("BinaryOp({op})"), - detail, - vec![l, r], - ) - } - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => { - unreachable!("dag_export::build reached a scalar QueryExpr variant directly: {other:?}") - } + semantics, + } => serde_json::json!({ + "Arithmetic": { + "op": op, + "left": sub(left), + "right": sub(right), + "semantics": semantics, + } + }), + ScalarExpr::Case { + operand, + branches, + else_expr, + } => serde_json::json!({ + "Case": { + "operand": operand.as_deref().map(sub), + "branches": branches + .iter() + .map(|(when, then)| serde_json::json!([sub(when), sub(then)])) + .collect::>(), + "else_expr": else_expr.as_deref().map(sub), + } + }), + ScalarExpr::CurrentTimestamp => serde_json::json!("CurrentTimestamp"), + ScalarExpr::EvalTimestamp => serde_json::json!("EvalTimestamp"), + ScalarExpr::PromqlScalarFromVector(node) => serde_json::json!({ + "PromqlScalarFromVector": scalar_ref(node, ids) + }), + ScalarExpr::ScalarSubquery(node) => serde_json::json!({ + "ScalarSubquery": scalar_ref(node, ids) + }), + ScalarExpr::Exists { subquery, negated } => serde_json::json!({ + "Exists": { "subquery": scalar_ref(subquery, ids), "negated": negated } + }), + ScalarExpr::InSubquery { + expr, + subquery, + negated, + } => serde_json::json!({ + "InSubquery": { + "expr": sub(expr), + "subquery": scalar_ref(subquery, ids), + "negated": negated, + } + }), } } @@ -1407,14 +1062,20 @@ mod tests { use std::rc::Rc; use super::*; + use crate::ir::operator_properties::{GroupKeys, JoinKind, Reduction}; + use crate::ir::Predicate; + use crate::post_asap::{ + BoundExpr, CompositionOperator, ErrorMetric, GroupingStrategy, GuaranteeSource, + ProbabilityExpr, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, SummaryUpdate, + }; use crate::pre_asap::agg_intent::AggIntent; - use crate::pre_asap::expr_ir::ScalarValue; - use crate::pre_asap::query_expr::{GroupKeys, Predicate, Reduction}; + use crate::pre_asap::expr_ir::{ColumnRef, ScalarValue}; use crate::pre_asap::schema::{DataType, Field, Schema}; + use crate::types::AccuracyTarget; - fn scan(table: &str, columns: Vec) -> QueryExpr { - QueryExpr::Scan { + fn scan(table: &str, columns: Vec) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: table.into(), }, @@ -1425,25 +1086,95 @@ mod tests { unique_keys: vec![], closed: true, }, - } + })) + .unwrap() } fn value_col() -> Vec { vec![Field::plain("value", DataType::Float64, false)] } + fn true_pred() -> Predicate { + Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))) + } + + fn count_agg(child: Rc) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(GroupKeys::none()), + measures: vec![AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child, + })) + .unwrap() + } + + fn join(left: Rc, right: Rc) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left, + right, + })) + .unwrap() + } + + /// A KLL `SummaryAgg` over `leaf`'s `v` column, read out as a quantile. + fn quantile_evaluation( + leaf: Rc, + guarantee: Option, + ) -> (Rc, Rc) { + let family = FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 40 }), + GroupingStrategy::default(), + ); + let agg = std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: leaf, + family: family.clone(), + input: SummaryUpdate::column(ColumnRef::Named("v".into())), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }), + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(None), + ); + let evaluation = std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: Rc::clone(&agg), + query: SketchStatistic::Quantile { q: 0.99 }, + }), + Schema::lifted( + vec![Field::plain("quantile", DataType::Float64, false)], + None, + ), + ) + .with_guarantee(guarantee), + ); + (agg, evaluation) + } + #[test] fn leaf_scan_is_a_single_node() { let dag = export(&scan("metrics", value_col())); assert_eq!(dag.nodes.len(), 1); assert_eq!(dag.root, 0); assert_eq!(dag.nodes[0].kind, "Scan"); + assert_eq!(dag.nodes[0].label, "Scan(metrics)"); assert!(dag.nodes[0].children.is_empty()); + assert!(dag.nodes[0].source_node.is_some()); } /// `export` itself never populates higher-layer annotations. Empty - /// annotations must not appear in serialized JSON, so ordinary (non-ASAP) - /// exports retain their existing shape. + /// annotations must not appear in serialized JSON, so ordinary exports + /// retain their existing shape. #[test] fn export_omits_empty_higher_layer_annotations() { let dag = export(&scan("metrics", value_col())); @@ -1460,6 +1191,10 @@ mod tests { !json.contains("decision"), "empty `decision` must be skipped, not serialized as `null`: {json}" ); + assert!( + !json.contains("source_node"), + "`source_node` is in-process only: {json}" + ); let dag_json = serde_json::to_string(&dag).unwrap(); assert!( !dag_json.contains("edge_annotations"), @@ -1473,22 +1208,29 @@ mod tests { // single child slot) share the exact same `Rc` Scan — // `export_post_asap` must merge them onto one node id. Sharing alone // is not physical cost evidence, so no edge cost may be fabricated. - let shared_scan = Rc::new(scan("metrics", value_col())); - let left_branch = QueryExpr::Dedup { - cols: vec![0], - child: Rc::clone(&shared_scan), - }; - let right_branch = QueryExpr::Limit { - n: 5, - offset: 0, - child: Rc::clone(&shared_scan), - }; - let root = QueryExpr::Concat { + let shared_scan = scan("metrics", value_col()); + let left_branch = + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: Rc::clone(&shared_scan), + })) + .unwrap(); + let right_branch = + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(5), + offset: 0, + partition_by: GroupKeys::none(), + child: Rc::clone(&shared_scan), + })) + .unwrap(); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Concat { children: vec![left_branch, right_branch], discriminator_unique_key: None, - }; + })) + .unwrap(); let dag = export_post_asap(&root, &mut |_| None); + assert_eq!(dag.nodes.len(), 4, "Scan, Dedup, Limit, Concat"); assert_eq!( dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1, @@ -1497,20 +1239,15 @@ mod tests { assert!(dag.edge_annotations.is_empty()); } - /// Regression test: a single parent referencing the same shared child - /// from two of its own operand slots at once (a `Join` whose left and - /// right sides are the exact same `Rc`, post pointer-dedup) is *one* - /// downstream consumer, not two — this must not inflate - /// produce an edge-cost annotation without explicit physical evidence. + /// A single parent referencing the same shared child from two of its + /// own operand slots at once (a `Join` whose left and right sides are + /// the exact same `Rc`) is *one* downstream consumer, not two — this + /// must not produce an edge-cost annotation without explicit physical + /// evidence. #[test] fn a_single_parent_referencing_a_shared_child_twice_is_one_consumer_not_two() { - let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { - kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::clone(&shared_scan), - right: Rc::clone(&shared_scan), - }; + let shared_scan = scan("metrics", value_col()); + let root = join(Rc::clone(&shared_scan), Rc::clone(&shared_scan)); let dag = export_post_asap(&root, &mut |_| None); assert_eq!( @@ -1518,6 +1255,7 @@ mod tests { 1, "the shared Scan must be merged onto one node, not duplicated" ); + assert_eq!(dag.nodes[dag.root as usize].children, vec![0, 0]); assert!( dag.edge_annotations.is_empty(), "a single parent referencing the same child twice is one consumer, not a genuine \ @@ -1527,38 +1265,33 @@ mod tests { } #[test] - fn export_never_produces_edge_annotations_since_it_never_shares_nodes() { - // Plain `export` (no `export_post_asap`) never deduplicates by `Rc` - // pointer identity — even a workload-level shared sub-DAG renders as - // two independent DAG nodes here, so there is nothing to annotate. - let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { - kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::clone(&shared_scan), - right: Rc::clone(&shared_scan), - }; - let dag = export(&root); - assert_eq!(dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 2); + fn export_merges_pointer_shared_nodes_but_not_equal_copies() { + // Plain `export` deduplicates by `Rc` pointer identity: the same + // `Rc` reached twice is one node ... + let shared_scan = scan("metrics", value_col()); + let dag = export(&join(Rc::clone(&shared_scan), Rc::clone(&shared_scan))); + assert_eq!(dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1); assert!(dag.edge_annotations.is_empty()); + + // ... while two structurally equal but distinct `Rc`s stay two + // nodes (with equal hashes — that is CSE's job, not the export's). + let dag = export(&join( + scan("metrics", value_col()), + scan("metrics", value_col()), + )); + let scans: Vec<_> = dag.nodes.iter().filter(|n| n.kind == "Scan").collect(); + assert_eq!(scans.len(), 2); + assert_eq!(scans[0].hash, scans[1].hash); } #[test] fn chain_preserves_shape_and_child_links() { - let expr = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::none()), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan("metrics", value_col())), - }), - }; - let dag = export(&expr); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: true_pred(), + child: count_agg(scan("metrics", value_col())), + })) + .unwrap(); + let dag = export(&root); assert_eq!(dag.nodes.len(), 3, "Filter -> Aggregate -> Scan"); let filter = &dag.nodes[dag.root as usize]; @@ -1567,6 +1300,7 @@ mod tests { let agg = &dag.nodes[filter.children[0] as usize]; assert_eq!(agg.kind, "Aggregate"); + assert_eq!(agg.label, "Aggregate(1 measures)"); assert_eq!(agg.children.len(), 1); let leaf = &dag.nodes[agg.children[0] as usize]; @@ -1576,18 +1310,76 @@ mod tests { #[test] fn merge_keeps_every_branch_as_a_child() { - let expr = QueryExpr::concat(vec![ - scan("a", value_col()), - scan("b", value_col()), - scan("c", value_col()), - ]); - let dag = export(&expr); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Concat { + children: vec![ + scan("a", value_col()), + scan("b", value_col()), + scan("c", value_col()), + ], + discriminator_unique_key: None, + })) + .unwrap(); + let dag = export(&root); assert_eq!(dag.nodes.len(), 4, "3 branches + the Concat node"); let merge = &dag.nodes[dag.root as usize]; assert_eq!(merge.kind, "Concat"); assert_eq!(merge.children.len(), 3); } + /// An operator node read from a scalar expression is a child of the + /// owning operator (after its operator inputs), and the expression's + /// `detail` points at it by id instead of inlining it. + #[test] + fn scalar_operator_references_are_children_rendered_as_scalar_refs() { + let subquery = scan("other", value_col()); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Exists { + subquery: Rc::clone(&subquery), + negated: false, + }), + child: scan("metrics", value_col()), + })) + .unwrap(); + let dag = export(&root); + assert_eq!(dag.nodes.len(), 3); + let filter = &dag.nodes[dag.root as usize]; + assert_eq!( + filter.children.len(), + 2, + "operator input, then the scalar reference" + ); + let input = &dag.nodes[filter.children[0] as usize]; + let referenced = &dag.nodes[filter.children[1] as usize]; + assert_eq!(input.label, "Scan(metrics)"); + assert_eq!(referenced.label, "Scan(other)"); + assert_eq!( + filter.detail["pred"]["Exists"]["subquery"]["scalar_ref"], + serde_json::json!(referenced.id) + ); + assert_eq!(filter.detail["pred"]["Exists"]["negated"], false); + let json = serde_json::to_string(&filter.detail).unwrap(); + assert!( + !json.contains("other"), + "the referenced sub_dag must not be inlined into detail: {json}" + ); + } + + #[test] + fn limit_without_n_is_offset_only() { + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Limit { + n: None, + offset: 3, + partition_by: GroupKeys::none(), + child: scan("metrics", value_col()), + })) + .unwrap(); + let dag = export(&root); + let limit = &dag.nodes[dag.root as usize]; + assert_eq!(limit.label, "Limit(offset 3)"); + assert_eq!(limit.detail["n"], serde_json::Value::Null); + assert_eq!(limit.detail["offset"], 3); + } + #[test] fn identical_sub_dags_hash_equal_and_differing_ones_dont() { let left = scan("metrics", value_col()); @@ -1615,17 +1407,18 @@ mod tests { // Two roots that each wrap the *same* Scan shape in a different outer // node — the exported hash should still flag the shared Scan even // though it's embedded at different depths / under different parents. - let shared_shape = || scan("metrics", value_col()); - - let q1 = QueryExpr::Limit { - n: 10, + let q1 = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), offset: 0, - child: Rc::new(shared_shape()), - }; - let q2 = QueryExpr::Dedup { + partition_by: GroupKeys::none(), + child: scan("metrics", value_col()), + })) + .unwrap(); + let q2 = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Dedup { cols: vec![0], - child: Rc::new(shared_shape()), - }; + child: scan("metrics", value_col()), + })) + .unwrap(); let g1 = export(&q1); let g2 = export(&q2); @@ -1647,9 +1440,9 @@ mod tests { fn root_hash_matches_cse_structural_hash_for_the_same_node() { // Not just "hashes equal for equal inputs" (any two consistent hash // functions would do that) — the exported root's `hash` must be the - // literal `u64` `crate::pre_asap::cse::structural_hash` produces for - // this exact node, because it's the same function call, not a - // parallel reimplementation that happens to agree. + // literal `u64` `crate::ir::cse::structural_hash` produces for this + // exact node, because it's the same function call, not a parallel + // reimplementation that happens to agree. let leaf = scan("metrics", value_col()); let dag = export(&leaf); assert_eq!( @@ -1661,24 +1454,15 @@ mod tests { #[test] fn every_node_hash_matches_cse_structural_hash_on_its_own_sub_dag() { - // A multi-level DAG: check the parity holds at every depth, not - // just the root — each `DAGNode::hash` must equal - // `structural_hash` applied to the actual `QueryExpr` sub-DAG that - // node represents. - let agg = QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::none()), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan("metrics", value_col())), - }; - let root = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(agg.clone()), - }; + // A multi-level tree: check the parity holds at every depth, not + // just the root — each `DAGNode::hash` must equal `structural_hash` + // applied to the actual node it represents. + let agg = count_agg(scan("metrics", value_col())); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: true_pred(), + child: Rc::clone(&agg), + })) + .unwrap(); let dag = export(&root); assert_eq!( @@ -1697,39 +1481,23 @@ mod tests { ); } - /// Issue #172: a readout's guarantee is exported structurally — metric, + /// Issue #172: a evaluation's guarantee is exported structurally — metric, /// symbolic bound, failure probability, provenance (allocation - /// included) — and a rejection carries its typed reason. + /// included) — and a rejection carries its typed reason. A relational + /// node below a summary is its own node, in the same dag. #[test] fn export_carries_guarantee_allocation_and_rejection_reason() { - use crate::post_asap::{ - BoundExpr, CompositionOperator, ErrorMetric, FieldDataType, GroupingStrategy, - GuaranteeSource, ProbabilityExpr, Schema, SketchAlgorithm, SketchKind, SketchParams, - SketchStatistic, - }; - let leaf = Rc::new(scan("t", vec![Field::plain("v", DataType::Float64, false)])); - let kept = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::clone(&leaf)), - schema: Schema::lifted(vec![], None), - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), - }); - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: kept, - family: FieldDataType::Sketch( - SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 40 }), - GroupingStrategy::default(), - ), - input: crate::post_asap::SummaryUpdate::column( - crate::pre_asap::expr_ir::ColumnRef::Named("v".into()), - ), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![], None), - guarantee: None, - }); + let leaf = Rc::new( + OperatorNode::new(Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Schema::lifted(vec![Field::plain("v", DataType::Float64, false)], None), + })) + .unwrap() + .with_guarantee(Some(ResultGuarantee::exact("Scan"))), + ); let guarantee = ResultGuarantee { metric: ErrorMetric::Rank, bound: BoundExpr::Sum { @@ -1755,15 +1523,15 @@ mod tests { }, ], }; - let root = SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: agg, - query: SketchStatistic::Quantile { q: 0.99 }, - }, - schema: Schema::lifted(vec![], None), - guarantee: Some(guarantee), - }; + let (_, root) = quantile_evaluation(Rc::clone(&leaf), Some(guarantee)); let dag = export_summary(&root); + assert_eq!( + dag.nodes.iter().map(|n| n.kind).collect::>(), + ["scan", "summary_agg", "summary_estimate"] + ); + assert_eq!(dag.nodes[1].label, "SummaryAgg(Sketch(Kll))"); + assert!(dag.nodes[2].label.starts_with("SummaryEstimate(Quantile")); + assert!(dag.nodes.iter().all(|n| n.schema.is_some())); let json = serde_json::to_value(&dag).unwrap(); let root_json = &json["nodes"][dag.root as usize]; assert_eq!(root_json["guarantee"]["metric"], "rank"); @@ -1779,10 +1547,18 @@ mod tests { assert!(provenance.iter().any(|s| s["kind"] == "composition_step")); // Raw sketch state carries none; the exact leaf carries zero error. let state = &json["nodes"][1]; - assert_eq!(state["kind"], "SummaryAgg"); assert!(state.get("guarantee").is_none()); assert_eq!(json["nodes"][0]["guarantee"]["bound"]["op"], "zero"); + // The `DAGNode` shape carries the same guarantee inside `detail`. + let dag = export(&root); + assert_eq!( + dag.nodes.iter().map(|n| n.kind).collect::>(), + ["Scan", "SummaryAgg", "SummaryEstimate"] + ); + assert_eq!(dag.nodes[2].detail["guarantee"]["metric"], "rank"); + assert!(dag.nodes[1].detail.get("guarantee").is_none()); + let named = NamedDAG { name: "q".into(), source: None, @@ -1792,7 +1568,7 @@ mod tests { workload_cost: None, rejections: vec![TargetRejection { target_pre_id: 0, - strategy: "SketchAlgorithmStrategy".into(), + strategy: "ASAPStrategies".into(), description: "quantile over quantile".into(), error: AccuracyError::UnsupportedComposition { operator: CompositionOperator::ApproximateAggregate, @@ -1819,38 +1595,166 @@ mod tests { .is_none()); } - fn viewer_kind_categories() -> std::collections::BTreeMap { - const START: &str = "const KIND_CATEGORY_JSON = `"; - let source = include_str!(concat!( - env!("CARGO_MANIFEST_DIR"), - "/../../tools/dag-viewer/node-style.js" - )); - let json = source - .split_once(START) - .expect("node-style.js must declare KIND_CATEGORY_JSON") - .1 - .split_once("`;") - .expect("KIND_CATEGORY_JSON must be a template literal") - .0; - serde_json::from_str(json).expect("KIND_CATEGORY_JSON must be valid JSON") + /// `export_post_asap` splices a winning summary in place of its target, + /// tags every node the splice introduced with the decision, and leaves + /// the rest of the query — including an input the summary reuses that + /// was already exported — untagged and shared. + #[test] + fn export_post_asap_splices_a_summary_substitution_in_place() { + let leaf = scan("t", vec![Field::plain("v", DataType::Float64, false)]); + let target = count_agg(Rc::clone(&leaf)); + // `leaf` is exported through the Join's left side before the target + // (its right side) is reached and substituted. + let root = join(Rc::clone(&leaf), Rc::clone(&target)); + let (_, evaluation) = quantile_evaluation(Rc::clone(&leaf), None); + let decision = DAGDecision { + id: 7, + strategy: "Sketch".into(), + rationale: "quantile via KLL".into(), + rank: 0, + cost: 1.0, + role: "", + baseline_cost: None, + selected_cost: None, + benefit: None, + }; + let mut calls = Vec::new(); + let dag = export_post_asap(&root, &mut |node| { + calls.push(node.operator.kind_name()); + Rc::ptr_eq(node, &target).then(|| PostAsapSubstitution::Summary { + replacement: Rc::clone(&evaluation), + decision: decision.clone(), + }) + }); + + let kinds: Vec<_> = dag.nodes.iter().map(|n| n.kind).collect(); + assert_eq!(kinds, ["Scan", "SummaryAgg", "SummaryEstimate", "Join"]); + assert!(!kinds.contains(&"Aggregate"), "the target itself is gone"); + let join_node = &dag.nodes[dag.root as usize]; + assert_eq!(join_node.children, vec![0, 2]); + assert!(join_node.decision.is_none()); + let estimate = &dag.nodes[2]; + assert_eq!(estimate.kind, "SummaryEstimate"); + assert_eq!( + estimate.decision.as_ref().map(|d| (d.id, d.role)), + Some((7, "replacement_root")) + ); + let agg = &dag.nodes[estimate.children[0] as usize]; + assert_eq!( + agg.decision.as_ref().map(|d| (d.id, d.role)), + Some((7, "replacement_region")) + ); + let scan_node = &dag.nodes[agg.children[0] as usize]; + assert_eq!( + scan_node.id, 0, + "the summary reuses the already-exported input" + ); + assert!( + scan_node.decision.is_none(), + "a node exported before the splice is not tagged by it" + ); + assert_eq!( + calls, + ["Join", "Scan", "Aggregate", "SummaryAgg"], + "the substitution's own top level (SummaryEstimate) is never re-queried; its \ + descendants are, except the input already exported" + ); } + /// A `SharedSubDAGStrategy`-shaped substitution returns the target + /// itself as its replacement; the walk must still terminate and render + /// the target once. #[test] - fn viewer_categorizes_exactly_the_exported_node_kinds() { - let expected: std::collections::BTreeSet<_> = QUERY_KIND_TAGS - .iter() - .chain(SUMMARY_KIND_TAGS) - .copied() - .chain(std::iter::once("KeepPreAsap")) - .collect(); + fn export_post_asap_terminates_when_the_replacement_is_the_target() { + let target = count_agg(scan("t", value_col())); + let decision = DAGDecision { + id: 1, + strategy: "SharedSubDAG".into(), + rationale: "share".into(), + rank: 0, + cost: f64::NAN, + role: "", + baseline_cost: None, + selected_cost: None, + benefit: None, + }; + let dag = export_post_asap(&target, &mut |node| { + Rc::ptr_eq(node, &target).then(|| PostAsapSubstitution::Rewrite { + replacement: Rc::clone(&target), + decision: decision.clone(), + }) + }); + assert_eq!(dag.nodes.len(), 2); + assert_eq!(dag.nodes[dag.root as usize].kind, "Aggregate"); assert_eq!( - expected.len(), - QUERY_KIND_TAGS.len() + SUMMARY_KIND_TAGS.len() + 1, - "exported kind tags must be unique" + dag.nodes[dag.root as usize] + .decision + .as_ref() + .map(|d| d.role), + Some("replacement_root") ); - let categories = viewer_kind_categories(); - let actual: std::collections::BTreeSet<_> = categories.keys().map(String::as_str).collect(); + } - assert_eq!(actual, expected); + /// The snake_case `kind` table is exactly `kind_name` re-cased, for + /// every variant: a `SummaryDAGNode` and a `DAGNode` for the same node + /// never disagree on what it is. + #[test] + fn summary_kind_is_the_operator_kind_name_in_snake_case() { + fn to_snake(name: &str) -> String { + let mut out = String::new(); + let chars: Vec = name.chars().collect(); + for (i, &c) in chars.iter().enumerate() { + if c.is_ascii_uppercase() { + let prev_lower = i > 0 && !chars[i - 1].is_ascii_uppercase(); + let next_lower = chars.get(i + 1).is_some_and(|n| n.is_ascii_lowercase()); + if i > 0 && (prev_lower || next_lower) { + out.push('_'); + } + out.push(c.to_ascii_lowercase()); + } else { + out.push(c); + } + } + out + } + let leaf = scan("t", vec![Field::plain("v", DataType::Float64, false)]); + let (_, evaluation) = quantile_evaluation(Rc::clone(&leaf), None); + let finalize = std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: evaluation }), + Schema::lifted(vec![], None), + ) + .with_guarantee(None), + ); + let root = + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::SQLWindowFunc { + func: crate::ir::operator_properties::WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::none(), + order_by: vec![], + frame: None, + output_name: "rn".into(), + child: finalize, + })) + .unwrap(); + let dag = export(&root); + let summary = export_summary(&root); + assert_eq!(dag.nodes.len(), summary.nodes.len()); + for (a, b) in dag.nodes.iter().zip(&summary.nodes) { + assert_eq!(a.id, b.id); + assert_eq!(b.kind, to_snake(a.kind), "{}", a.kind); + assert_eq!(a.children, b.children); + assert_eq!(a.label, b.label); + } + assert_eq!( + summary.nodes.iter().map(|n| n.kind).collect::>(), + [ + "scan", + "summary_agg", + "summary_estimate", + "finalize_exact_accumulator", + "sql_window_func", + ] + ); } } diff --git a/crates/types/src/parsed_workload.rs b/crates/types/src/parsed_workload.rs index ff955e6a7..9dcf21b83 100644 --- a/crates/types/src/parsed_workload.rs +++ b/crates/types/src/parsed_workload.rs @@ -1,5 +1,5 @@ //! [`ParsedWorkload`] — a [`PlanningWorkload`] whose queries have been lowered -//! to pre-ASAP IR. +//! to the operator IR. //! //! This is the boundary between the frontend stage and the optimization stage //! (issues #429, #430). Everything downstream of lowering consumes this type @@ -10,7 +10,7 @@ use std::rc::Rc; -use crate::pre_asap::query_expr::QueryExpr; +use crate::ir::{OperatorNode, QueryRoot, ScalarExpr}; use crate::workload::{ DataWorkload, PlanningWorkload, QueryWorkload, QueryWorkloadEntry, WorkloadError, }; @@ -33,7 +33,9 @@ pub enum ParsedWorkloadError { #[derive(Debug, Clone)] pub struct ParsedWorkload { workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, + operator_indices: Vec, + scalars: Vec<(usize, ScalarExpr)>, } impl ParsedWorkload { @@ -41,16 +43,43 @@ impl ParsedWorkload { /// `i`-th entry. pub fn new( workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, + ) -> Result { + Self::from_roots( + workload, + exprs.into_iter().map(QueryRoot::Operator).collect(), + ) + } + + pub fn from_roots( + workload: PlanningWorkload, + roots: Vec, ) -> Result { let entries = workload.query_workload.entries().count(); - if entries != exprs.len() { + if entries != roots.len() { return Err(ParsedWorkloadError::LengthMismatch { entries, - lowered: exprs.len(), + lowered: roots.len(), }); } - Ok(Self { workload, exprs }) + let mut exprs = Vec::new(); + let mut operator_indices = Vec::new(); + let mut scalars = Vec::new(); + for (index, root) in roots.into_iter().enumerate() { + match root { + QueryRoot::Operator(node) => { + operator_indices.push(index); + exprs.push(node); + } + QueryRoot::Scalar(expr) => scalars.push((index, expr)), + } + } + Ok(Self { + workload, + exprs, + operator_indices, + scalars, + }) } pub fn planning_workload(&self) -> &PlanningWorkload { @@ -65,26 +94,37 @@ impl ParsedWorkload { self.workload.data_workload.as_ref() } - pub fn exprs(&self) -> &[Rc] { + pub fn exprs(&self) -> &[Rc] { &self.exprs } pub fn len(&self) -> usize { - self.exprs.len() + self.exprs.len() + self.scalars.len() } pub fn is_empty(&self) -> bool { - self.exprs.is_empty() + self.len() == 0 } /// Normalized entries paired with their lowered expression. - pub fn entries(&self) -> impl Iterator)> + '_ { + pub fn entries(&self) -> impl Iterator)> + '_ { self.workload .query_workload .entries() + .enumerate() + .filter(|(index, _)| self.operator_indices.binary_search(index).is_ok()) + .map(|(_, entry)| entry) .zip(self.exprs.iter()) } + pub fn operator_indices(&self) -> &[usize] { + &self.operator_indices + } + + pub fn scalar_roots(&self) -> &[(usize, ScalarExpr)] { + &self.scalars + } + /// The retained workload's own validation — entry legality and data-workload /// consistency. The PromQL-specific checks it also runs were already a /// precondition of the lowering that produced `self`. diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 74f1d47d2..38219fa2c 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -816,6 +816,8 @@ pub enum ExactOperationSchemaError { NonPlainInput, #[error("schema derivation failed: {0}")] Schema(#[from] QueryExprError), + #[error("schema derivation failed: {0}")] + Derivation(#[from] crate::ir::SchemaDerivationError), } #[cfg(test)] diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs index bd45748aa..732a09db5 100644 --- a/crates/types/src/post_asap/expr.rs +++ b/crates/types/src/post_asap/expr.rs @@ -33,7 +33,7 @@ pub enum ValueOperation { /// Maintain the full declared population, including membership changes, /// so removing a TopK member can promote another. MaintainPopulation { - population: super::maintained_population::MaintainedPopulation, + population: super::maintained_population::MaintainedPopulation, }, /// Read an aggregate or TopK prefix from the maintained population. ReadPopulation { diff --git a/crates/types/src/post_asap/guarantee.rs b/crates/types/src/post_asap/guarantee.rs index cfc6f3f01..68cd7cea9 100644 --- a/crates/types/src/post_asap/guarantee.rs +++ b/crates/types/src/post_asap/guarantee.rs @@ -16,9 +16,9 @@ //! ## What a guarantee says //! //! [`ResultGuarantee`] is attached to a finalized, caller-visible value — -//! [`super::SummaryNode::guarantee`] on a `SummaryEstimate` readout, an +//! [`crate::ir::OperatorNode::guarantee`] on a `SummaryEstimate` evaluation, an //! exact accumulator, or a kept pre-ASAP sub-DAG — never to raw summary -//! state (a `SummaryAgg` sketch node carries `None`; its readout carries the +//! state (a `SummaryAgg` sketch node carries `None`; its evaluation carries the //! guarantee). Its statement is: //! //! ```text @@ -325,13 +325,13 @@ pub enum GuaranteeSource { /// Deterministic exact computation — zero error by construction. Exact { /// What made it exact (e.g. `"ExactAggregate(Sum)"`, - /// `"KeepPreAsap"`). + /// `"RetainedExact"`). reason: String, }, - /// The target this readout's sketch was sized against. + /// The target this evaluation's sketch was sized against. AccuracyTarget { target: AccuracyTarget }, - /// The concrete sketch a readout's local guarantee was derived from. - SketchReadout { + /// The concrete sketch a evaluation's local guarantee was derived from. + SketchEvaluation { algorithm: String, /// Stable estimator/analysis contract used to derive this guarantee. #[serde(default)] @@ -421,8 +421,8 @@ impl ResultGuarantee { self.bound.is_zero() && self.failure_probability.is_zero() } - /// How many approximate sketch readouts contributed to this value — - /// `1` for a plain readout, `0` for an exact value, and the transitive + /// How many approximate sketch evaluations contributed to this value — + /// `1` for a plain evaluation, `0` for an exact value, and the transitive /// count through every [`GuaranteeSource::ChildGuarantee`] for a /// composition. An `AccuracyBudgetAllocator` uses this as the number /// of layers a budget must be split across. @@ -430,7 +430,7 @@ impl ResultGuarantee { self.provenance .iter() .map(|source| match source { - GuaranteeSource::SketchReadout { .. } => 1, + GuaranteeSource::SketchEvaluation { .. } => 1, GuaranteeSource::ChildGuarantee { guarantee, .. } => { guarantee.approximate_layer_count() } diff --git a/crates/types/src/post_asap/maintained_population.rs b/crates/types/src/post_asap/maintained_population.rs index 73ddd38ba..577544bd7 100644 --- a/crates/types/src/post_asap/maintained_population.rs +++ b/crates/types/src/post_asap/maintained_population.rs @@ -114,7 +114,7 @@ impl CurrentSeriesInput { /// Membership is part of state identity. Table rows must never acquire implicit /// latest-per-series selection, stale markers, or a PromQL lookback. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum PopulationInput { +pub enum PopulationInput { CurrentSeries(CurrentSeriesInput), Rows { input: std::rc::Rc, @@ -124,13 +124,13 @@ pub enum PopulationInput { } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct MaintainedPopulation { +pub struct MaintainedPopulation { pub input: PopulationInput, pub max_k: usize, pub quantiles: bool, } -impl MaintainedPopulation { +impl MaintainedPopulation { pub fn matches_input(&self, input: &crate::pre_asap::QueryExpr) -> bool { match &self.input { PopulationInput::CurrentSeries(spec) => spec.matches_input(input), diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index f71aaca0f..b1a7da67a 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -27,6 +27,7 @@ //! alongside `reduction` and on sketch-valued edge types //! — see `asap_aware_mapping::grouping`'s module docs for why. +// Legacy summary IR: no longer re-exported; removed by the cleanup PR. pub mod cse; pub mod execution_data_state; pub mod expr; @@ -40,21 +41,29 @@ pub mod summary_maintenance_lifecycle; pub mod summary_window; pub use crate::pre_asap::schema::{Field, FieldDataType, Schema}; -pub use cse::share_common_summary_sub_dags; pub use execution_data_state::{ - assigned_child_data_state, exact_operation_output_schema, produced_data_state, - validate_execution_data_states, validate_execution_data_states_at, DataPrimitive, - ExactOperationSchemaError, ExecutionDataState, ExecutionDataStateAssignment, + lift_plain, DataPrimitive, ExactOperationSchemaError, ExecutionDataState, ExecutionDataStateError, ExecutionTiming, }; -pub use expr::{ +// Legacy summary IR names, kept for the legacy modules above only. +#[allow(unused_imports)] +pub(crate) use cse::share_common_summary_sub_dags; +#[allow(unused_imports)] +pub(crate) use execution_data_state::{ + assigned_child_data_state, exact_operation_output_schema, produced_data_state, + validate_execution_data_states, validate_execution_data_states_at, + ExecutionDataStateAssignment, +}; +#[allow(unused_imports)] +pub(crate) use expr::{ BinaryOperator, CandidateCompleteness, ExactOperation, SummaryExpr, SummaryNode, ValueOperation, }; pub use guarantee::{ AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, GuaranteeSource, ProbabilityExpr, ResultGuarantee, }; -pub use post_asap_dag::{ +#[allow(unused_imports)] +pub(crate) use post_asap_dag::{ compile_post_asap_dag, compile_post_asap_dag_with_node_ids, EdgeRole, GroupingEdgeCompatibility, PostAsapDAG, PostAsapDAGCompilation, PostAsapDAGDocument, PostAsapDAGEdge, PostAsapDAGNode, PostAsapDAGValidationError, PostAsapNodeId, diff --git a/crates/types/src/post_asap/query_time/error_estimation.rs b/crates/types/src/post_asap/query_time/error_estimation.rs index 663d2da12..f2bf229da 100644 --- a/crates/types/src/post_asap/query_time/error_estimation.rs +++ b/crates/types/src/post_asap/query_time/error_estimation.rs @@ -41,7 +41,7 @@ //! //! ## What this is *not* — no runtime sketch exists yet to wire this into //! -//! This issue names two possible integration points: (1) runtime/readout-time +//! This issue names two possible integration points: (1) runtime/evaluation-time //! accuracy reporting from a sketch's *actual* counters, and (2) tighter //! plan-time sizing. As of this module landing, **this repository has no //! vendored CMS/CountSketch/CU-Sketch runtime and no counter-array data @@ -52,12 +52,12 @@ //! planning-time sizing metadata. There is no `A[row][col]` counter matrix //! anywhere in the workspace for these functions to be handed at query //! time. So integration point (1) — reporting an *actual* query's posterior -//! error from real counters at readout — has nothing to wire into today. +//! error from real counters at evaluation — has nothing to wire into today. //! //! The functions here are deliberately **sketch-object-agnostic**: they take //! plain counter slices (`&[u64]` / `&[i64]`) and numeric parameters, not a //! concrete sketch type, specifically so that the moment a real CMS/ -//! Count-Sketch/CU-Sketch runtime lands in this workspace, its readout path +//! Count-Sketch/CU-Sketch runtime lands in this workspace, its evaluation path //! can call these functions directly on its real counter arrays with zero //! changes needed here. That wiring is out of scope for this module — see //! issue #239. @@ -255,8 +255,8 @@ fn posterior_rank(w: usize, rows: u32, delta: f64) -> Option { /// The `k`-th largest value in `values` (1-indexed: `k=1` is the max). /// `select_nth_unstable_by` partitions in O(w) average instead of fully /// sorting in O(w log w) — this only ever needs one rank, not a total -/// order, and both call sites (this module's per-query readout math) are -/// documented as meant to run on a future runtime's hot readout path. +/// order, and both call sites (this module's per-query evaluation math) are +/// documented as meant to run on a future runtime's hot evaluation path. fn kth_largest(values: &[u64], k: usize) -> u64 { let mut buf: Vec = values.to_vec(); let idx = k - 1; diff --git a/crates/types/src/post_asap/sketch.rs b/crates/types/src/post_asap/sketch.rs index a5e1edcc1..65e78d24f 100644 --- a/crates/types/src/post_asap/sketch.rs +++ b/crates/types/src/post_asap/sketch.rs @@ -6,7 +6,7 @@ use crate::pre_asap::ColumnRef; /// An exact, mergeable accumulator family — zero approximation error. The /// partial state built for one of these *is* the answer; no -/// `SummaryEstimate` readout is needed to get a value out of it. +/// `SummaryEstimate` evaluation is needed to get a value out of it. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] pub enum ExactKind { /// Exact sum accumulator (mergeable by addition). @@ -48,7 +48,7 @@ pub enum ExactParams { /// [`SketchKind::new`] is where that classification is made. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] pub enum SketchAlgorithm { - /// Universal frequency-vector summary with shared statistic readouts. + /// Universal frequency-vector summary with shared statistic evaluations. UnivMon, /// KLL quantile sketch (mergeable, ε-accurate rank queries). Kll, @@ -345,7 +345,7 @@ pub enum StatModelParams { /// really the universal-sketch composition (L layers of Count-Sketch plus a /// heavy-hitter heap, Theorems 1+2 combined) estimating entropy/L1-norm/ /// L2-norm/cardinality/frequency-moments as one instance. Standalone UnivMon -/// and its frequency readouts are represented here, but sharing a Hydra +/// and its frequency evaluations are represented here, but sharing a Hydra /// grid across populations still needs its own collision/error contract; /// standalone support does not establish that contract. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] @@ -500,7 +500,7 @@ pub fn default_hydra_params( /// enums themselves, for exactly the reason explained in this section's /// module docs above. /// -/// Carried both on `SummaryExpr::SummaryAgg` (where planning consults it) +/// Carried both on `ASAPOp::SummaryAgg` (where planning consults it) /// and on sketch-valued `FieldDataType` edges (where it prevents /// incompatible shared and independent physical states from type-checking /// as merge-compatible). @@ -590,7 +590,8 @@ pub enum SummaryInputExpr { EntityIdentity(EntityIdentity), } -/// What to extract from a built summary. Carried by `SummaryEstimate`. +/// The statistic computed from summary state by `SummaryEstimate`, for example +/// `Quantile { q: 0.99 }`. This is a result operation, not a workload query. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub enum SketchStatistic { /// sqrt(sum_v frequency(v)^2), not the norm of numeric input values. @@ -605,8 +606,8 @@ pub enum SketchStatistic { /// `value: Some(v)` is a per-item point lookup (e.g. /// `count(cms_metric{item="checkout"})` — `key` is `item`, `value` is /// `"checkout"`). `value` is carried here rather than resolved by the - /// `SummaryExecutor` from a `Filter` predicate because `readout`'s - /// trait signature has no DAG access — see `CostModel::readout_extension`. + /// `SummaryExecutor` from a `Filter` predicate because `evaluation`'s + /// trait signature has no dag access — see `CostModel::evaluation_extension`. PointCount { key: ColumnRef, value: Option, @@ -622,7 +623,7 @@ mod tests { use super::*; #[test] - fn keyed_summary_input_and_topk_readout_round_trip() { + fn keyed_summary_input_and_topk_evaluation_round_trip() { for input in [ SummaryUpdate { item: Some(SummaryInputExpr::EntityIdentity( diff --git a/crates/types/src/post_asap/summary_maintenance.rs b/crates/types/src/post_asap/summary_maintenance.rs index d1e50d7e5..4e7eaf205 100644 --- a/crates/types/src/post_asap/summary_maintenance.rs +++ b/crates/types/src/post_asap/summary_maintenance.rs @@ -1,6 +1,6 @@ //! Planner-level construction mode for a materialized summary. //! -//! A [`super::SummaryNode`] is a logical summary expression and deliberately +//! An ASAP [`crate::ir::OperatorNode`] is a logical summary expression and deliberately //! does not carry this choice: the same candidate may be built directly for //! one workload or maintained incrementally for another. Planner search //! attaches the selected mode to its lifecycle guarantee; downstream physical diff --git a/crates/types/src/post_asap/summary_window.rs b/crates/types/src/post_asap/summary_window.rs index 0e344c1ed..6d45649b4 100644 --- a/crates/types/src/post_asap/summary_window.rs +++ b/crates/types/src/post_asap/summary_window.rs @@ -57,7 +57,7 @@ pub enum PaneCoverageError { }, } -/// Validate that a pane-only readout covers a query exactly. A mismatched +/// Validate that a pane-only evaluation covers a query exactly. A mismatched /// phase is sound only when the physical plan explicitly supplies an exact /// residual for the partial edge panes. pub fn validate_pane_coverage( @@ -147,7 +147,7 @@ mod tests { } #[test] - fn pane_only_readout_rejects_source_and_query_phase_mismatch() { + fn pane_only_evaluation_rejects_source_and_query_phase_mismatch() { let layout = PaneLayout { pane_width_ms: 60_000, pane_origin_ms: Some(26_000), diff --git a/crates/types/tests/planner_vocabulary.rs b/crates/types/tests/planner_vocabulary.rs index f14a56ee8..f3d53cf59 100644 --- a/crates/types/tests/planner_vocabulary.rs +++ b/crates/types/tests/planner_vocabulary.rs @@ -1,6 +1,5 @@ -use asap_types::post_asap::{ - validate_pane_coverage, PaneLayout, WindowEdgeCompatibility, WindowEdgeCoverage, -}; +use asap_types::ir::physical_export::WindowEdgeCompatibility; +use asap_types::post_asap::{validate_pane_coverage, PaneLayout, WindowEdgeCoverage}; use asap_types::pre_asap::{SchemaResolver, Source, UnresolvedQueryExpr}; use asap_types::resources::{PhysicalHandoffBytes, PhysicalHandoffKind};