diff --git a/Cargo.lock b/Cargo.lock index b0dd1db9a..d698d1714 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -304,35 +304,53 @@ dependencies = [ ] [[package]] -name = "asap-aware-mapping" +name = "asap-devtools" version = "0.1.0" dependencies = [ "asap-frontend-promql", + "asap-frontend-sql", + "asap-logical-optimizer", + "asap-plan-selection", "asap-types", - "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "serde", "serde_json", - "thiserror 2.0.18", + "serde_yaml", + "tokio", ] [[package]] -name = "asap-devtools" +name = "asap-executor" version = "0.1.0" dependencies = [ - "asap-aware-mapping", "asap-frontend-promql", - "asap-frontend-sql", + "asap-logical-optimizer", + "asap-plan-selection", "asap-types", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", + "chrono", + "futures", + "regex", + "rmp-serde", "serde", "serde_json", - "serde_yaml", - "tokio", + "thiserror 2.0.18", + "tracing", +] + +[[package]] +name = "asap-frontend-common" +version = "0.1.0" +dependencies = [ + "asap-types", + "serde", + "thiserror 2.0.18", ] [[package]] name = "asap-frontend-metricsql" version = "0.1.0" dependencies = [ + "asap-frontend-common", "asap-types", "metricsql_parser", "thiserror 2.0.18", @@ -342,7 +360,9 @@ dependencies = [ name = "asap-frontend-promql" version = "0.1.0" dependencies = [ - "asap-aware-mapping", + "asap-frontend-common", + "asap-logical-optimizer", + "asap-plan-selection", "asap-types", "promql-parser", ] @@ -351,7 +371,8 @@ dependencies = [ name = "asap-frontend-sql" version = "0.1.0" dependencies = [ - "asap-aware-mapping", + "asap-frontend-common", + "asap-logical-optimizer", "asap-sql-function-catalog", "asap-types", "datafusion", @@ -365,10 +386,13 @@ dependencies = [ name = "asap-integration-tests" version = "0.1.0" dependencies = [ - "asap-aware-mapping", + "asap-executor", "asap-frontend-promql", "asap-frontend-sql", - "asap-physical-operators", + "asap-logical-optimizer", + "asap-physical-optimizer", + "asap-plan-selection", + "asap-planner", "asap-types", "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "futures", @@ -377,30 +401,48 @@ dependencies = [ ] [[package]] -name = "asap-physical-operators" +name = "asap-logical-optimizer" version = "0.1.0" dependencies = [ - "asap-aware-mapping", "asap-frontend-promql", "asap-types", - "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", - "futures", - "regex", - "rmp-serde", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", + "serde_json", + "thiserror 2.0.18", +] + +[[package]] +name = "asap-physical-optimizer" +version = "0.1.0" +dependencies = [ + "asap-frontend-promql", + "asap-logical-optimizer", + "asap-types", + "thiserror 2.0.18", +] + +[[package]] +name = "asap-plan-selection" +version = "0.1.0" +dependencies = [ + "asap-frontend-promql", + "asap-logical-optimizer", + "asap-physical-optimizer", + "asap-types", "serde", "serde_json", "thiserror 2.0.18", - "tracing", ] [[package]] name = "asap-planner" version = "0.1.0" dependencies = [ - "asap-aware-mapping", "asap-frontend-metricsql", "asap-frontend-promql", "asap-frontend-sql", + "asap-logical-optimizer", + "asap-plan-selection", "asap-types", "thiserror 2.0.18", "tokio", diff --git a/Cargo.toml b/Cargo.toml index a2b019af8..ad0ed9ce6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,9 +1,12 @@ [workspace] members = [ - "crates/asap-physical-operators", + "crates/executor", "crates/types", + "crates/frontend-common", "crates/sql-function-catalog", - "crates/asap-aware-mapping", + "crates/logical-optimizer", + "crates/physical-optimizer", + "crates/plan-selection", "crates/frontend-promql", "crates/frontend-metricsql", "crates/metricsql-common-parser-support", diff --git a/crates/asap-aware-mapping/Cargo.toml b/crates/asap-aware-mapping/Cargo.toml deleted file mode 100644 index b2c03ed12..000000000 --- a/crates/asap-aware-mapping/Cargo.toml +++ /dev/null @@ -1,18 +0,0 @@ -[package] -name = "asap-aware-mapping" -version = "0.1.0" -edition = "2021" - -# The cost-aware optimizer layer (L4 decisions) over the L3 intent algebra: -# selects and sizes the summary family for each intent. Consumes and produces -# asap-types (L3 intent algebra + L4 sketch-bound IR) — never depends on a -# front end. -[dependencies] -asap_sketchlib = { workspace = true } -asap-types = { path = "../types" } -thiserror = "2" -serde = { version = "1", features = ["derive"] } -serde_json = "1" - -[dev-dependencies] -asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-aware-mapping/src/empirical_comparison.rs b/crates/asap-aware-mapping/src/empirical_comparison.rs deleted file mode 100644 index ecd2a6cd4..000000000 --- a/crates/asap-aware-mapping/src/empirical_comparison.rs +++ /dev/null @@ -1,913 +0,0 @@ -//! Query-matched, fixed-snapshot offline recommendations. Observed error is an -//! explicit acceptance criterion, never a replacement for formal guarantees. - -use asap_types::post_asap::{SketchAlgorithm, SketchParams}; -use serde::{Deserialize, Serialize}; - -use crate::empirical_cost::{ - DistributionDescriptor, EmpiricalEvidenceProvider, EnvironmentDescriptor, EvidenceArtifact, - EvidenceContext, Measurement, MeasurementProvenance, OfflineMeasurement, -}; - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OfflineQueryDescriptor { - pub kind: String, - pub value_type: String, - /// Identifies the exact probe population used for timing and observed error. - pub probe_set: String, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct MeasurementQueryBinding { - pub record_id: String, - pub query: OfflineQueryDescriptor, -} - -pub use crate::empirical_resources::ExactResourceMeasurements; - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OfflineExactMeasurement { - pub id: String, - pub distribution: DistributionDescriptor, - pub environment: EnvironmentDescriptor, - pub query: OfflineQueryDescriptor, - pub measured_at_unix_seconds: u64, - pub valid_until_unix_seconds: u64, - pub provenance: MeasurementProvenance, - pub metrics: ExactResourceMeasurements, -} - -/// The companion format binds otherwise query-agnostic sketch primitives to -/// their measured readout and exact reference. Bindings describe state after -/// ingestion, without merges or intervening updates during the read sequence. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OfflineComparisonEvidence { - pub schema_version: u32, - /// Every CPU phase excludes other phases and destruction of retained state. - pub timing_contract: String, - pub sketch_evidence: EvidenceArtifact, - pub query_bindings: Vec, - pub exact_records: Vec, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct EmpiricalAccuracyRequirement { - pub metric: String, - pub max_observed_mean: f64, - pub minimum_trials: u32, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OfflineWorkload { - pub input_items_per_state: u64, - /// Reads of the fixed, fully ingested snapshot; no updates between reads. - pub reads_per_state: u64, - pub merges_per_state: u64, - pub state_instances: u64, - pub horizon_seconds: f64, -} - -/// Explicit scalarization: CPU ns × cpu_ns_weight + byte-seconds × -/// retained_byte_seconds_weight. Weights must be nonnegative and not both zero. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ResourceWeights { - pub cpu_ns_weight: f64, - pub retained_byte_seconds_weight: f64, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SketchConfiguration { - pub algorithm: SketchAlgorithm, - pub params: SketchParams, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OfflineComparisonRequest { - pub context: EvidenceContext, - pub exact_environment: EnvironmentDescriptor, - pub query: OfflineQueryDescriptor, - pub accuracy: EmpiricalAccuracyRequirement, - pub workload: OfflineWorkload, - pub weights: ResourceWeights, - /// `Some` restricts selection to these algorithms and configurations at - /// least as large as their formally legal deployment parameters. `None` - /// requests a purely offline recommendation, unsuitable for formal binding. - pub formal_minimums: Option>, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct OfflineResourceEstimate { - pub cpu_ns: f64, - pub retained_bytes: Option, - pub retained_byte_seconds: Option, - /// Conservative sum of per-state construction/ingestion peaks. - pub peak_bytes_upper_bound: Option, - /// Per-state snapshot sizes, not charged as writes in this in-memory model. - pub serialized_bytes_per_state: Option, - pub disk_bytes_per_state: Option, - pub objective_cost: f64, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct OfflineCandidateEstimate { - pub record_id: String, - pub configuration: Option, - pub observed_error_mean: Option, - pub resources: Option, - pub rejection: Option, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct OfflineRecommendation { - pub query: OfflineQueryDescriptor, - pub selected: OfflineCandidateEstimate, - pub exact_baseline: OfflineCandidateEstimate, - pub candidates: Vec, - pub estimated_cpu_savings_ns: f64, - pub estimated_retained_bytes_savings: Option, - /// True only if the request supplied and every selected sketch passed - /// explicit formal minima; observed error alone cannot authorize binding. - pub checked_formal_minimums: bool, - pub assumptions: Vec, -} - -impl OfflineRecommendation { - /// An exact selection deliberately returns no sketch. The caller must - /// preserve its exact path; do not silently substitute a default sketch. - pub fn selected_sketch(&self) -> Option<&SketchConfiguration> { - self.selected.configuration.as_ref() - } -} - -/// Compare complete CPU components for a fixed snapshot and retain every -/// unavailable alternative with its reason. Without an applicable exact -/// baseline no benefit can be established, so this returns an error. -pub fn recommend_offline( - evidence: &OfflineComparisonEvidence, - request: &OfflineComparisonRequest, -) -> Result { - validate_request(request)?; - if evidence.schema_version != 1 { - return Err("unsupported offline comparison schema version".into()); - } - if evidence.timing_contract != "disjoint_live_state_v1" { - return Err( - "comparison requires disjoint CPU phases measured with retained state alive".into(), - ); - } - let provider = - EmpiricalEvidenceProvider::new(evidence.sketch_evidence.clone(), request.context.clone()) - .map_err(|e| e.to_string())?; - let mut bindings = std::collections::HashMap::new(); - for binding in &evidence.query_bindings { - if bindings - .insert(&binding.record_id, &binding.query) - .is_some() - { - return Err("duplicate query binding".into()); - } - if !evidence - .sketch_evidence - .records - .iter() - .any(|r| r.id == binding.record_id) - { - return Err("query binding refers to missing record".into()); - } - } - let exact: Vec<_> = evidence - .exact_records - .iter() - .filter(|r| { - r.distribution == request.context.distribution - && r.environment == request.exact_environment - && r.query == request.query - && r.measured_at_unix_seconds <= request.context.now_unix_seconds - && request.context.now_unix_seconds <= r.valid_until_unix_seconds - }) - .collect(); - if exact.len() != 1 { - return Err("missing, stale, incompatible or ambiguous exact baseline".into()); - } - let exact = exact[0]; - validate_exact(exact)?; - let baseline_resources = estimate_exact(exact, request)?; - let exact_baseline = OfflineCandidateEstimate { - record_id: exact.id.clone(), - configuration: None, - observed_error_mean: Some(0.0), - resources: Some(baseline_resources), - rejection: None, - }; - let mut candidates = Vec::new(); - for row in &evidence.sketch_evidence.records { - let mut candidate = OfflineCandidateEstimate { - record_id: row.id.clone(), - configuration: Some(SketchConfiguration { - algorithm: row.algorithm.clone(), - params: row.params.clone(), - }), - observed_error_mean: row.error.as_ref().and_then(|e| e.mean), - resources: None, - rejection: None, - }; - let result = (|| { - let matched = provider - .lookup(&row.algorithm, &row.params) - .map_err(|e| e.to_string())?; - if matched.id != row.id { - return Err("record belongs to another applicability context".into()); - } - if bindings.get(&row.id).copied() != Some(&request.query) { - return Err("missing or incompatible measured query binding".into()); - } - if let Some(minimums) = &request.formal_minimums { - if !minimums.iter().any(|m| { - m.algorithm == row.algorithm && parameters_at_least(&row.params, &m.params) - }) { - return Err( - "configuration does not meet the deployment's formal minimum".into(), - ); - } - } - let error = row - .error - .as_ref() - .ok_or("missing offline error observation")?; - if error.query.get("kind").and_then(|v| v.as_str()) != Some(request.query.kind.as_str()) - || error.query.get("value_type").and_then(|v| v.as_str()) - != Some(request.query.value_type.as_str()) - { - return Err("offline error observation has incompatible readout semantics".into()); - } - if error.metric != request.accuracy.metric - || error.trials < request.accuracy.minimum_trials - { - return Err("incompatible error metric or insufficient offline trials".into()); - } - let mean = error.mean.ok_or("missing observed mean error")?; - if mean > request.accuracy.max_observed_mean { - return Err("observed error exceeds explicit offline acceptance budget".into()); - } - estimate_sketch(row, request) - })(); - match result { - Ok(resources) => candidate.resources = Some(resources), - Err(reason) => candidate.rejection = Some(reason), - } - candidates.push(candidate); - } - let mut selected = exact_baseline.clone(); - for candidate in &candidates { - if let Some(resources) = &candidate.resources { - if resources.objective_cost < selected.resources.as_ref().unwrap().objective_cost { - selected = candidate.clone(); - } - } - } - let baseline = exact_baseline.resources.as_ref().unwrap(); - let chosen = selected.resources.as_ref().unwrap(); - Ok(OfflineRecommendation { query: request.query.clone(), - estimated_cpu_savings_ns: baseline.cpu_ns - chosen.cpu_ns, - estimated_retained_bytes_savings: baseline.retained_bytes.zip(chosen.retained_bytes).map(|(a,b)| a-b), - selected, exact_baseline, candidates, checked_formal_minimums: request.formal_minimums.is_some(), - assumptions: vec![ - "Fixed snapshot: construct, ingest all measured input, prepare exact index once, then read without updates or merges".into(), - "Read CPU is the measured average over all distinct keys; individual key latency and error can differ".into(), - "CPU sums disjoint measured construction/update/prepare/read phases while state remains alive; the comparison ends with retained state and excludes retirement".into(), - "No serialization or disk-write CPU is modeled; reported disk/serialized bytes describe one optional persisted snapshot only".into(), - "Offline mean-error acceptance is restricted to the measured input and probe population; it is not a formal or runtime error guarantee".into(), - ] }) -} - -fn validate_request(request: &OfflineComparisonRequest) -> Result<(), String> { - let w = &request.workload; - if request.query.kind != "point_frequency" - || request.query.value_type != "i64" - || request.query.probe_set != "all_distinct_keys" - { - return Err("unsupported offline query contract".into()); - } - if request.accuracy.metric.trim().is_empty() - || !nonnegative(request.accuracy.max_observed_mean) - || request.accuracy.minimum_trials == 0 - { - return Err("invalid empirical accuracy requirement".into()); - } - if w.input_items_per_state != request.context.distribution.sample_count - || w.state_instances == 0 - || !w.horizon_seconds.is_finite() - || w.horizon_seconds <= 0.0 - { - return Err( - "workload must match the measured snapshot and have positive states/horizon".into(), - ); - } - if w.merges_per_state != 0 { - return Err("no post-merge error or exact merge baseline was measured".into()); - } - if !nonnegative(request.weights.cpu_ns_weight) - || !nonnegative(request.weights.retained_byte_seconds_weight) - || request.weights.cpu_ns_weight == 0.0 - && request.weights.retained_byte_seconds_weight == 0.0 - { - return Err("invalid resource objective weights".into()); - } - let a = &request.context.environment; - let b = &request.exact_environment; - if a.cpu != b.cpu || a.os != b.os || a.runtime != b.runtime { - return Err( - "sketch and exact measurements must share hardware, OS and benchmark runtime".into(), - ); - } - Ok(()) -} - -fn validate_exact(row: &OfflineExactMeasurement) -> Result<(), String> { - let p = &row.provenance; - if row.id.trim().is_empty() - || [ - &p.command, - &p.dataset, - &p.source_revision, - &row.environment.id, - &row.environment.implementation, - &row.environment.implementation_version, - ] - .iter() - .any(|s| s.trim().is_empty()) - || p.repetitions == 0 - || row.measured_at_unix_seconds > row.valid_until_unix_seconds - { - return Err("invalid exact baseline provenance".into()); - } - let m = &row.metrics.resources; - for measurement in [ - &m.cpu.build_cpu_ns, - &m.cpu.update_cpu_ns, - &m.cpu.merge_cpu_ns, - &m.cpu.prepare_cpu_ns, - &m.cpu.read_cpu_ns, - &m.retained_memory_bytes, - &m.peak_memory_bytes, - &m.serialized_bytes, - &m.disk_bytes, - &m.scan_bytes, - ] - .into_iter() - .flatten() - { - if !nonnegative(measurement.value) - || measurement.samples == 0 - || measurement.stddev.is_some_and(|s| !nonnegative(s)) - { - return Err("invalid exact baseline measurement".into()); - } - } - Ok(()) -} - -fn charge(measurement: &Option, count: u64, name: &str) -> Result { - if count == 0 { - return Ok(0.0); - } - let value = measurement - .as_ref() - .ok_or_else(|| format!("missing {name}"))? - .value - * count as f64; - if !nonnegative(value) { - return Err(format!("invalid or overflowing {name}")); - } - Ok(value) -} - -fn estimate_sketch( - row: &OfflineMeasurement, - request: &OfflineComparisonRequest, -) -> Result { - let m = &row.metrics.resources; - let w = &request.workload; - let cpu = charge(&m.cpu.build_cpu_ns, 1, "empty sketch construction CPU")? - + charge( - &m.cpu.update_cpu_ns, - w.input_items_per_state, - "sketch update CPU", - )? - + charge( - &m.cpu.read_cpu_ns, - w.reads_per_state, - "query-matched sketch read CPU", - )? - + charge(&m.cpu.merge_cpu_ns, w.merges_per_state, "sketch merge CPU")? - + crate::empirical_cost::snapshot_prepare_cpu(row) - .ok_or("missing or invalid sketch snapshot preparation CPU")?; - estimate_resources( - cpu, - &m.retained_memory_bytes, - &m.peak_memory_bytes, - m.serialized_bytes.as_ref().map(|m| m.value), - m.disk_bytes.as_ref().map(|m| m.value), - request, - ) -} - -fn estimate_exact( - row: &OfflineExactMeasurement, - request: &OfflineComparisonRequest, -) -> Result { - let m = &row.metrics.resources; - let w = &request.workload; - let cpu = charge(&m.cpu.build_cpu_ns, 1, "exact empty construction CPU")? - + charge( - &m.cpu.update_cpu_ns, - w.input_items_per_state, - "exact update CPU", - )? - + charge(&m.cpu.prepare_cpu_ns, 1, "exact snapshot preparation CPU")? - + charge(&m.cpu.read_cpu_ns, w.reads_per_state, "exact read CPU")?; - estimate_resources( - cpu, - &m.retained_memory_bytes, - &m.peak_memory_bytes, - m.serialized_bytes.as_ref().map(|m| m.value), - m.disk_bytes.as_ref().map(|m| m.value), - request, - ) -} - -fn estimate_resources( - per_state_cpu: f64, - retained: &Option, - peak: &Option, - serialized: Option, - disk: Option, - request: &OfflineComparisonRequest, -) -> Result { - let count = request.workload.state_instances as f64; - let cpu_ns = per_state_cpu * count; - let retained_bytes = retained.as_ref().map(|m| m.value * count); - let retained_byte_seconds = retained_bytes.map(|v| v * request.workload.horizon_seconds); - let peak_bytes_upper_bound = peak.as_ref().map(|m| m.value * count); - let memory_cost = if request.weights.retained_byte_seconds_weight == 0.0 { - 0.0 - } else { - retained_byte_seconds.ok_or("missing retained memory for weighted resource objective")? - * request.weights.retained_byte_seconds_weight - }; - let objective_cost = cpu_ns * request.weights.cpu_ns_weight + memory_cost; - if [ - Some(cpu_ns), - retained_bytes, - retained_byte_seconds, - peak_bytes_upper_bound, - Some(objective_cost), - ] - .into_iter() - .flatten() - .any(|v| !nonnegative(v)) - { - return Err("overflowing resource estimate".into()); - } - Ok(OfflineResourceEstimate { - cpu_ns, - retained_bytes, - retained_byte_seconds, - peak_bytes_upper_bound, - serialized_bytes_per_state: serialized, - disk_bytes_per_state: disk, - objective_cost, - }) -} - -fn nonnegative(value: f64) -> bool { - value.is_finite() && value >= 0.0 -} - -/// Conservative componentwise dominance for known planner sizing families. -/// A deployment still checks its own catalog/layout constraints before binding. -pub fn parameters_at_least(candidate: &SketchParams, minimum: &SketchParams) -> bool { - match (candidate, minimum) { - (SketchParams::Cms { width: a, depth: b }, SketchParams::Cms { width: c, depth: d }) - | ( - SketchParams::CountSketch { width: a, depth: b }, - SketchParams::CountSketch { width: c, depth: d }, - ) => a >= c && b >= d, - (SketchParams::Kll { k: a }, SketchParams::Kll { k: b }) - | (SketchParams::Kmv { k: a }, SketchParams::Kmv { k: b }) - | (SketchParams::Theta { k: a }, SketchParams::Theta { k: b }) => a >= b, - (SketchParams::Hll { precision: a }, SketchParams::Hll { precision: b }) => a >= b, - (SketchParams::DDSketch { alpha: a }, SketchParams::DDSketch { alpha: b }) => { - a.is_finite() && *a > 0.0 && a <= b - } - ( - SketchParams::CmsWithHeap { - width: a, - depth: b, - heap_size: c, - }, - SketchParams::CmsWithHeap { - width: d, - depth: e, - heap_size: f, - }, - ) - | ( - SketchParams::CountSketchWithHeap { - width: a, - depth: b, - heap_size: c, - }, - SketchParams::CountSketchWithHeap { - width: d, - depth: e, - heap_size: f, - }, - ) => a >= d && b >= e && c >= f, - _ => false, - } -} - -#[cfg(test)] -mod tests { - use super::*; - - fn m(value: f64) -> Option { - Some(Measurement { - value, - stddev: None, - samples: 3, - method: Some("synthetic test fixture, not measured".into()), - }) - } - - fn fixture() -> (OfflineComparisonEvidence, OfflineComparisonRequest) { - let mut sketches: EvidenceArtifact = serde_json::from_str(include_str!( - "../tests/data/offline-evidence-synthetic.json" - )) - .unwrap(); - let query = OfflineQueryDescriptor { - kind: "point_frequency".into(), - value_type: "i64".into(), - probe_set: "all_distinct_keys".into(), - }; - let row = &mut sketches.records[0]; - row.metrics.resources.cpu.build_cpu_ns = m(1.0); - row.metrics.resources.cpu.update_cpu_ns = m(1.0); - row.metrics.resources.cpu.read_cpu_ns = m(1.0); - row.metrics.resources.retained_memory_bytes = m(100.0); - row.error.as_mut().unwrap().mean = Some(0.5); - row.error.as_mut().unwrap().query = - serde_json::json!({"kind":"point_frequency","value_type":"i64"}); - let mut wide = row.clone(); - wide.id = "synthetic-wide-cms".into(); - wide.params = SketchParams::Cms { - width: 544, - depth: 5, - }; - wide.metrics.resources.cpu.update_cpu_ns = m(2.0); - wide.metrics.resources.retained_memory_bytes = m(200.0); - wide.error.as_mut().unwrap().mean = Some(0.001); - let context = EvidenceContext { - distribution: row.distribution.clone(), - environment: row.environment.clone(), - now_unix_seconds: 150, - }; - let mut exact_environment = row.environment.clone(); - exact_environment.id = "synthetic-exact".into(); - exact_environment.implementation = "synthetic exact baseline".into(); - let exact = OfflineExactMeasurement { - id: "exact".into(), - distribution: row.distribution.clone(), - environment: exact_environment.clone(), - query: query.clone(), - measured_at_unix_seconds: 100, - valid_until_unix_seconds: 200, - provenance: row.provenance.clone(), - metrics: ExactResourceMeasurements { - resources: asap_types::resources::MeasuredResources { - cpu: asap_types::resources::MeasuredCpu { - build_cpu_ns: m(1.0), - update_cpu_ns: m(5.0), - prepare_cpu_ns: m(1000.0), - read_cpu_ns: m(5.0), - ..Default::default() - }, - retained_memory_bytes: m(1000.0), - ..Default::default() - }, - }, - }; - let metric = row.error.as_ref().unwrap().metric.clone(); - sketches.records.push(wide); - let bindings = sketches - .records - .iter() - .map(|r| MeasurementQueryBinding { - record_id: r.id.clone(), - query: query.clone(), - }) - .collect(); - let request = OfflineComparisonRequest { - context, - exact_environment, - query, - accuracy: EmpiricalAccuracyRequirement { - metric, - max_observed_mean: 0.01, - minimum_trials: 1, - }, - workload: OfflineWorkload { - input_items_per_state: 1000, - reads_per_state: 1000, - merges_per_state: 0, - state_instances: 1, - horizon_seconds: 10.0, - }, - weights: ResourceWeights { - cpu_ns_weight: 1.0, - retained_byte_seconds_weight: 0.0, - }, - formal_minimums: None, - }; - ( - OfflineComparisonEvidence { - schema_version: 1, - timing_contract: "disjoint_live_state_v1".into(), - sketch_evidence: sketches, - query_bindings: bindings, - exact_records: vec![exact], - }, - request, - ) - } - - /// Observed acceptance rejects a cheap inaccurate rung and chooses a larger - /// measured configuration, with every CPU component and state counted. - #[test] - fn accuracy_requirement_changes_selected_configuration_and_cost() { - let (evidence, mut request) = fixture(); - request.formal_minimums = Some(vec![SketchConfiguration { - algorithm: SketchAlgorithm::Cms, - params: SketchParams::Cms { - width: 512, - depth: 5, - }, - }]); - let chosen = recommend_offline(&evidence, &request).unwrap(); - assert_eq!(chosen.selected.record_id, "synthetic-wide-cms"); - assert_eq!(chosen.selected.resources.as_ref().unwrap().cpu_ns, 3001.0); - assert_eq!( - chosen.exact_baseline.resources.as_ref().unwrap().cpu_ns, - 11001.0 - ); - assert_eq!(chosen.estimated_cpu_savings_ns, 8000.0); - assert_eq!(chosen.estimated_retained_bytes_savings, Some(800.0)); - assert!(chosen.checked_formal_minimums); - request.formal_minimums = None; - request.accuracy.max_observed_mean = 1.0; - assert_eq!( - recommend_offline(&evidence, &request) - .unwrap() - .selected - .record_id, - "synthetic-test-cms" - ); - } - - /// Optional sketch preparation is charged once; exact preparation never - /// borrows the independent merge measurement, and snapshot bytes survive. - #[test] - fn preparation_and_exact_resource_dimensions_keep_their_meaning() { - let (mut evidence, request) = fixture(); - let old = recommend_offline(&evidence, &request).unwrap(); - evidence.sketch_evidence.records[1] - .metrics - .resources - .cpu - .prepare_cpu_ns = m(37.0); - let exact = &mut evidence.exact_records[0].metrics.resources; - exact.cpu.merge_cpu_ns = m(1e9); - exact.serialized_bytes = m(256.0); - exact.disk_bytes = m(4096.0); - exact.scan_bytes = m(8000.0); - let changed = recommend_offline(&evidence, &request).unwrap(); - assert_eq!( - changed.selected.resources.as_ref().unwrap().cpu_ns, - old.selected.resources.as_ref().unwrap().cpu_ns + 37.0 - ); - let baseline = changed.exact_baseline.resources.unwrap(); - assert_eq!( - baseline.cpu_ns, - old.exact_baseline.resources.unwrap().cpu_ns - ); - assert_eq!(baseline.serialized_bytes_per_state, Some(256.0)); - assert_eq!(baseline.disk_bytes_per_state, Some(4096.0)); - } - - /// Exact observations validate optional dimensions even when the comparison - /// does not execute the corresponding operation. - #[test] - fn optional_exact_dimensions_cannot_hide_invalid_measurements() { - let selectors: [fn( - &mut asap_types::resources::MeasuredResources, - ) -> &mut Option; 4] = [ - |r| &mut r.cpu.merge_cpu_ns, - |r| &mut r.serialized_bytes, - |r| &mut r.disk_bytes, - |r| &mut r.scan_bytes, - ]; - for select in selectors { - let (mut evidence, request) = fixture(); - *select(&mut evidence.exact_records[0].metrics.resources) = m(-1.0); - assert!(recommend_offline(&evidence, &request).is_err()); - } - } - - /// A family outside the established CMS/CountSketch contract cannot gain - /// an apparently cheap comparison by omitting its preparation measurement. - #[test] - fn sketch_comparison_requires_unknown_preparation_phase() { - let (evidence, request) = fixture(); - let mut row = evidence.sketch_evidence.records[1].clone(); - let original_cpu = estimate_sketch(&row, &request).unwrap().cpu_ns; - row.algorithm = SketchAlgorithm::Kll; - row.params = SketchParams::Kll { k: 269 }; - assert!(estimate_sketch(&row, &request) - .unwrap_err() - .contains("preparation CPU")); - row.metrics.resources.cpu.prepare_cpu_ns = m(37.0); - assert_eq!( - estimate_sketch(&row, &request).unwrap().cpu_ns, - original_cpu + 37.0 - ); - } - - /// Missing required measurements and failing observed-error budgets return - /// the applicable exact baseline, never an optimistically free sketch. - #[test] - fn missing_cost_or_failed_error_acceptance_selects_exact() { - let (mut evidence, mut request) = fixture(); - request.accuracy.max_observed_mean = 0.0; - assert!(recommend_offline(&evidence, &request) - .unwrap() - .selected_sketch() - .is_none()); - request.accuracy.max_observed_mean = 0.01; - evidence.sketch_evidence.records[1] - .metrics - .resources - .cpu - .build_cpu_ns = None; - let chosen = recommend_offline(&evidence, &request).unwrap(); - assert!(chosen.selected_sketch().is_none()); - assert!(chosen.candidates[1] - .rejection - .as_ref() - .unwrap() - .contains("construction")); - evidence.exact_records[0] - .metrics - .resources - .cpu - .prepare_cpu_ns = None; - assert!(recommend_offline(&evidence, &request) - .unwrap_err() - .contains("preparation")); - } - - /// Query, environment, snapshot cardinality, metric, trial count and - /// validity are required independently; nearby evidence is not extrapolated. - #[test] - fn applicability_is_checked_before_recommendation() { - let (evidence, request) = fixture(); - for case in ["query", "environment", "cardinality", "expired", "merges"] { - let mut request = request.clone(); - match case { - "query" => request.query.kind = "total_count".into(), - "environment" => request.exact_environment.cpu = "other CPU".into(), - "cardinality" => request.workload.input_items_per_state = 1001, - "expired" => request.context.now_unix_seconds = 201, - "merges" => request.workload.merges_per_state = 1, - _ => unreachable!(), - } - assert!(recommend_offline(&evidence, &request).is_err(), "{case}"); - } - for case in ["metric", "trials", "binding", "readout"] { - let mut evidence = evidence.clone(); - let mut request = request.clone(); - match case { - "metric" => request.accuracy.metric = "rank_error".into(), - "trials" => request.accuracy.minimum_trials = 100, - "binding" => evidence.query_bindings.clear(), - "readout" => { - for row in &mut evidence.sketch_evidence.records { - row.error.as_mut().unwrap().query["kind"] = - serde_json::json!("total_count"); - } - } - _ => unreachable!(), - } - assert!( - recommend_offline(&evidence, &request) - .unwrap() - .selected_sketch() - .is_none(), - "{case}" - ); - } - } - - /// Resource weights have explicit dimensions; missing memory blocks a - /// memory-weighted objective but does not become a zero-memory estimate. - #[test] - fn resource_objective_and_unknown_memory_are_explicit() { - let (mut evidence, mut request) = fixture(); - evidence.sketch_evidence.records[1] - .metrics - .resources - .retained_memory_bytes = None; - let chosen = recommend_offline(&evidence, &request).unwrap(); - assert!(chosen.selected.resources.unwrap().retained_bytes.is_none()); - request.weights.retained_byte_seconds_weight = 1.0; - assert!(recommend_offline(&evidence, &request) - .unwrap() - .selected_sketch() - .is_none()); - evidence.sketch_evidence.records[1] - .metrics - .resources - .retained_memory_bytes = m(10000.0); - assert!(recommend_offline(&evidence, &request) - .unwrap() - .selected_sketch() - .is_none()); - request.weights.cpu_ns_weight = f64::NAN; - assert!(recommend_offline(&evidence, &request).is_err()); - } - - /// A measured rung below a deployment's formal sizing floor cannot be - /// selected even if its error happened to be zero on the offline input. - #[test] - fn formal_minimums_cannot_be_relaxed_by_observed_accuracy() { - let (evidence, mut request) = fixture(); - request.formal_minimums = Some(vec![SketchConfiguration { - algorithm: SketchAlgorithm::Cms, - params: SketchParams::Cms { - width: 1024, - depth: 5, - }, - }]); - assert!(recommend_offline(&evidence, &request) - .unwrap() - .selected_sketch() - .is_none()); - assert!(!parameters_at_least( - &SketchParams::Cms { - width: 1024, - depth: 4 - }, - &SketchParams::Cms { - width: 512, - depth: 5 - } - )); - assert!(!parameters_at_least( - &SketchParams::CountSketch { - width: 1024, - depth: 5 - }, - &SketchParams::Cms { - width: 512, - depth: 5 - } - )); - } - - /// Consuming-wrapper CPU phases and ambiguous exact generations cannot - /// establish a complete fixed-snapshot comparison. - #[test] - fn comparison_requires_disjoint_phases_and_one_exact_generation() { - let (mut evidence, request) = fixture(); - evidence.timing_contract = "consuming_upstream_wrappers".into(); - assert!(recommend_offline(&evidence, &request) - .unwrap_err() - .contains("disjoint")); - evidence.timing_contract = "disjoint_live_state_v1".into(); - evidence - .exact_records - .push(evidence.exact_records[0].clone()); - assert!(recommend_offline(&evidence, &request) - .unwrap_err() - .contains("ambiguous")); - } -} diff --git a/crates/asap-aware-mapping/src/erp.rs b/crates/asap-aware-mapping/src/erp.rs deleted file mode 100644 index 6b4ad6737..000000000 --- a/crates/asap-aware-mapping/src/erp.rs +++ /dev/null @@ -1,859 +0,0 @@ -//! Distribution-conditioned Error–Resource Profiles (ERP). -//! -//! This is a discrete, auditable profile selector. It does not infer a formal -//! sketch guarantee from benchmark observations and it does not interpolate -//! across distributions, implementations, or parameter points. - -use std::collections::{BTreeMap, HashSet}; - -use serde::{Deserialize, Serialize}; - -pub const ERP_SCHEMA_VERSION: u32 = 1; - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpResourceProfile { - pub memory_bytes: f64, - pub update_cpu_seconds: f64, - pub merge_cpu_seconds: f64, - pub query_cpu_seconds: f64, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpRecord { - pub id: String, - pub sketch: String, - pub implementation: String, - pub parameters: serde_json::Value, - /// Opaque but equality-matched sketch-bench workload descriptor. - pub distribution: serde_json::Value, - pub trials: u32, - pub error_metrics: BTreeMap, - pub resources: ErpResourceProfile, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpArtifact { - pub schema_version: u32, - pub producer_version: String, - pub records: Vec, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum AccuracyMode { - /// ERP may select a smaller empirical configuration. The result is not a - /// formal guarantee and must be labelled accordingly by the deployment. - Empirical, - /// Same selection as empirical mode, but absence of applicable evidence is - /// reported so the caller can fall back to formal sizing or exact execution. - Hybrid, -} - -#[derive(Debug, Clone)] -pub struct ErpSelectionRequest { - pub distribution: serde_json::Value, - pub implementation: Option, - pub allowed_sketches: Vec, - pub error_metric: String, - pub max_error: f64, - pub min_trials: u32, - pub expected_updates: f64, - pub expected_queries: f64, - pub expected_merges: f64, - pub retention_seconds: f64, - pub cpu_weight: f64, - pub byte_second_weight: f64, - pub mode: AccuracyMode, -} - -#[derive(Debug, Clone, PartialEq)] -pub struct ErpSelection<'a> { - pub record: &'a ErpRecord, - pub observed_error: f64, - pub estimated_cost: f64, - pub accuracy_mode: AccuracyMode, -} - -/// Runtime-observable input shape used to match a benchmark scenario. Input -/// volume is a sufficiency gate, not a distance axis: once the benchmark has -/// enough samples, repeating the same stationary distribution adds little -/// information about sketch error. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpDataShape { - pub cardinality: u64, - /// Stable distribution family, for example `uniform`, `zipf`, - /// `power_law`, or `empirical`. Different families are never interpolated. - pub family: String, - /// Family-specific numeric parameters. Zipf uses `exponent`; a continuous - /// power law may use `alpha` and `minimum`. Uniform has no parameters. - #[serde(default)] - pub parameters: BTreeMap, - pub benchmark_events: u64, -} - -/// One hypothesis fitted to the same bounded runtime observation. Lower -/// goodness-of-fit is better; confidence is in [0, 1]. Keeping all plausible -/// fits avoids prematurely classifying unknown data as one named family. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpShapeFit { - pub family: String, - #[serde(default)] - pub parameters: BTreeMap, - pub goodness_of_fit: f64, - pub confidence: f64, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpShapeObservation { - pub cardinality: u64, - pub observed_events: u64, - pub fits: Vec, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub empirical_fingerprint: Option, -} - -#[derive(Debug, Clone)] -pub struct ErpMultiFitSelectionRequest { - pub selection: ErpSelectionRequest, - pub observed: ErpShapeObservation, - pub minimum_benchmark_events: u64, - pub max_log2_cardinality_distance: f64, - pub max_parameter_distance: f64, - pub max_goodness_of_fit: f64, - pub minimum_confidence: f64, - /// Required confidence separation between the best and second-best fit. - pub minimum_confidence_margin: f64, -} - -#[derive(Debug, Clone)] -pub struct ErpNearestSelectionRequest { - pub selection: ErpSelectionRequest, - pub observed: ErpDataShape, - pub minimum_benchmark_events: u64, - pub max_log2_cardinality_distance: f64, - /// Maximum normalized distance for every common distribution parameter. - pub max_parameter_distance: f64, -} - -#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpWindowWorkload { - pub input_updates: u64, - pub query_executions: u64, - pub panes_per_query: u64, - pub retained_panes: u64, - pub materializations: u64, -} - -#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct ErpOperationCounts { - pub updates: u64, - pub merges: u64, - pub queries: u64, - pub retained_sketches: u64, -} - -#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] -pub enum ErpError { - #[error("unsupported ERP schema version {0}")] - UnsupportedVersion(u32), - #[error("invalid ERP artifact: {0}")] - Invalid(&'static str), - #[error("no applicable ERP configuration satisfies the empirical accuracy requirement")] - NoApplicableConfiguration, -} - -impl ErpArtifact { - pub fn validate(&self) -> Result<(), ErpError> { - if self.schema_version != ERP_SCHEMA_VERSION { - return Err(ErpError::UnsupportedVersion(self.schema_version)); - } - if self.producer_version.trim().is_empty() { - return Err(ErpError::Invalid("missing producer version")); - } - let mut ids = HashSet::new(); - for row in &self.records { - if row.id.trim().is_empty() || !ids.insert(row.id.as_str()) { - return Err(ErpError::Invalid("empty or duplicate record id")); - } - if row.sketch.trim().is_empty() - || row.implementation.trim().is_empty() - || row.trials == 0 - || !row.resources.valid() - || row - .error_metrics - .values() - .any(|value| !value.is_finite() || *value < 0.0) - { - return Err(ErpError::Invalid("invalid profile record")); - } - } - Ok(()) - } - - /// Select the least-cost measured parameter point satisfying the requested - /// empirical error. Distribution and implementation matching are exact by - /// design; a later model may add conservative interpolation explicitly. - pub fn select(&self, request: &ErpSelectionRequest) -> Result, ErpError> { - self.validate()?; - if !request.valid() { - return Err(ErpError::Invalid("invalid selection request")); - } - self.records - .iter() - .filter(|row| row.distribution == request.distribution) - .filter(|row| { - request - .implementation - .as_ref() - .is_none_or(|wanted| &row.implementation == wanted) - }) - .filter(|row| { - request.allowed_sketches.is_empty() - || request - .allowed_sketches - .iter() - .any(|name| name == &row.sketch) - }) - .filter(|row| row.trials >= request.min_trials) - .filter_map(|row| { - let observed_error = *row.error_metrics.get(&request.error_metric)?; - (observed_error <= request.max_error).then(|| ErpSelection { - record: row, - observed_error, - estimated_cost: request.cost(&row.resources), - accuracy_mode: request.mode, - }) - }) - .min_by(|left, right| { - left.estimated_cost - .total_cmp(&right.estimated_cost) - .then_with(|| left.record.id.cmp(&right.record.id)) - }) - .ok_or(ErpError::NoApplicableConfiguration) - } - - /// Select against the nearest compatible measured shape. Shape metadata is - /// read from `distribution.erp_shape`, keeping ERP v1 wire compatibility. - pub fn select_nearest( - &self, - request: &ErpNearestSelectionRequest, - ) -> Result, ErpError> { - self.validate()?; - if !request.selection.valid() || !request.valid() { - return Err(ErpError::Invalid("invalid nearest-shape request")); - } - self.records - .iter() - .filter_map(|row| Some((row, data_shape(&row.distribution)?))) - .filter(|(_, shape)| shape.benchmark_events >= request.minimum_benchmark_events) - .filter_map(|(row, shape)| Some((row, request.distance(shape)?))) - .filter(|(_, distance)| *distance <= 1.0) - .filter(|(row, _)| { - request - .selection - .implementation - .as_ref() - .is_none_or(|wanted| &row.implementation == wanted) - && (request.selection.allowed_sketches.is_empty() - || request - .selection - .allowed_sketches - .iter() - .any(|name| name == &row.sketch)) - && row.trials >= request.selection.min_trials - }) - .filter_map(|(row, distance)| { - let observed_error = *row.error_metrics.get(&request.selection.error_metric)?; - (observed_error <= request.selection.max_error).then_some(( - distance, - ErpSelection { - record: row, - observed_error, - estimated_cost: request.selection.cost(&row.resources), - accuracy_mode: request.selection.mode, - }, - )) - }) - .min_by(|(left_distance, left), (right_distance, right)| { - left_distance - .total_cmp(right_distance) - .then_with(|| left.estimated_cost.total_cmp(&right.estimated_cost)) - .then_with(|| left.record.id.cmp(&right.record.id)) - }) - .map(|(_, selection)| selection) - .ok_or(ErpError::NoApplicableConfiguration) - } - - /// Select using every statistically plausible fit for one observation. - /// Ambiguous and poor fits fail closed so callers can use their - /// theoretical-then-exact fallback policy. - pub fn select_multi_fit( - &self, - request: &ErpMultiFitSelectionRequest, - ) -> Result, ErpError> { - self.validate()?; - if !request.selection.valid() || !request.valid() { - return Err(ErpError::Invalid("invalid multi-fit request")); - } - if let Some(selected) = - request - .observed - .empirical_fingerprint - .as_deref() - .and_then(|wanted| { - self.records - .iter() - .filter(|row| benchmark_fingerprint(&row.distribution) == Some(wanted)) - .filter_map(|row| request.selection.evaluate(row).map(|value| (row, value))) - .min_by(|(left_row, left), (right_row, right)| { - left.estimated_cost - .total_cmp(&right.estimated_cost) - .then_with(|| left_row.id.cmp(&right_row.id)) - }) - .map(|(_, selected)| selected) - }) - { - return Ok(selected); - } - let mut fits = request - .observed - .fits - .iter() - .filter(|fit| { - fit.confidence >= request.minimum_confidence - && fit.goodness_of_fit <= request.max_goodness_of_fit - }) - .collect::>(); - fits.sort_by(|left, right| right.confidence.total_cmp(&left.confidence)); - let Some(best) = fits.first() else { - return Err(ErpError::NoApplicableConfiguration); - }; - if fits.get(1).is_some_and(|second| { - best.confidence - second.confidence < request.minimum_confidence_margin - }) { - return Err(ErpError::NoApplicableConfiguration); - } - // Compare every plausible fit with every compatible benchmark shape. - // Confidence and fit quality contribute to the joint score; neither - // field first collapses the observation to one family. - self.records - .iter() - .filter_map(|row| Some((row, data_shape(&row.distribution)?))) - .filter(|(_, shape)| shape.benchmark_events >= request.minimum_benchmark_events) - .flat_map(|(row, shape)| { - fits.iter().filter_map(move |fit| { - let nearest = ErpNearestSelectionRequest { - selection: request.selection.clone(), - observed: ErpDataShape { - cardinality: request.observed.cardinality, - family: fit.family.clone(), - parameters: fit.parameters.clone(), - benchmark_events: request.observed.observed_events, - }, - minimum_benchmark_events: request.minimum_benchmark_events, - max_log2_cardinality_distance: request.max_log2_cardinality_distance, - max_parameter_distance: request.max_parameter_distance, - }; - let shape_distance = nearest.distance(shape.clone())?; - (shape_distance <= 1.0).then_some((row, fit, shape_distance)) - }) - }) - .filter_map(|(row, fit, shape_distance)| { - let selected = request.selection.evaluate(row)?; - let fit_distance = - fit.goodness_of_fit / request.max_goodness_of_fit.max(f64::EPSILON); - let confidence_distance = 1.0 - fit.confidence; - Some(( - shape_distance.max(fit_distance).max(confidence_distance), - selected, - )) - }) - .min_by(|(left_distance, left), (right_distance, right)| { - left_distance - .total_cmp(right_distance) - .then_with(|| left.estimated_cost.total_cmp(&right.estimated_cost)) - .then_with(|| left.record.id.cmp(&right.record.id)) - }) - .map(|(_, selected)| selected) - .ok_or(ErpError::NoApplicableConfiguration) - } -} - -fn benchmark_fingerprint(distribution: &serde_json::Value) -> Option<&str> { - distribution - .pointer("/erp_shape/empirical_fingerprint") - .or_else(|| distribution.pointer("/workload/external/fingerprint")) - .and_then(serde_json::Value::as_str) -} - -fn data_shape(distribution: &serde_json::Value) -> Option { - serde_json::from_value(distribution.get("erp_shape")?.clone()).ok() -} - -impl ErpNearestSelectionRequest { - fn valid(&self) -> bool { - self.minimum_benchmark_events > 0 - && self.observed.cardinality > 0 - && self.max_log2_cardinality_distance.is_finite() - && self.max_log2_cardinality_distance > 0.0 - && !self.observed.family.trim().is_empty() - && self - .observed - .parameters - .values() - .all(|value| value.is_finite()) - && self.max_parameter_distance.is_finite() - && self.max_parameter_distance > 0.0 - } - - fn distance(&self, candidate: ErpDataShape) -> Option { - if candidate.cardinality == 0 - || candidate.family != self.observed.family - || candidate - .parameters - .keys() - .ne(self.observed.parameters.keys()) - { - return None; - } - let cardinality = ((candidate.cardinality as f64).log2() - - (self.observed.cardinality as f64).log2()) - .abs() - / self.max_log2_cardinality_distance; - let parameters = candidate - .parameters - .iter() - .map(|(name, value)| { - let observed = self.observed.parameters.get(name)?; - (value.is_finite() && observed.is_finite()) - .then_some((value - observed).abs() / self.max_parameter_distance) - }) - .collect::>>()? - .into_iter() - .fold(0.0_f64, f64::max); - Some(cardinality.max(parameters)) - } -} - -impl ErpMultiFitSelectionRequest { - fn valid(&self) -> bool { - self.observed.cardinality > 0 - && self.observed.observed_events > 0 - && (!self.observed.fits.is_empty() || self.observed.empirical_fingerprint.is_some()) - && self.max_goodness_of_fit.is_finite() - && self.max_goodness_of_fit >= 0.0 - && self.minimum_confidence.is_finite() - && (0.0..=1.0).contains(&self.minimum_confidence) - && self.minimum_confidence_margin.is_finite() - && (0.0..=1.0).contains(&self.minimum_confidence_margin) - && self.observed.fits.iter().all(|fit| { - !fit.family.trim().is_empty() - && fit.goodness_of_fit.is_finite() - && fit.goodness_of_fit >= 0.0 - && fit.confidence.is_finite() - && (0.0..=1.0).contains(&fit.confidence) - && fit.parameters.values().all(|value| value.is_finite()) - }) - } -} - -impl ErpOperationCounts { - /// Compose atomic benchmark costs with a pane-based window plan. A query - /// over one pane needs no merge; N panes need N-1 merges. - pub fn from_window(workload: ErpWindowWorkload) -> Self { - Self { - updates: workload - .input_updates - .saturating_mul(workload.materializations), - merges: workload - .query_executions - .saturating_mul(workload.panes_per_query.saturating_sub(1)), - queries: workload.query_executions, - retained_sketches: workload - .retained_panes - .saturating_mul(workload.materializations), - } - } - - pub fn cpu_seconds(self, resources: &ErpResourceProfile) -> f64 { - self.updates as f64 * resources.update_cpu_seconds - + self.merges as f64 * resources.merge_cpu_seconds - + self.queries as f64 * resources.query_cpu_seconds - } -} - -impl ErpResourceProfile { - fn valid(&self) -> bool { - [ - self.memory_bytes, - self.update_cpu_seconds, - self.merge_cpu_seconds, - self.query_cpu_seconds, - ] - .into_iter() - .all(|value| value.is_finite() && value >= 0.0) - } -} - -impl ErpSelectionRequest { - fn evaluate<'a>(&self, row: &'a ErpRecord) -> Option> { - if self - .implementation - .as_ref() - .is_some_and(|wanted| &row.implementation != wanted) - || (!self.allowed_sketches.is_empty() - && !self.allowed_sketches.iter().any(|name| name == &row.sketch)) - || row.trials < self.min_trials - { - return None; - } - let observed_error = *row.error_metrics.get(&self.error_metric)?; - (observed_error <= self.max_error).then(|| ErpSelection { - record: row, - observed_error, - estimated_cost: self.cost(&row.resources), - accuracy_mode: self.mode, - }) - } - - fn valid(&self) -> bool { - !self.error_metric.trim().is_empty() - && self.max_error.is_finite() - && self.max_error >= 0.0 - && self.min_trials > 0 - && [ - self.expected_updates, - self.expected_queries, - self.expected_merges, - self.retention_seconds, - self.cpu_weight, - self.byte_second_weight, - ] - .into_iter() - .all(|value| value.is_finite() && value >= 0.0) - } - - fn cost(&self, resources: &ErpResourceProfile) -> f64 { - self.cpu_weight - * (self.expected_updates * resources.update_cpu_seconds - + self.expected_queries * resources.query_cpu_seconds - + self.expected_merges * resources.merge_cpu_seconds) - + self.byte_second_weight * self.retention_seconds * resources.memory_bytes - } -} - -#[cfg(test)] -mod tests { - use super::*; - - fn row(id: &str, width: u64, error: f64, memory: f64) -> ErpRecord { - ErpRecord { - id: id.into(), - sketch: "cms".into(), - implementation: "oxide".into(), - parameters: serde_json::json!({"width": width, "depth": 3}), - distribution: serde_json::json!({"synthetic":{"description":{"kind":"zipf"}}}), - trials: 20, - error_metrics: BTreeMap::from([("relative_error".into(), error)]), - resources: ErpResourceProfile { - memory_bytes: memory, - update_cpu_seconds: 1e-7, - merge_cpu_seconds: 1e-5, - query_cpu_seconds: 1e-6, - }, - } - } - - fn request() -> ErpSelectionRequest { - ErpSelectionRequest { - distribution: serde_json::json!({"synthetic":{"description":{"kind":"zipf"}}}), - implementation: Some("oxide".into()), - allowed_sketches: vec!["cms".into()], - error_metric: "relative_error".into(), - max_error: 0.01, - min_trials: 10, - expected_updates: 1_000.0, - expected_queries: 100.0, - expected_merges: 0.0, - retention_seconds: 60.0, - cpu_weight: 1.0, - byte_second_weight: 1e-9, - mode: AccuracyMode::Hybrid, - } - } - - #[test] - fn selects_cheapest_applicable_accurate_configuration() { - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![ - row("too-small", 256, 0.02, 6_144.0), - row("winner", 512, 0.009, 12_288.0), - row("overprovisioned", 2048, 0.001, 49_152.0), - ], - }; - let selected = artifact.select(&request()).unwrap(); - assert_eq!(selected.record.id, "winner"); - } - - #[test] - fn distribution_mismatch_fails_closed() { - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![row("cms", 512, 0.009, 12_288.0)], - }; - let mut request = request(); - request.distribution = serde_json::json!({"external":{"dataset":"production"}}); - assert_eq!( - artifact.select(&request), - Err(ErpError::NoApplicableConfiguration) - ); - } - - fn multi_fit_request() -> ErpMultiFitSelectionRequest { - ErpMultiFitSelectionRequest { - selection: request(), - observed: ErpShapeObservation { - cardinality: 1_000, - observed_events: 20_000, - fits: vec![ErpShapeFit { - family: "zipf".into(), - parameters: BTreeMap::from([("exponent".into(), 1.2)]), - goodness_of_fit: 0.03, - confidence: 0.95, - }], - empirical_fingerprint: None, - }, - minimum_benchmark_events: 10_000, - max_log2_cardinality_distance: 1.0, - max_parameter_distance: 0.2, - max_goodness_of_fit: 0.1, - minimum_confidence: 0.8, - minimum_confidence_margin: 0.1, - } - } - - #[test] - fn multi_fit_selects_only_confident_well_fitting_family() { - let mut measured = row("zipf", 512, 0.009, 12_288.0); - measured.distribution = serde_json::json!({"erp_shape": { - "cardinality": 1000, "family": "zipf", - "parameters": {"exponent": 1.22}, "benchmark_events": 20000 - }}); - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![measured], - }; - assert_eq!( - artifact - .select_multi_fit(&multi_fit_request()) - .unwrap() - .record - .id, - "zipf" - ); - } - - #[test] - fn multi_fit_rejects_ambiguous_or_poor_observations() { - let mut measured = row("zipf", 512, 0.009, 12_288.0); - measured.distribution = serde_json::json!({"erp_shape": { - "cardinality": 1000, "family": "zipf", - "parameters": {"exponent": 1.2}, "benchmark_events": 20000 - }}); - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![measured], - }; - let mut ambiguous = multi_fit_request(); - ambiguous.observed.fits.push(ErpShapeFit { - family: "normal".into(), - parameters: BTreeMap::from([("mean".into(), 0.0), ("stddev".into(), 1.0)]), - goodness_of_fit: 0.04, - confidence: 0.90, - }); - assert_eq!( - artifact.select_multi_fit(&ambiguous), - Err(ErpError::NoApplicableConfiguration) - ); - ambiguous.observed.fits.truncate(1); - ambiguous.observed.fits[0].goodness_of_fit = 0.5; - assert_eq!( - artifact.select_multi_fit(&ambiguous), - Err(ErpError::NoApplicableConfiguration) - ); - } - - #[test] - fn multi_fit_jointly_ranks_all_plausible_families() { - let mut distant_high_confidence = row("zipf-distant", 512, 0.009, 1.0); - distant_high_confidence.distribution = serde_json::json!({"erp_shape": { - "cardinality": 1000, "family": "zipf", - "parameters": {"exponent": 1.39}, "benchmark_events": 20000 - }}); - let mut close_lower_confidence = row("normal-close", 512, 0.009, 100.0); - close_lower_confidence.distribution = serde_json::json!({"erp_shape": { - "cardinality": 1000, "family": "normal", - "parameters": {"mean": 4.0, "stddev": 1.0}, "benchmark_events": 20000 - }}); - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![distant_high_confidence, close_lower_confidence], - }; - let mut request = multi_fit_request(); - request.observed.fits.push(ErpShapeFit { - family: "normal".into(), - parameters: BTreeMap::from([("mean".into(), 4.0), ("stddev".into(), 1.0)]), - goodness_of_fit: 0.01, - confidence: 0.82, - }); - request.minimum_confidence_margin = 0.1; - assert_eq!( - artifact.select_multi_fit(&request).unwrap().record.id, - "normal-close" - ); - } - - #[test] - fn exact_empirical_fingerprint_precedes_fits_and_requires_identity() { - let mut exact = row("exact-trace", 512, 0.009, 100.0); - exact.distribution = serde_json::json!({"workload": {"external": { - "dataset": "trace-a", "fingerprint": "sha256:abc" - }}}); - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![exact], - }; - let mut request = multi_fit_request(); - request.observed.empirical_fingerprint = Some("sha256:abc".into()); - request.observed.fits.clear(); - assert_eq!( - artifact.select_multi_fit(&request).unwrap().record.id, - "exact-trace" - ); - request.observed.empirical_fingerprint = Some("sha256:different".into()); - assert_eq!( - artifact.select_multi_fit(&request), - Err(ErpError::NoApplicableConfiguration) - ); - } - - /// Verifies the JSON contract emitted by sketch-bench is directly - /// consumable without a backend-specific translation layer. - #[test] - fn deserializes_sketch_bench_wire_format() { - let json = serde_json::json!({ - "schema_version": 1, - "producer_version": "sketch-bench-rev", - "records": [{ - "id": "erp-0", - "sketch": "cms", - "implementation": "oxide", - "parameters": {"width": 512, "depth": 3}, - "distribution": {"synthetic": {"description": {"kind": "zipf"}}}, - "trials": 20, - "error_metrics": {"relative_error": 0.009}, - "resources": { - "memory_bytes": 12288.0, - "update_cpu_seconds": 1e-7, - "merge_cpu_seconds": 1e-5, - "query_cpu_seconds": 1e-6 - } - }] - }); - let artifact: ErpArtifact = serde_json::from_value(json).unwrap(); - let selected = artifact.select(&request()).unwrap(); - assert_eq!(selected.record.id, "erp-0"); - assert_eq!(selected.accuracy_mode, AccuracyMode::Hybrid); - } - - #[test] - fn nearest_shape_prefers_cardinality_and_skew_then_cost() { - let mut close = row("close", 512, 0.009, 12_288.0); - close.distribution = serde_json::json!({"erp_shape": { - "cardinality": 1000, "family": "zipf", "parameters": {"exponent": 1.2}, "benchmark_events": 100000 - }}); - let mut cheap_but_far = row("far", 256, 0.009, 1.0); - cheap_but_far.distribution = serde_json::json!({"erp_shape": { - "cardinality": 8000, "family": "zipf", "parameters": {"exponent": 1.2}, "benchmark_events": 100000 - }}); - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![cheap_but_far, close], - }; - let selected = artifact - .select_nearest(&ErpNearestSelectionRequest { - selection: request(), - observed: ErpDataShape { - cardinality: 1200, - family: "zipf".into(), - parameters: BTreeMap::from([("exponent".into(), 1.1)]), - benchmark_events: 0, - }, - minimum_benchmark_events: 10_000, - max_log2_cardinality_distance: 4.0, - max_parameter_distance: 0.5, - }) - .unwrap(); - assert_eq!(selected.record.id, "close"); - } - - #[test] - fn nearest_shape_rejects_distribution_family_and_small_benchmarks() { - let mut row = row("uniform", 512, 0.009, 12_288.0); - row.distribution = serde_json::json!({"erp_shape": { - "cardinality": 1000, "family": "uniform", "parameters": {}, "benchmark_events": 999 - }}); - let artifact = ErpArtifact { - schema_version: ERP_SCHEMA_VERSION, - producer_version: "bench-1".into(), - records: vec![row], - }; - let nearest = ErpNearestSelectionRequest { - selection: request(), - observed: ErpDataShape { - cardinality: 1000, - family: "zipf".into(), - parameters: BTreeMap::from([("exponent".into(), 1.0)]), - benchmark_events: 0, - }, - minimum_benchmark_events: 1_000, - max_log2_cardinality_distance: 1.0, - max_parameter_distance: 0.5, - }; - assert_eq!( - artifact.select_nearest(&nearest), - Err(ErpError::NoApplicableConfiguration) - ); - } - - #[test] - fn pane_window_composes_atomic_operation_costs() { - let counts = ErpOperationCounts::from_window(ErpWindowWorkload { - input_updates: 1_000, - query_executions: 10, - panes_per_query: 12, - retained_panes: 24, - materializations: 2, - }); - assert_eq!(counts.updates, 2_000); - assert_eq!(counts.merges, 110); - assert_eq!(counts.queries, 10); - assert_eq!(counts.retained_sketches, 48); - assert!((counts.cpu_seconds(&row("cost", 1, 0.0, 0.0).resources) - 0.00131).abs() < 1e-12); - } -} diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs deleted file mode 100644 index 95e50a552..000000000 --- a/crates/asap-aware-mapping/src/lib.rs +++ /dev/null @@ -1,236 +0,0 @@ -//! `asap-plan` — the cost-aware optimizer layer over the pre-ASAP intent algebra. -//! -//! This crate sits between the language-agnostic IR ([`asap_ir`]) and -//! any runtime: it consumes pre-ASAP [`QueryExpr`](asap_types::pre_asap::QueryExpr) -//! DAGs and makes the cost-aware decisions the pre-ASAP IR deliberately -//! leaves open — which sketch (if any) realises each approximate intent. -//! -//! **Common sub-expression elimination (CSE) is not this crate's job.** -//! Detection is a primary pass over the pre-ASAP `QueryExpr` IR itself -//! (`asap_types::pre_asap`, design tracked in issue #223), run before a -//! DAG ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 -//! for why (batch query optimization needs to see shared work across a -//! `QueryWorkload` before summary binding, not after). This crate may -//! eventually run a second, narrower CSE pass of its own over an -//! already-bound `SummaryExpr`/`SummaryNode` DAG, recognizing sharing that's invisible -//! at the pre-ASAP level by construction — e.g. `Quantile(x, 0.99)` and -//! `Quantile(x, 0.95)` are structurally distinct `AggIntent`s but can -//! still share one built sketch, read out twice. That post-ASAP pass is -//! secondary to, and downstream of, the primary pre-ASAP pass, not a -//! replacement for it. -//! -//! It depends only on the IR crate, never on a front end — the layering -//! invariant (arrows point up) holds here too. -//! -//! Post-lowering **canonicalization** is *not* here: it landed in -//! `asap_types::pre_asap::canonicalize`, run inside the shared `resolve_root` -//! so every front end normalizes before the pre-ASAP IR leaves resolution -//! (issue #34, closed). -//! -//! ## Planning workflows -//! -//! Candidate search returns [`CandidateLogicalASAPDAGs`](replacement::CandidateLogicalASAPDAGs), a compact -//! logical choice space with one [`TargetSubDAGCandidates`] per target sub-DAG. -//! [`ReplacementStrategy`] implementations propose local alternatives; search -//! applies the applicable semantic and accuracy checks. Candidate presence does -//! not certify physical deployability or an unknown accuracy guarantee. -//! -//! Integrators choose among these workflows: -//! -//! - Inspect the candidate space, optionally using [`CandidateLogicalASAPDAGs::cost_sorted`] -//! to obtain ranked views, and perform selection downstream. -//! - Call [`CandidateLogicalASAPDAGs::global_selection`] once for the workload, then -//! [`GlobalSelection::assemble_selected_dag`] for each query root. This -//! coordinates logical choices and preserves shared nodes, but makes no -//! summary-maintenance lifecycle decision. -//! - When Planner owns maintenance-versus-recomputation decisions, use -//! [`global_selection_with_summary_maintenance_lifecycles`] followed by -//! [`assemble_selected_dag_with_summary_maintenance_lifecycles`] per root. -//! This alternative workflow returns [`SummaryMaintenanceLifecyclePlan`] -//! values containing DAG roots and maintenance decisions; callers do not need -//! to run ordinary selection/assembly first. -//! -//! Models and evidence determine which choices the helpers can justify. -//! Physical operator binding, placement, storage, deployment, and execution -//! remain downstream responsibilities. Neither taking the first candidate nor -//! assembling a logical DAG creates an executable deployment plan. -//! -//! ## Supporting components -//! -//! - [`cost_model`] — the [`CostModel`](cost_model::CostModel) trait every -//! deployment's cost-based sketch selection plugs into (issues #6, #33). -//! `asap-plan` itself only ships [`DefaultCostModel`](cost_model::DefaultCostModel), -//! which preserves [`replacement`]'s built-in static preference order and -//! — via [`CostModel::estimate_cost`](cost_model::CostModel::estimate_cost) -//! — exposes an actual numeric cost per candidate, not just a relative -//! rank, for a caller (e.g. a DAG-visualization view) that wants to show -//! "candidate A costs ≈ X" next to "candidate B costs ≈ Y". -//! - [`explanation`] — this crate's explanation of a replacement: a -//! reporting *view* over [`replacement`]'s candidate-plan space (issue -//! #257, part of #33) that translates every discovered `TargetSubDAG` with -//! a non-trivial candidate list into an -//! [`explanation::ReplacementExplanation`] (why a replacement exists, -//! where, reusing the candidate's own rationale rather than inventing new -//! prose), meant for the same downstream consumer (e.g. a -//! DAG-visualization view) the crate doc's planning workflows section above -//! already names for [`replacement::CandidateLogicalASAPDAGs`] itself. Superseded PR -//! #247's own rule-based traversal, which re-walked the DAG once per -//! optimization before [`replacement::search_workload`] existed to read -//! from instead — see that module's docs for the full reframing. -//! - [`rollup`] — [`rollup::RollupStrategy`] wraps group-by-lattice roll-up -//! reuse (issue #254, part of #33) as a [`ReplacementStrategy`]: given a -//! coarser `Aggregate` target and a caller-supplied sibling set, proposes -//! re-deriving it from an already-computed, strictly finer sibling -//! `Aggregate` over identical child IR instead of an independent pass -//! over the raw source — the cross-aggregate sibling of -//! `pre_asap::cse::share_common_sub_dags`'s identical-sub-DAG sharing. -//! [`rollup::is_legal_rollup_source`] is the standalone legality predicate -//! other axes (e.g. issue #256's `GroupingStrategy`) are expected to -//! consult directly, so it and this module's `RollupStrategy` can never -//! disagree about which siblings qualify. -//! - [`grouping`] — [`grouping::HydraGroupingStrategy`] (issue #256, part of -//! #33) is an additional `ReplacementStrategy`: the orthogonal -//! `GroupingStrategy` axis (one summary instance per `by` subpopulation -//! versus one shared Hydra-family structure serving all of them), offered -//! alongside the candidates [`replacement::SketchAlgorithmStrategy`] -//! enumerates for the same target. -//! - [`rewrite`] — the "semantic-equivalent rewriting (e.g. `avg` → -//! `sum`/`count`) to increase how often the [sharing/sketch] optimizations -//! above apply" degree of freedom `docs/design_docs/asap_aware_mapping.md` -//! names (issue #253, part of #33): [`rewrite::AvgToSumOverCountStrategy`] -//! is a [`replacement::ReplacementStrategy`] that reshapes a bare `avg` -//! node — which [`replacement::realizations_for_intent`] can only -//! dispatch to `Realization::PassThrough`, so it can never be a -//! [`replacement::SharedSubDAGStrategy`] target — into a `sum`/`count` -//! pair under the same grouping, re-divided back by a wrapping `Project`, -//! so those *are* ordinary mergeable accumulators sharing/sketching can -//! reach. It only reshapes; [`replacement::search_workload`]'s cost-based -//! ranking (or a downstream consumer reading [`replacement::CandidateLogicalASAPDAGs`]) -//! is what decides whether the reshaped form is actually worth picking, -//! the same propose-don't-decide split every other strategy here keeps. -//! -//! ## Terminology -//! -//! Schema resolution, candidate realization, and runtime placement are distinct stages. -//! -//! | Term | Meaning | Entry point | -//! |---|---|---| -//! | Schema resolution | Derive input schemas and resolve column names to positions | `asap_types::pre_asap::SchemaResolver::resolve_schema`, `resolve_root` | -//! | Realization | Enumerate ranked physical forms for one aggregate intent | `replacement::realizations_for_intent` | -//! | Replacement | Construct each candidate summary sub-DAG | [`replacement::SketchAlgorithmStrategy`] | -//! | Search | Enumerate and compare alternatives across a workload | [`replacement::search_workload`] | -//! | Runtime placement | Choose deployment locations and concrete executors | Downstream physical plan providers | -//! -//! A related question (tracked alongside issues #6/#33): whether this -//! crate should also own a **matching** predicate — "does an already -//! *available* `Realization` satisfy a *required* one" — the way a -//! database's materialized-view matching / "answering queries using -//! views" layer does. It owns the *question*, not an *answer*: -//! [`replacement::Matcher`] is a trait with no default implementation and -//! no shipped instance, the same shape as [`cost_model::CostModel`] and for -//! the same reason — which `Realization`s are actually *available* -//! anywhere is entirely a downstream deployment's concern (an inventory -//! this crate has no way to see), and even the pure sketch-algebra -//! compatibility rules (e.g. a heap-bearing top-k sketch also satisfying a -//! bare frequency point-query) turned out to have deployment-specific -//! competitors (e.g. single-vs-multi-population re-aggregation) that -//! don't reduce to a fact about a summary family's kind alone. `control_plane`'s own -//! `sketch_algebra::capability::Capability`/`is_satisfied_by` is the -//! reference downstream implementation. -//! -//! - [`accuracy`] — the [`AccuracyModel`](accuracy::AccuracyModel) / -//! [`AccuracyBudgetAllocator`](accuracy::AccuracyBudgetAllocator) -//! extension points (issue #172): the planning-time algebra that derives -//! a machine-readable [`ResultGuarantee`](asap_types::post_asap::ResultGuarantee) -//! for every finalized post-ASAP value, propagates it through -//! approximate-over-approximate compositions under conservative rules -//! (no independence assumptions, unknown statistics stay unknown), and -//! rejects — before any `CostModel` ranks anything — every candidate with -//! no sound rule or one that misses the applicable `AccuracyTarget`. -//! Legality and cost are separate responsibilities; see that module's -//! docs for the pipeline order and the root-vs-per-node precedence rules. - -pub mod accuracy; -pub mod analytical_cost; -pub mod cost_model; -pub mod empirical_comparison; -pub mod empirical_cost; -pub mod empirical_resources; -pub mod erp; -pub mod exact_composition; -pub mod explanation; -mod function_rules; -pub mod grouping; -pub mod pane_sharing; -pub mod pass; -pub mod physical_handoff_cost; -pub mod physical_operator_statistics; -pub mod physical_plan_cost_model; -pub mod query_physical_lowering; -pub mod recurrence; -pub mod replacement; -pub mod rewrite; -pub mod rollup; -pub mod storage_io; -pub mod summary_maintenance_cost; -pub mod summary_maintenance_dag_export; -pub mod summary_maintenance_lifecycle; -#[cfg(test)] -mod test_support; -pub mod topk_reuse; - -pub use accuracy::reconciliation::AccuracyReconciliationStrategy; -pub use accuracy::{ - AccuracyAllocation, AccuracyBudgetAllocator, AccuracyEvidenceProvider, AccuracyModel, - CompositionShape, DefaultAccuracyModel, EqualSplitAllocator, NoAccuracyEvidence, - PropagationStats, WorkloadAccuracyEvidence, -}; -pub use cost_model::CompleteSummaryCandidateEstimate; -pub use cost_model::{ - maintenance_operation_plan_cost_rate, raw_recompute_cost_rate, read_operation_plan_cost_rate, - CostModel, CostProvenance, CostUnit, DefaultCostModel, ExactCompositionCostInputs, - ExactCompositionCostRequest, ValueOperationCapabilities, -}; -pub use exact_composition::{ExactComposition, ExactCompositionStrategy, OperationPlacement}; -pub use explanation::{ - explain_replacements, explain_replacements_with, ExplanationKind, ReplacementExplanation, -}; -pub use grouping::{has_subpopulations, HydraGroupingStrategy}; -pub use pass::{ - optimize, LifecycleInput, MajorPass, OptimizationInput, OptimizationInputError, - OptimizationPass, OptimizeError, PassNameConflict, PassRegistry, PlanOutput, PlanningModels, - QueryLifecyclePlan, -}; -pub use recurrence::{ - evaluation_rate_of, total_cost, update_rate_from_data_workload, CostRate, EvaluationRate, - Horizon, RecurrenceCostExplanation, RecurrenceError, RecurrenceProfile, RootRecurrence, - UpdateRate, -}; -pub use replacement::{ - default_strategies, default_strategies_with, search_workload, search_workload_with, - search_workload_with_targets, summary_candidates, CandidateLogicalASAPDAGs, - CompositionDecision, GlobalSelection, Matcher, Proposals, RankedTargetSubDAGCandidates, - Realization, RealizationError, RecurrenceProfileMap, RejectedCandidate, Replacement, - ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubDAGStrategy, - SketchAlgorithmStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, - MAX_SEARCH_ITERATIONS, -}; -pub use rewrite::{AvgToSumOverCountStrategy, SemanticEquivalentRewriteStrategy}; -pub use summary_maintenance_dag_export::{ - export_summary_maintenance_plan, SummaryMaintenanceDAGExport, - SummaryMaintenanceDeploymentExport, SummaryMaintenanceLifecycleAlternativeExport, -}; -pub use summary_maintenance_lifecycle::{ - assemble_selected_dag_with_summary_maintenance_lifecycles, - enumerate_summary_maintenance_lifecycles, global_selection_with_summary_maintenance_lifecycles, - plan_summary_maintenance_lifecycles, SummaryMaintenanceCapabilities, - SummaryMaintenanceDeployment, SummaryMaintenanceLifecycleAlternative, - SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCandidates, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleChoiceError, - SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecyclePlan, - SummaryMaintenanceLifecyclePlanError, SummaryMaintenanceLifecycleRejection, - SummaryMaintenanceLifecycleSelectionError, SummaryMaintenanceTimingError, WorkloadDemand, -}; -pub use topk_reuse::TopKLimitReuseStrategy; - -pub mod maintained_population; diff --git a/crates/asap-aware-mapping/src/pane_sharing.rs b/crates/asap-aware-mapping/src/pane_sharing.rs deleted file mode 100644 index b5f5bff3a..000000000 --- a/crates/asap-aware-mapping/src/pane_sharing.rs +++ /dev/null @@ -1,113 +0,0 @@ -//! Costed reuse of compatible physical pane producers. The executor supplies -//! an equality key covering source, state, phase and evidence. This pass never -//! changes logical readout windows or assumes compatibility from metric names. - -/// A concrete mergeable-pane implementation and its horizon costs. -#[derive(Debug, Clone)] -pub struct PaneReuseCandidate { - pub compatibility: K, - pub lookback_ms: u64, - /// Build, update, residency and retirement for this producer. Candidates - /// with the same key must use the same unit costs and pane width, making - /// the longest-lived producer sufficient for every readout in the group. - pub producer_cost: f64, - /// Readout cost for all consumers of this distinct producer. - pub read_cost: f64, -} - -#[derive(Debug, PartialEq)] -pub struct SharedPaneGroup { - pub members: Vec, - pub lookback_ms: u64, - pub cost: f64, -} - -/// Select build-once reuse when it is cheaper than independent producers. -/// Nonfinite quotes are ineligible, not zero-cost alternatives. The caller -/// retains independent implementations for candidates absent from the result. -pub fn select_shared_panes(candidates: &[PaneReuseCandidate]) -> Vec { - let mut groups: Vec> = Vec::new(); - for (index, candidate) in candidates.iter().enumerate() { - if candidate.lookback_ms == 0 - || ![candidate.producer_cost, candidate.read_cost] - .into_iter() - .all(|cost| cost.is_finite() && cost >= 0.0) - { - continue; - } - if let Some(group) = groups - .iter_mut() - .find(|group| candidates[group[0]].compatibility == candidate.compatibility) - { - group.push(index); - } else { - groups.push(vec![index]); - } - } - groups - .into_iter() - .filter_map(|members| { - if members.len() < 2 { - return None; - } - let mut independent = 0.0; - let mut producer = 0.0_f64; - let mut reads = 0.0; - let mut lookback_ms = 0; - for &index in &members { - let c = &candidates[index]; - independent += c.producer_cost + c.read_cost; - producer = producer.max(c.producer_cost); - reads += c.read_cost; - lookback_ms = lookback_ms.max(c.lookback_ms); - } - let cost = producer + reads; - (independent.is_finite() && cost.is_finite() && cost < independent).then_some( - SharedPaneGroup { - members, - lookback_ms, - cost, - }, - ) - }) - .collect() -} - -#[cfg(test)] -mod tests { - use super::*; - fn offer(key: &str, lookback_ms: u64, producer_cost: f64) -> PaneReuseCandidate<&str> { - PaneReuseCandidate { - compatibility: key, - lookback_ms, - producer_cost, - read_cost: 2.0, - } - } - // Share source work once while retaining both readout charges and longest history. - #[test] - fn shares_compatible_windows() { - assert_eq!( - select_shared_panes(&[ - offer("a", 60_000, 10.0), - offer("a", 600_000, 15.0), - offer("b", 600_000, 15.0) - ]), - vec![SharedPaneGroup { - members: vec![0, 1], - lookback_ms: 600_000, - cost: 19.0 - }] - ); - } - // Invalid evidence and free producers must not fabricate a sharing benefit. - #[test] - fn excludes_invalid_and_nonbeneficial_quotes() { - for cost in [0.0, f64::NAN, f64::INFINITY, -1.0, f64::MAX] { - assert!( - select_shared_panes(&[offer("a", 60_000, cost), offer("a", 600_000, cost)]) - .is_empty() - ); - } - } -} diff --git a/crates/asap-aware-mapping/src/pass/major.rs b/crates/asap-aware-mapping/src/pass/major.rs deleted file mode 100644 index 3671ece58..000000000 --- a/crates/asap-aware-mapping/src/pass/major.rs +++ /dev/null @@ -1,202 +0,0 @@ -//! [`MajorPass`] — the shipped two-phase algorithm, behind the -//! [`OptimizationPass`](super::OptimizationPass) trait. -//! -//! This is the same pipeline the crate has always run (candidate search, -//! whole-workload selection, per-root assembly); moving it here is what makes -//! it *one* pass rather than *the* algorithm. `ReplacementStrategy` is -//! therefore a concept of this pass, not of the optimization interface. - -use std::rc::Rc; - -use asap_types::post_asap::{share_common_summary_sub_dags, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::types::AccuracyTarget; - -use super::{OptimizationInput, OptimizationPass, OptimizeError, PlanOutput, QueryLifecyclePlan}; -use crate::replacement::{default_strategies_with_evidence, search_workload_with_targets}; -use crate::summary_maintenance_lifecycle::{ - global_selection_with_summary_maintenance_lifecycles, plan_assembled_dag, shared_state_cost, - summary_states, WorkloadDemand, -}; - -/// The shipped algorithm. Unit struct: its strategy set is the crate default, -/// and a caller who wants a different one now has a better option than -/// swapping rules — write another [`OptimizationPass`]. -#[derive(Debug, Default, Clone, Copy)] -pub struct MajorPass; - -impl OptimizationPass for MajorPass { - fn name(&self) -> &'static str { - "major" - } - - fn optimize(&self, input: OptimizationInput<'_>) -> Result { - let workload = input.workload; - let models = input.models; - let strategies = default_strategies_with_evidence(models.cost, models.evidence); - - // `Id` is the entry's position in `QueryWorkload::entries()`, so the - // search result carries the workload binding the lifecycle stage and - // the output both need. CSE may make two identical queries share one - // `Rc`, but it never drops or reorders a root, so this stays aligned. - let roots: Vec<(usize, Rc, Option)> = workload - .entries() - .enumerate() - .map(|(index, (entry, expr))| { - ( - index, - Rc::clone(expr), - Some(entry.requirements.accuracy.target()), - ) - }) - .collect(); - - let space = search_workload_with_targets(roots, &strategies, models.accuracy); - - let lifecycle = input.lifecycle; - - // One index per root, in `CandidateLogicalASAPDAGs::roots` order — which is the order - // the roots went in, which is `entries()` order. - let entry_indices: Vec = (0..workload.len()).collect(); - let demand = WorkloadDemand { - workload: workload.query_workload(), - data_workload: workload.data_workload(), - entry_indices: &entry_indices, - }; - - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - demand, - lifecycle.now_ms, - lifecycle.horizon, - lifecycle.capabilities, - models.cost, - ) - .map_err(OptimizeError::LifecycleSelection)?; - // Each root's lifecycle is planned against the entries that consume - // it — the same binding selection costed it with — not the whole - // workload, so one query's reads never amortize another's state. - let bindings = space - .workload_entries_by_target(demand.workload, &entry_indices) - .map_err(|error| OptimizeError::LifecycleSelection(error.into()))?; - - // Assemble every root, then intern structurally identical summary - // producers across them once, so two queries that selected the same - // `SummaryAgg` reach one `Rc` (consumers dedupe states by pointer). - let mut assembled = Vec::with_capacity(space.roots.len()); - for (entry_index, root) in &space.roots { - let dag = selection - .assemble_selected_dag(root) - .map_err(|source| OptimizeError::LifecycleAssembly { - entry_index: *entry_index, - source: source.into(), - })? - .ok_or_else(|| self.missing_group(*entry_index))?; - assembled.push(dag); - } - let interned = - share_common_summary_sub_dags(assembled.iter().cloned().enumerate().collect()); - let states: Vec<_> = interned - .iter() - .map(|(_, dag)| summary_states(dag)) - .collect(); - - // A state reached from several roots is planned once against all of - // their entries, in every plan that reaches it, so each plan picks - // the same lifecycle for it. When that union cannot be costed the - // roots keep their own, unshared DAG and entries. - let mut shared_entries: Vec<(Rc, Option>)> = Vec::new(); - for (position, (entry_index, root)) in space.roots.iter().enumerate() { - for state in &states[position] { - if shared_entries.iter().any(|(s, _)| Rc::ptr_eq(s, state)) { - continue; - } - let readers: Vec<_> = (0..space.roots.len()) - .filter(|&other| states[other].iter().any(|s| Rc::ptr_eq(s, state))) - .map(|other| &space.roots[other].1) - .collect(); - if readers.iter().all(|reader| Rc::ptr_eq(reader, root)) { - continue; - } - let mut entries: Vec = readers - .iter() - .flat_map(|reader| bindings[&Rc::as_ptr(reader)].iter().copied()) - .collect(); - entries.sort_unstable(); - entries.dedup(); - let cost = shared_state_cost( - state, - WorkloadDemand { - entry_indices: &entries, - ..demand - }, - lifecycle.now_ms, - lifecycle.horizon, - lifecycle.capabilities, - models.cost, - ) - .map_err(|source| OptimizeError::LifecycleAssembly { - entry_index: *entry_index, - source: source.into(), - })?; - shared_entries.push((Rc::clone(state), cost.map(|_| entries))); - } - } - - let mut plans = Vec::with_capacity(space.roots.len()); - for (position, (entry_index, root)) in space.roots.iter().enumerate() { - let mut entries = bindings[&Rc::as_ptr(root)].clone(); - let mut dag = Rc::clone(&interned[position].1); - for (state, shared) in &shared_entries { - if !states[position].iter().any(|s| Rc::ptr_eq(s, state)) { - continue; - } - match shared { - Some(shared) => entries.extend(shared), - None => { - entries = bindings[&Rc::as_ptr(root)].clone(); - dag = Rc::clone(&assembled[position]); - break; - } - } - } - entries.sort_unstable(); - entries.dedup(); - let plan = plan_assembled_dag( - dag, - root, - WorkloadDemand { - entry_indices: &entries, - ..demand - }, - lifecycle.now_ms, - lifecycle.horizon, - lifecycle.capabilities, - models.cost, - ) - .map_err(|source| OptimizeError::LifecycleAssembly { - entry_index: *entry_index, - source, - })?; - plans.push(QueryLifecyclePlan { - entry_index: *entry_index, - plan, - }); - } - Ok(PlanOutput::new(plans)) - } -} - -impl MajorPass { - /// Assembly returns `Ok(None)` only for a target that is not a discovered - /// site. Every root is one — `discover_targets` walks each root and - /// `search_cse_workload_with` gives every discovered target a group — so - /// reaching this means the invariant broke, not that the query had no - /// optimization available. - fn missing_group(&self, entry_index: usize) -> OptimizeError { - OptimizeError::ContractViolation { - pass: self.name(), - detail: format!("entry {entry_index}: root has no candidate group"), - } - } -} diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs deleted file mode 100644 index 5bf42d2e4..000000000 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs +++ /dev/null @@ -1,1320 +0,0 @@ -use super::*; - -pub(super) fn estimate_heterogeneous_summary( - root: &SummaryNode, - deployments: &[CostedSummaryDeployment<'_>], - evidence: &SummaryNodeEvidence, - scope: &ComparisonScope, - raw: &RawInputEvidence, - window_frameworks: &[Option], -) -> Result { - if window_frameworks.len() != deployments.len() { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "window framework assignments", - )); - } - let frameworks_by_node: HashMap<_, _> = deployments - .iter() - .zip(window_frameworks) - .map(|(deployment, framework)| (deployment.summary as *const _, framework)) - .collect(); - validate_summary_edges_and_physical_ids(root, evidence, &frameworks_by_node)?; - fn summary_source_selections( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut Vec, - ) -> Result<(), AnalyticalCostError> { - if !seen.insert(node as *const _) { - return Ok(()); - } - match &node.expr { - SummaryExpr::KeepPreAsap(query) => query_source_selections(query, out)?, - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - summary_source_selections(child, seen, out)? - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - summary_source_selections(child, seen, out)?; - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - summary_source_selections(left, seen, out)?; - summary_source_selections(right, seen, out)?; - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - summary_source_selections(summary_input, seen, out)? - } - } - Ok(()) - } - let evaluation_count = scope.validate()?; - let by_node: HashMap<_, _> = deployments - .iter() - .map(|deployment| (deployment.summary as *const _, deployment)) - .collect(); - let mut cpu_ops = 0.0; - let mut persistent_bytes = 0_u64; - let mut ephemeral_state_bytes = 0_u64; - let mut scans = HashMap::::new(); - let mut physical_states = HashMap::< - String, - ( - SummaryAggregateEvidence, - SummaryMaintenanceLifecycleGuarantee, - String, - Option, - ), - >::new(); - for deployment in deployments { - let node_evidence = evidence - .aggregation(deployment.summary) - .ok_or(AnalyticalCostError::MissingOrStale("summary_agg"))?; - let SummaryExpr::SummaryAgg { child, .. } = &deployment.summary.expr else { - return Err(AnalyticalCostError::UnsupportedCandidate); - }; - let inputs = node_evidence.inputs.validate()?; - validate_arrival_rate(scope.data_arrival, inputs.ingestion_rate_per_second)?; - match node_evidence.source_coverage_index { - Some(index) => { - let declared = - scope - .sources - .get(index) - .ok_or(AnalyticalCostError::MissingComparisonScope( - "summary source coverage", - ))?; - if !matches!(&child.expr, SummaryExpr::KeepPreAsap(_)) - || inputs.initial_input_rows != raw.planning_time_input_rows - || inputs.initial_input_bytes != raw.planning_time_input_bytes - || inputs.initial_source_scan_bytes != raw.planning_time_source_scan_bytes - || inputs.ingestion_rate_per_second != raw.ingestion_rate_per_second - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "source-root bootstrap evolution", - )); - } - let mut actual_selections = Vec::new(); - summary_source_selections(child, &mut HashSet::new(), &mut actual_selections)?; - let actual_selections = deduplicate_source_selections(actual_selections); - let expected = ( - declared.source.clone(), - declared.predicates.clone(), - declared.info_matchers.clone(), - ); - if actual_selections.as_slice() != [expected] { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "summary source lineage", - )); - } - } - None => { - if matches!(&child.expr, SummaryExpr::KeepPreAsap(_)) - || inputs.initial_source_scan_bytes != 0 - || !node_evidence.bootstrap_read_identity.is_empty() - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "intermediate bootstrap source ownership", - )); - } - } - } - validate_guarantee(deployment.guarantee, scope.data_arrival)?; - let logical_state = format!("{:?}", deployment.summary.expr); - let window_framework = (*frameworks_by_node - .get(&(deployment.summary as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale( - "window framework assignment", - ))?) - .clone(); - match physical_states.entry(node_evidence.physical_id.clone()) { - std::collections::hash_map::Entry::Vacant(entry) => { - entry.insert(( - node_evidence.clone(), - deployment.guarantee.clone(), - logical_state, - window_framework, - )); - } - std::collections::hash_map::Entry::Occupied(entry) - if entry.get() - != &( - node_evidence.clone(), - deployment.guarantee.clone(), - logical_state, - window_framework, - ) => - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "summary physical identity", - )); - } - std::collections::hash_map::Entry::Occupied(_) => continue, - } - let ephemeral = matches!( - deployment.guarantee.summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::Ephemeral - ); - let (bootstrap, updates, source_scan_bytes) = if ephemeral { - ( - ephemeral_rows_over_horizon(inputs, scope)?, - 0, - ephemeral_scan_bytes_over_horizon(inputs, raw, scope)?, - ) - } else { - let (bootstrap, updates, _) = - lifecycle_row_counts(inputs, deployment.guarantee, scope)?; - let bootstrap_extra_rows = bootstrap - .checked_sub(inputs.initial_input_rows) - .ok_or(AnalyticalCostError::Overflow)?; - let source_scan_bytes = if node_evidence.source_coverage_index.is_some() { - inputs - .initial_source_scan_bytes - .checked_add( - bootstrap_extra_rows - .checked_mul(raw.arriving_source_row_bytes) - .ok_or(AnalyticalCostError::Overflow)?, - ) - .ok_or(AnalyticalCostError::Overflow)? - } else { - 0 - }; - (bootstrap, updates, source_scan_bytes) - }; - let insert = validated_operator_cpu("insert_cpu_ops", node_evidence.insert_cpu_ops)?; - let insert_calls = bootstrap - .checked_mul(inputs.bootstrap_window_count) - .and_then(|calls| { - updates - .checked_mul(inputs.active_window_count) - .and_then(|updates| calls.checked_add(updates)) - }) - .ok_or(AnalyticalCostError::Overflow)?; - cpu_ops += insert_calls as f64 * insert; - let live_window_count = if ephemeral { - inputs.bootstrap_window_count - } else { - inputs - .active_window_count - .checked_add(inputs.retained_window_count) - .ok_or(AnalyticalCostError::Overflow)? - }; - let state_bytes = live_window_count - .checked_mul(inputs.physical_summary_count) - .and_then(|states| states.checked_mul(inputs.state_bytes_per_summary)) - .ok_or(AnalyticalCostError::Overflow)?; - if ephemeral { - ephemeral_state_bytes = ephemeral_state_bytes - .checked_add(state_bytes) - .ok_or(AnalyticalCostError::Overflow)?; - } else { - persistent_bytes = persistent_bytes - .checked_add(state_bytes) - .ok_or(AnalyticalCostError::Overflow)?; - } - if let Some(source_index) = node_evidence.source_coverage_index { - if node_evidence.bootstrap_read_identity.is_empty() { - return Err(AnalyticalCostError::MissingOrStale( - "bootstrap_read_identity", - )); - } - match scans.entry(node_evidence.bootstrap_read_identity.clone()) { - std::collections::hash_map::Entry::Vacant(entry) => { - entry.insert((source_index, source_scan_bytes)); - } - std::collections::hash_map::Entry::Occupied(entry) - if *entry.get() != (source_index, source_scan_bytes) => - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "bootstrap source bytes", - )); - } - _ => {} - } - } - } - let covered_sources: HashSet<_> = scans.values().map(|(index, _)| *index).collect(); - if covered_sources.len() != scope.sources.len() - || !(0..scope.sources.len()).all(|index| covered_sources.contains(&index)) - { - return Err(AnalyticalCostError::ComparisonScopeMismatch("sources")); - } - - #[expect(clippy::too_many_arguments, reason = "CPU and I/O traversal state")] - fn visit_ops( - node: &SummaryNode, - seen: &mut HashSet, - by_node: &HashMap<*const SummaryNode, &CostedSummaryDeployment<'_>>, - evidence: &SummaryNodeEvidence, - scope: &ComparisonScope, - evaluation_count: u64, - cpu_ops: &mut f64, - io_bytes: &mut u64, - ) -> Result<(), AnalyticalCostError> { - let physical_id = summary_physical_id(node, evidence)?; - if !seen.insert(physical_id) { - return Ok(()); - } - match &node.expr { - SummaryExpr::BinaryOp { lhs, rhs, .. } - | SummaryExpr::RelationalJoin { - left: lhs, - right: rhs, - .. - } => { - let operation = summary_operation_evidence(node, evidence)?.resource(); - *cpu_ops += evaluation_count as f64 - * validated_operator_executions("exact_binary", operation)? as f64 - * validated_operator_cpu("exact_binary", operation.cpu_ops)?; - add_operator_io(io_bytes, operation, evaluation_count)?; - visit_ops( - lhs, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - visit_ops( - rhs, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - - SummaryExpr::ValueOperation { child, .. } => { - let operation = summary_operation_evidence(node, evidence)?.resource(); - *cpu_ops += evaluation_count as f64 - * validated_operator_executions("value_operation", operation)? as f64 - * validated_operator_cpu("value_operation", operation.cpu_ops)?; - add_operator_io(io_bytes, operation, evaluation_count)?; - visit_ops( - child, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - SummaryExpr::KeepPreAsap(_) => { - let retained = evidence - .retained_queries - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap"))?; - if !retained.preprocessing_cpu_ops_over_horizon.is_finite() - || retained.preprocessing_cpu_ops_over_horizon < 0.0 - { - return Err(AnalyticalCostError::InvalidOperationCost( - "keep_pre_asap", - retained.preprocessing_cpu_ops_over_horizon, - )); - } - *cpu_ops += retained.preprocessing_cpu_ops_over_horizon; - } - SummaryExpr::SummaryAgg { child, .. } => { - visit_ops( - child, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - SummaryExpr::SummaryMerge { children, .. } => { - let operation = summary_operation_evidence(node, evidence)?.resource(); - let merge = validated_operator_cpu("summary_merge", operation.cpu_ops)?; - *cpu_ops += evaluation_count as f64 - * validated_operator_executions("summary_merge", operation)? as f64 - * merge; - add_operator_io(io_bytes, operation, evaluation_count)?; - for child in children { - visit_ops( - child, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - } - SummaryExpr::SummarySubtract { left, right } => { - let operation = summary_operation_evidence(node, evidence)?.resource(); - *cpu_ops += evaluation_count as f64 - * validated_operator_executions("summary_subtract", operation)? as f64 - * validated_operator_cpu("summary_subtract", operation.cpu_ops)?; - add_operator_io(io_bytes, operation, evaluation_count)?; - visit_ops( - left, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - visit_ops( - right, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - SummaryExpr::SummaryDelete { summary_input, .. } => { - let delete = summary_operation_evidence(node, evidence)?; - let SummaryOperatorEvidence::Delete { - resource: operation, - events_per_second, - routing_fanout, - } = delete - else { - unreachable!("operation kind was validated") - }; - let state_ptr = evidence - .operation_state_owners - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("summary_delete_owner"))?; - fn collect_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut Vec<*const SummaryNode>, - ) { - if !seen.insert(node as *const _) { - return; - } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - out.push(node as *const _); - collect_aggs(child, seen, out); - } - SummaryExpr::ValueOperation { child, .. } => collect_aggs(child, seen, out), - SummaryExpr::SummaryMerge { children, .. } => { - children - .iter() - .for_each(|child| collect_aggs(child, seen, out)); - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - collect_aggs(left, seen, out); - collect_aggs(right, seen, out); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_aggs(summary_input, seen, out) - } - SummaryExpr::KeepPreAsap(_) => {} - } - } - let mut reachable = Vec::new(); - collect_aggs(summary_input, &mut HashSet::new(), &mut reachable); - if reachable.as_slice() != [*state_ptr] { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "summary delete owner", - )); - } - let deployment = by_node - .get(state_ptr) - .copied() - .ok_or(AnalyticalCostError::MissingOrStale("summary_delete_owner"))?; - let state = evidence - .aggregation(deployment.summary) - .ok_or(AnalyticalCostError::MissingOrStale("summary_delete_owner"))?; - let (_, _, active_ms) = - lifecycle_row_counts(state.inputs, deployment.guarantee, scope)?; - if !events_per_second.is_finite() || *events_per_second < 0.0 { - return Err(AnalyticalCostError::InvalidIngestionRate( - *events_per_second, - )); - } - if *routing_fanout == 0 { - return Err(AnalyticalCostError::MissingOrZero("delete_routing_fanout")); - } - let delete_events = (events_per_second * active_ms as f64 / 1_000.0).ceil() - * *routing_fanout as f64; - if !delete_events.is_finite() || delete_events > u64::MAX as f64 { - return Err(AnalyticalCostError::Overflow); - } - let delete_events = delete_events as u64; - *cpu_ops += delete_events as f64 - * validated_operator_executions("summary_delete", operation)? as f64 - * validated_operator_cpu("summary_delete", operation.cpu_ops)?; - add_operator_io(io_bytes, operation, delete_events)?; - visit_ops( - summary_input, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - let operation = summary_operation_evidence(node, evidence)?.resource(); - *cpu_ops += evaluation_count as f64 - * validated_operator_executions("summary_readout", operation)? as f64 - * validated_operator_cpu("summary_readout", operation.cpu_ops)?; - add_operator_io(io_bytes, operation, evaluation_count)?; - visit_ops( - summary_input, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - let join = evidence - .joins - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("summary_join"))?; - if !join.cpu_ops_per_execution.is_finite() - || join.cpu_ops_per_execution <= 0.0 - || join.working_memory_bytes == 0 - || join.executions_per_evaluation == 0 - { - return Err(AnalyticalCostError::MissingOrStale("summary_join")); - } - *cpu_ops += evaluation_count as f64 - * join.executions_per_evaluation as f64 - * join.cpu_ops_per_execution; - let join_io = join - .io_bytes_per_execution - .ok_or(AnalyticalCostError::MissingOrStale("summary_join_io"))?; - *io_bytes = io_bytes - .checked_add( - join_io - .checked_mul(join.executions_per_evaluation) - .and_then(|bytes| bytes.checked_mul(evaluation_count)) - .ok_or(AnalyticalCostError::Overflow)?, - ) - .ok_or(AnalyticalCostError::Overflow)?; - visit_ops( - outer, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - visit_ops( - inner, - seen, - by_node, - evidence, - scope, - evaluation_count, - cpu_ops, - io_bytes, - )?; - } - } - Ok(()) - } - - let mut operator_io_bytes = 0; - visit_ops( - root, - &mut HashSet::new(), - &by_node, - evidence, - scope, - evaluation_count, - &mut cpu_ops, - &mut operator_io_bytes, - )?; - let transient_bytes = estimate_transient_liveness(root, evidence)?; - if !cpu_ops.is_finite() { - return Err(AnalyticalCostError::Overflow); - } - Ok(ResourceEstimate::new( - cpu_ops, - persistent_bytes - .checked_add(transient_bytes) - .and_then(|bytes| bytes.checked_add(ephemeral_state_bytes)) - .ok_or(AnalyticalCostError::Overflow)?, - scans - .values() - .try_fold(operator_io_bytes, |sum, (_, bytes)| { - sum.checked_add(*bytes).ok_or(AnalyticalCostError::Overflow) - })?, - )) -} - -fn add_operator_io( - total: &mut u64, - operation: &SummaryOperatorResourceEvidence, - execution_units: u64, -) -> Result<(), AnalyticalCostError> { - let bytes = operation - .io_bytes_per_execution - .ok_or(AnalyticalCostError::MissingOrStale("summary operator io"))?; - if operation.executions_per_evaluation == 0 { - return Err(AnalyticalCostError::MissingOrStale( - "summary operator executions", - )); - } - *total = total - .checked_add( - bytes - .checked_mul(operation.executions_per_evaluation) - .and_then(|value| value.checked_mul(execution_units)) - .ok_or(AnalyticalCostError::Overflow)?, - ) - .ok_or(AnalyticalCostError::Overflow)?; - Ok(()) -} - -fn validate_summary_edges_and_physical_ids( - root: &SummaryNode, - evidence: &SummaryNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, -) -> Result<(), AnalyticalCostError> { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - vec![child] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|child| child.as_ref()).collect() - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - } - } - fn metadata( - node: &SummaryNode, - evidence: &SummaryNodeEvidence, - ) -> Result<(String, Vec, EdgeStatistics), AnalyticalCostError> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - let retained = evidence - .retained_queries - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap"))?; - Ok((retained.physical_id.clone(), vec![], retained.output)) - } - SummaryExpr::SummaryAgg { .. } => { - let value = evidence - .aggregations - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("summary_agg"))?; - Ok((value.physical_id.clone(), vec![value.input], value.output)) - } - SummaryExpr::SummaryJoin { .. } => { - let value = evidence - .joins - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("summary_join"))?; - Ok(( - value.physical_id.clone(), - value.inputs.clone(), - value.output, - )) - } - _ => { - let value = summary_operation_evidence(node, evidence)?.resource(); - Ok(( - value.physical_id.clone(), - value.inputs.clone(), - value.output, - )) - } - } - } - fn visit( - node: &SummaryNode, - evidence: &SummaryNodeEvidence, - frameworks_by_node: &HashMap<*const SummaryNode, &Option>, - seen: &mut HashSet<*const SummaryNode>, - physical: &mut HashMap, EdgeStatistics, String)>, - ) -> Result { - if !seen.insert(node as *const _) { - return metadata(node, evidence).map(|(_, _, output)| output); - } - let child_nodes = children(node); - let child_outputs = child_nodes - .iter() - .map(|child| visit(child, evidence, frameworks_by_node, seen, physical)) - .collect::, _>>()?; - let child_physical_ids = child_nodes - .iter() - .map(|child| summary_physical_id(child, evidence)) - .collect::, _>>()?; - let (id, inputs, output) = metadata(node, evidence)?; - let local_fingerprint = match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - format!("{:?}", evidence.retained_queries.get(&(node as *const _))) - } - SummaryExpr::SummaryAgg { .. } => { - format!("{:?}", evidence.aggregations.get(&(node as *const _))) - } - SummaryExpr::SummaryJoin { .. } => { - format!("{:?}", evidence.joins.get(&(node as *const _))) - } - _ => format!("{:?}", evidence.operations.get(&(node as *const _))), - }; - // A provider identity names the complete physical operator, including - // its inputs. Equal local widths/costs do not make operators consuming - // different physical children the same deployment. - let framework = frameworks_by_node.get(&(node as *const _)); - let fingerprint = format!( - "logical={:?}|framework={framework:?}|{local_fingerprint}|children={child_physical_ids:?}", - node.expr - ); - if id.is_empty() - || inputs != child_outputs - || !output.is_consistent() - || inputs.iter().any(|edge| !edge.is_consistent()) - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "summary physical edge statistics", - )); - } - match physical.entry(id) { - std::collections::hash_map::Entry::Vacant(entry) => { - entry.insert((inputs, output, fingerprint)); - } - std::collections::hash_map::Entry::Occupied(entry) - if entry.get() != &(inputs, output, fingerprint) => - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "summary physical identity", - )); - } - _ => {} - } - Ok(output) - } - visit( - root, - evidence, - frameworks_by_node, - &mut HashSet::new(), - &mut HashMap::new(), - ) - .map(|_| ()) -} - -fn summary_physical_id( - node: &SummaryNode, - evidence: &SummaryNodeEvidence, -) -> Result { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => evidence - .retained_queries - .get(&(node as *const _)) - .map(|value| value.physical_id.clone()), - SummaryExpr::SummaryAgg { .. } => evidence - .aggregations - .get(&(node as *const _)) - .map(|value| value.physical_id.clone()), - SummaryExpr::SummaryJoin { .. } => evidence - .joins - .get(&(node as *const _)) - .map(|value| value.physical_id.clone()), - _ => summary_operation_evidence(node, evidence) - .ok() - .map(|value| value.resource().physical_id.clone()), - } - .ok_or(AnalyticalCostError::MissingOrStale( - "summary physical identity", - )) -} - -/// Simulate a deterministic child-before-parent physical schedule. Completed -/// child output buffers remain live until their final consumer executes; -/// operator workspace and its output buffer coexist during that execution. -pub(super) fn estimate_transient_liveness( - root: &SummaryNode, - evidence: &SummaryNodeEvidence, -) -> Result { - fn children(node: &SummaryNode) -> Vec<&SummaryNode> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::SummaryAgg { child, .. } | SummaryExpr::ValueOperation { child, .. } => { - vec![child] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|child| child.as_ref()).collect() - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - } - } - fn visit<'a>( - node: &'a SummaryNode, - evidence: &SummaryNodeEvidence, - seen: &mut HashSet, - uses: &mut HashMap, - order: &mut Vec<&'a SummaryNode>, - ) -> Result<(), AnalyticalCostError> { - if !seen.insert(summary_physical_id(node, evidence)?) { - return Ok(()); - } - for child in children(node) { - *uses - .entry(summary_physical_id(child, evidence)?) - .or_default() += 1; - visit(child, evidence, seen, uses, order)?; - } - order.push(node); - Ok(()) - } - fn memory( - node: &SummaryNode, - evidence: &SummaryNodeEvidence, - ) -> Result<(u64, u64), AnalyticalCostError> { - match &node.expr { - SummaryExpr::KeepPreAsap(_) => evidence - .retained_queries - .get(&(node as *const _)) - .map(|value| (value.working_memory_bytes, value.output_buffer_bytes)) - .ok_or(AnalyticalCostError::MissingOrStale("keep_pre_asap")), - SummaryExpr::SummaryAgg { .. } => Ok((0, 0)), - SummaryExpr::SummaryJoin { .. } => evidence - .joins - .get(&(node as *const _)) - .map(|value| (value.working_memory_bytes, value.output_buffer_bytes)) - .ok_or(AnalyticalCostError::MissingOrStale("summary_join")), - SummaryExpr::SummaryMerge { .. } - | SummaryExpr::BinaryOp { .. } - | SummaryExpr::RelationalJoin { .. } - | SummaryExpr::ValueOperation { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryEstimate { .. } => { - let value = summary_operation_evidence(node, evidence)?.resource(); - Ok((value.working_memory_bytes, value.output_buffer_bytes)) - } - } - } - - let mut uses = HashMap::new(); - let mut order = Vec::new(); - visit(root, evidence, &mut HashSet::new(), &mut uses, &mut order)?; - let outputs: HashMap<_, _> = order - .iter() - .map(|node| { - memory(node, evidence) - .and_then(|(_, output)| summary_physical_id(node, evidence).map(|id| (id, output))) - }) - .collect::>()?; - let mut live = 0_u64; - let mut peak = 0_u64; - for node in order { - let (workspace, output) = memory(node, evidence)?; - peak = peak.max( - live.checked_add(workspace) - .and_then(|bytes| bytes.checked_add(output)) - .ok_or(AnalyticalCostError::Overflow)?, - ); - live = live - .checked_add(output) - .ok_or(AnalyticalCostError::Overflow)?; - for child in children(node) { - let child_id = summary_physical_id(child, evidence)?; - let remaining = - uses.get_mut(&child_id) - .ok_or(AnalyticalCostError::InvalidPhysicalDAG( - "missing summary consumer count", - ))?; - *remaining -= 1; - if *remaining == 0 { - live = live - .checked_sub(outputs[&child_id]) - .ok_or(AnalyticalCostError::Overflow)?; - } - } - } - Ok(peak) -} -#[cfg(test)] -pub(super) fn evidence_nodes(root: &SummaryNode) -> (Vec<&SummaryNode>, Vec<&SummaryNode>) { - fn visit<'a>( - node: &'a SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - aggregations: &mut Vec<&'a SummaryNode>, - joins: &mut Vec<&'a SummaryNode>, - ) { - if !seen.insert(node as *const _) { - return; - } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - aggregations.push(node); - visit(child, seen, aggregations, joins); - } - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, aggregations, joins), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - visit(child, seen, aggregations, joins); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - if matches!(&node.expr, SummaryExpr::SummaryJoin { .. }) { - joins.push(node); - } - visit(left, seen, aggregations, joins); - visit(right, seen, aggregations, joins); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - visit(summary_input, seen, aggregations, joins); - } - SummaryExpr::KeepPreAsap(_) => {} - } - } - let mut aggregations = Vec::new(); - let mut joins = Vec::new(); - visit(root, &mut HashSet::new(), &mut aggregations, &mut joins); - (aggregations, joins) -} - -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -#[cfg(test)] -struct SummaryOperationCounts { - state_builds: u64, - merges_per_read: u64, - subtracts_per_read: u64, - deletes_per_update: u64, - readouts_per_read: u64, - joins_per_read: u64, -} - -/// Low-level diagnostic for a homogeneous deployment. Final planner ranking -/// uses the per-node whole-DAG estimator above. Shared `Rc` nodes are visited -/// once; explicit delete frequency comes from deletion evidence. -#[cfg(test)] -pub(super) fn estimate_incremental_summary_maintenance( - root: &SummaryNode, - guarantee: &SummaryMaintenanceLifecycleGuarantee, - inputs: SummaryMaintenanceInputs, - cpu: SummaryOperationCpuEvidence, - scope: &ComparisonScope, -) -> Result { - estimate_incremental_summary_maintenance_with_join(root, guarantee, inputs, cpu, None, scope) -} -#[cfg(test)] -pub(super) fn estimate_incremental_summary_maintenance_with_join( - root: &SummaryNode, - guarantee: &SummaryMaintenanceLifecycleGuarantee, - inputs: SummaryMaintenanceInputs, - cpu: SummaryOperationCpuEvidence, - join: Option, - scope: &ComparisonScope, -) -> Result { - let inputs = inputs.validate()?; - let evaluation_count = scope.validate()?; - validate_guarantee(guarantee, scope.data_arrival)?; - let (bootstrap_input_rows, arriving_input_rows, active_ms) = - lifecycle_row_counts(inputs, guarantee, scope)?; - let counts = count_operations(root)?; - if counts.state_builds == 0 { - return Err(AnalyticalCostError::UnsupportedCandidate); - } - - let insert = required_cpu("insert_cpu_ops", cpu.insert_cpu_ops)?; - let merge = required_cpu_when(counts.merges_per_read, "merge_cpu_ops", cpu.merge_cpu_ops)?; - let subtract = required_cpu_when( - counts.subtracts_per_read, - "subtract_cpu_ops", - cpu.subtract_cpu_ops, - )?; - let delete = required_cpu_when( - counts.deletes_per_update, - "delete_cpu_ops", - cpu.delete_cpu_ops, - )?; - let delete_events = if counts.deletes_per_update == 0 { - 0_u64 - } else { - let rate = cpu - .delete_events_per_second - .filter(|rate| rate.is_finite() && *rate >= 0.0) - .ok_or(AnalyticalCostError::MissingOrStale( - "delete_events_per_second", - ))?; - let fanout = cpu - .delete_routing_fanout - .filter(|fanout| *fanout > 0) - .ok_or(AnalyticalCostError::MissingOrStale("delete_routing_fanout"))?; - let events = (rate * active_ms as f64 / 1_000.0).ceil(); - if !events.is_finite() || events > u64::MAX as f64 { - return Err(AnalyticalCostError::Overflow); - } - (events as u64) - .checked_mul(fanout) - .ok_or(AnalyticalCostError::Overflow)? - }; - let readout = required_cpu_when( - counts.readouts_per_read, - "readout_cpu_ops", - cpu.readout_cpu_ops, - )?; - let join_cpu = match (counts.joins_per_read, join.as_ref()) { - (0, _) => 0.0, - (_, Some(evidence)) - if evidence.cpu_ops_per_execution.is_finite() - && evidence.cpu_ops_per_execution > 0.0 - && evidence.working_memory_bytes > 0 => - { - evidence.cpu_ops_per_execution - } - (_, Some(evidence)) - if !evidence.cpu_ops_per_execution.is_finite() - || evidence.cpu_ops_per_execution <= 0.0 => - { - return Err(AnalyticalCostError::InvalidOperationCost( - "summary_join_cpu_ops_per_execution", - evidence.cpu_ops_per_execution, - )); - } - _ => return Err(AnalyticalCostError::MissingOrStale("summary_join")), - }; - - let build_inserts = bootstrap_input_rows - .checked_mul(inputs.bootstrap_window_count) - .ok_or(AnalyticalCostError::Overflow)? - .checked_mul(counts.state_builds) - .ok_or(AnalyticalCostError::Overflow)?; - let update_inserts = arriving_input_rows - .checked_mul(inputs.active_window_count) - .and_then(|n| n.checked_mul(counts.state_builds)) - .ok_or(AnalyticalCostError::Overflow)?; - let instances = inputs.physical_summary_count as f64; - let evaluations = evaluation_count as f64; - let cpu_ops = (build_inserts as f64 + update_inserts as f64) * insert - + evaluations * counts.merges_per_read as f64 * instances * merge - + evaluations * counts.subtracts_per_read as f64 * instances * subtract - + delete_events as f64 * counts.deletes_per_update as f64 * delete - + evaluations * counts.readouts_per_read as f64 * instances * readout - + evaluations * counts.joins_per_read as f64 * join_cpu; - if !cpu_ops.is_finite() { - return Err(AnalyticalCostError::Overflow); - } - - let state_instances = inputs - .active_window_count - .checked_add(inputs.retained_window_count) - .and_then(|n| n.checked_mul(inputs.physical_summary_count)) - .and_then(|n| n.checked_mul(counts.state_builds)) - .ok_or(AnalyticalCostError::Overflow)?; - let retained_bytes = state_instances - .checked_mul(inputs.state_bytes_per_summary) - .ok_or(AnalyticalCostError::Overflow)?; - // Merge/subtract may stream over persistent inputs but still needs one - // result state per physical instance. Persistent retained windows are - // already included above and are not loaded a second time. - let transient_bytes = if counts.merges_per_read > 0 || counts.subtracts_per_read > 0 { - inputs - .physical_summary_count - .checked_mul(inputs.state_bytes_per_summary) - .ok_or(AnalyticalCostError::Overflow)? - } else { - 0 - }; - let join_bytes = match (counts.joins_per_read, join.as_ref()) { - (0, _) => 0, - (_, Some(evidence)) => evidence.working_memory_bytes, - _ => return Err(AnalyticalCostError::MissingOrStale("summary_join")), - }; - let bootstrap_row_buffer = if inputs.initial_input_rows == 0 { - 0 - } else { - inputs - .initial_input_bytes - .div_ceil(inputs.initial_input_rows) - }; - Ok(ResourceEstimate::new( - cpu_ops, - retained_bytes - .checked_add(transient_bytes) - .and_then(|bytes| bytes.checked_add(join_bytes)) - .ok_or(AnalyticalCostError::Overflow)? - .max(bootstrap_row_buffer), - inputs.initial_source_scan_bytes, - )) -} - -pub(super) fn lifecycle_row_counts( - inputs: SummaryMaintenanceInputs, - guarantee: &SummaryMaintenanceLifecycleGuarantee, - scope: &ComparisonScope, -) -> Result<(u64, u64, u64), AnalyticalCostError> { - validate_arrival_rate(scope.data_arrival, inputs.ingestion_rate_per_second)?; - let horizon_end = scope - .planning_time - .0 - .checked_add(scope.horizon.0) - .ok_or(AnalyticalCostError::Overflow)?; - let (bootstrap_extra_ms, active_ms) = match guarantee.summary_maintenance_lifecycle { - SummaryMaintenanceLifecycle::Prepared { - activate_at, - retire_at, - } => { - if activate_at.0 >= retire_at.0 { - return Err(AnalyticalCostError::IncompatibleLifecycleGuarantee); - } - let covers_every_evaluation = evaluation_offsets_ms(scope)?.into_iter().all(|offset| { - scope - .planning_time - .0 - .checked_add(offset) - .is_some_and(|at| at >= activate_at.0 && at < retire_at.0) - }); - if !covers_every_evaluation { - return Err(AnalyticalCostError::IncompatibleLifecycleGuarantee); - } - let activation = activate_at.0.max(scope.planning_time.0).min(horizon_end); - let bootstrap_extra_ms = activation.saturating_sub(scope.planning_time.0); - let start = activation; - let end = retire_at.0.min(horizon_end); - (bootstrap_extra_ms, end.saturating_sub(start)) - } - SummaryMaintenanceLifecycle::Shared { .. } => (0, scope.horizon.0), - SummaryMaintenanceLifecycle::ContinuouslyMaintained => (0, scope.horizon.0), - SummaryMaintenanceLifecycle::Ephemeral => { - return Err(AnalyticalCostError::IncompatibleLifecycleGuarantee) - } - }; - let bootstrap_extra = inputs.ingestion_rate_per_second * bootstrap_extra_ms as f64 / 1000.0; - let updates = inputs.ingestion_rate_per_second * active_ms as f64 / 1000.0; - if !bootstrap_extra.is_finite() - || !updates.is_finite() - || bootstrap_extra > u64::MAX as f64 - || updates > u64::MAX as f64 - { - return Err(AnalyticalCostError::Overflow); - } - Ok(( - inputs - .initial_input_rows - .checked_add(bootstrap_extra.ceil() as u64) - .ok_or(AnalyticalCostError::Overflow)?, - updates.ceil() as u64, - active_ms, - )) -} - -fn validate_guarantee( - guarantee: &SummaryMaintenanceLifecycleGuarantee, - arrival: DataArrival, -) -> Result<(), AnalyticalCostError> { - if guarantee.output_representation != asap_types::post_asap::OutputRepresentation::SummaryState - || guarantee.summary_maintenance_mode - != maintenance_mode(&guarantee.summary_maintenance_lifecycle, arrival) - || guarantee.evaluation_schedule - != evaluation_schedule(&guarantee.summary_maintenance_lifecycle, arrival) - { - return Err(AnalyticalCostError::IncompatibleLifecycleGuarantee); - } - Ok(()) -} - -#[cfg(test)] -fn required_cpu(name: &'static str, value: Option) -> Result { - let value = value.ok_or(AnalyticalCostError::MissingOrStale(name))?; - if !value.is_finite() || value <= 0.0 { - return Err(AnalyticalCostError::InvalidOperationCost(name, value)); - } - Ok(value) -} - -pub(super) fn validated_operator_cpu( - name: &'static str, - value: f64, -) -> Result { - if !value.is_finite() || value <= 0.0 { - Err(AnalyticalCostError::InvalidOperationCost(name, value)) - } else { - Ok(value) - } -} - -fn validated_operator_executions( - name: &'static str, - evidence: &SummaryOperatorResourceEvidence, -) -> Result { - if evidence.executions_per_evaluation == 0 { - return Err(AnalyticalCostError::MissingOrZero(name)); - } - Ok(evidence.executions_per_evaluation) -} - -#[cfg(test)] -fn required_cpu_when( - count: u64, - name: &'static str, - value: Option, -) -> Result { - if count == 0 { - return Ok(0.0); - } - required_cpu(name, value) -} - -#[cfg(test)] -fn count_operations(root: &SummaryNode) -> Result { - fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - counts: &mut SummaryOperationCounts, - ) -> Result<(), AnalyticalCostError> { - if !seen.insert(node as *const SummaryNode) { - return Ok(()); - } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => { - counts.state_builds = counts - .state_builds - .checked_add(1) - .ok_or(AnalyticalCostError::Overflow)?; - visit(child, seen, counts)?; - } - SummaryExpr::SummaryMerge { children, .. } => { - if children.is_empty() { - return Err(AnalyticalCostError::InvalidPhysicalDAG( - "summary merge has no children", - )); - } - counts.merges_per_read = counts - .merges_per_read - .checked_add(children.len().saturating_sub(1) as u64) - .ok_or(AnalyticalCostError::Overflow)?; - for child in children { - visit(child, seen, counts)?; - } - } - SummaryExpr::SummarySubtract { left, right } => { - counts.subtracts_per_read = counts - .subtracts_per_read - .checked_add(1) - .ok_or(AnalyticalCostError::Overflow)?; - visit(left, seen, counts)?; - visit(right, seen, counts)?; - } - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - visit(lhs, seen, counts)?; - visit(rhs, seen, counts)?; - } - SummaryExpr::RelationalJoin { left, right, .. } => { - visit(left, seen, counts)?; - visit(right, seen, counts)?; - } - - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, counts)?, - SummaryExpr::SummaryDelete { summary_input, .. } => { - counts.deletes_per_update = counts - .deletes_per_update - .checked_add(1) - .ok_or(AnalyticalCostError::Overflow)?; - visit(summary_input, seen, counts)?; - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - counts.readouts_per_read = counts - .readouts_per_read - .checked_add(1) - .ok_or(AnalyticalCostError::Overflow)?; - visit(summary_input, seen, counts)?; - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - counts.joins_per_read = counts - .joins_per_read - .checked_add(1) - .ok_or(AnalyticalCostError::Overflow)?; - visit(outer, seen, counts)?; - visit(inner, seen, counts)?; - } - } - Ok(()) - } - - let mut counts = SummaryOperationCounts::default(); - visit(root, &mut HashSet::new(), &mut counts)?; - Ok(counts) -} diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs deleted file mode 100644 index f396e3cd8..000000000 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs +++ /dev/null @@ -1,381 +0,0 @@ -use super::*; - -/// Physical evidence that is not represented by [`DataWorkload`] for one -/// summary deployment. Window counts describe the -/// already-selected physical deployment; this layer does not define another -/// tumbling/sliding policy enum. -#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -pub struct SummaryPhysicalInputEvidence { - /// Logical bytes in the snapshot used to bootstrap the state. - pub initial_input_bytes: u64, - /// Source bytes read while bootstrapping. Arriving stream bytes are not a - /// disk scan and are therefore excluded. - pub initial_source_scan_bytes: u64, - /// Simultaneously open windows receiving each arriving item. - pub active_window_count: u64, - /// Window/state partitions receiving each bootstrap row. - pub bootstrap_window_count: u64, - /// Completed windows retained for query coverage. - pub retained_window_count: u64, - /// Independent state instances per window: one for shared - /// multi-subpopulation state, otherwise the resolved group count. - pub physical_summary_count: u64, - /// Resident bytes of one concrete state instance. - pub state_bytes_per_summary: u64, -} - -/// Workload-normalized inputs for summary construction and maintenance over one finite -/// comparison horizon. -#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -pub struct SummaryMaintenanceInputs { - pub initial_input_rows: u64, - pub initial_input_bytes: u64, - pub initial_source_scan_bytes: u64, - pub ingestion_rate_per_second: f64, - pub active_window_count: u64, - pub bootstrap_window_count: u64, - pub retained_window_count: u64, - pub physical_summary_count: u64, - pub state_bytes_per_summary: u64, -} - -impl SummaryMaintenanceInputs { - /// Resolve snapshot size, arriving rows, and reads from the canonical - /// workload. Positive fractional expected work rounds up conservatively. - /// - /// `AtRest` needs snapshot cardinality and implies zero arrivals; continuous - /// ingestion additionally requires fresh rate evidence. - /// `Mixed` fails closed because today's workload schema cannot distinguish - /// its at-rest backlog from its continuing-arrival cardinality. - pub fn from_workload( - physical: SummaryPhysicalInputEvidence, - data: &DataWorkload, - scope: &ComparisonScope, - ) -> Result { - let _ = scope.validate()?; - if scope.data_arrival != data.arrival { - return Err(AnalyticalCostError::ComparisonScopeMismatch("data arrival")); - } - let initial_input_rows = data - .input_cardinality - .value_at(scope.planning_time.0) - .copied() - .ok_or(AnalyticalCostError::MissingOrStale("input_cardinality"))?; - let ingestion_rate = match data.arrival { - DataArrival::AtRest => { - // A declared snapshot has no arrivals. Reject contradictory fresh - // evidence rather than silently pricing the wrong workload. - if let Some(rate) = data.ingestion_rate.value_at(scope.planning_time.0) { - validate_arrival_rate(data.arrival, rate.0)?; - } - 0.0 - } - DataArrival::ContinuouslyIngesting => { - data.ingestion_rate - .value_at(scope.planning_time.0) - .copied() - .ok_or(AnalyticalCostError::MissingOrStale("ingestion_rate"))? - .0 - } - arrival => return Err(AnalyticalCostError::UnsupportedDataArrival(arrival)), - }; - validate_arrival_rate(data.arrival, ingestion_rate)?; - Self { - initial_input_rows, - initial_input_bytes: physical.initial_input_bytes, - initial_source_scan_bytes: physical.initial_source_scan_bytes, - ingestion_rate_per_second: ingestion_rate, - active_window_count: physical.active_window_count, - bootstrap_window_count: physical.bootstrap_window_count, - retained_window_count: physical.retained_window_count, - physical_summary_count: physical.physical_summary_count, - state_bytes_per_summary: physical.state_bytes_per_summary, - } - .validate() - } - - pub fn validate(self) -> Result { - for (name, value) in [ - ("active_window_count", self.active_window_count), - ("bootstrap_window_count", self.bootstrap_window_count), - ("physical_summary_count", self.physical_summary_count), - ("state_bytes_per_summary", self.state_bytes_per_summary), - ] { - if value == 0 { - return Err(AnalyticalCostError::MissingOrZero(name)); - } - } - if (self.initial_input_rows == 0) != (self.initial_input_bytes == 0) - || (self.initial_input_rows == 0 && self.initial_source_scan_bytes != 0) - { - return Err(AnalyticalCostError::InconsistentBootstrapEvidence); - } - if !self.ingestion_rate_per_second.is_finite() || self.ingestion_rate_per_second < 0.0 { - return Err(AnalyticalCostError::InvalidIngestionRate( - self.ingestion_rate_per_second, - )); - } - Ok(self) - } -} - -/// CPU operations for one concrete state operation on one state instance. -/// Missing evidence is legal only when the selected summary DAG does not use -/// that operation. -#[cfg(test)] -#[derive(Debug, Clone, Copy, Default, PartialEq, Serialize, Deserialize)] -pub struct SummaryOperationCpuEvidence { - pub insert_cpu_ops: Option, - pub merge_cpu_ops: Option, - pub subtract_cpu_ops: Option, - pub delete_cpu_ops: Option, - /// Expirations/retractions routed to this DAG per second. Required only - /// when an explicit `SummaryDelete` is present. - pub delete_events_per_second: Option, - /// Concrete state instances touched by one delete event. - pub delete_routing_fanout: Option, - pub readout_cpu_ops: Option, -} - -/// Physical evidence for one `SummaryJoin` implementation. Total work, -/// cardinality, and memory cannot be inferred from the logical join key alone. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct SummaryJoinEvidence { - pub physical_id: String, - pub inputs: Vec, - pub output: EdgeStatistics, - /// Total build, probe, match-production, and output CPU for one complete - /// execution of the selected physical join algorithm. - pub cpu_ops_per_execution: f64, - pub working_memory_bytes: u64, - pub output_buffer_bytes: u64, - pub executions_per_evaluation: u64, - pub io_bytes_per_execution: Option, -} - -#[derive(Debug, Clone, PartialEq)] -pub struct SummaryAggregateEvidence { - pub physical_id: String, - pub input: EdgeStatistics, - pub output: EdgeStatistics, - /// Index into `ComparisonScope.sources` when this state bootstraps directly - /// from storage. `None` means its input is an already-materialized child - /// edge and therefore has no additional source read. - pub source_coverage_index: Option, - /// Provider-owned identity of the physical bootstrap read. Equal source - /// coverage alone does not prove two independent builds share I/O. - pub bootstrap_read_identity: String, - pub inputs: SummaryMaintenanceInputs, - /// CPU operations to insert one routed row into one state instance. - pub insert_cpu_ops: f64, -} - -#[derive(Debug, Clone, PartialEq)] -pub struct SummaryOperatorResourceEvidence { - pub physical_id: String, - pub inputs: Vec, - pub output: EdgeStatistics, - pub cpu_ops: f64, - pub working_memory_bytes: u64, - pub output_buffer_bytes: u64, - /// Executions of this physical operator for one query evaluation. - /// This is provider evidence, not inferred from a descendant state. - pub executions_per_evaluation: u64, - pub io_bytes_per_execution: Option, -} - -/// Evidence is structured by logical summary operation so delete-only facts -/// cannot be attached to merge, subtract, or readout nodes. -#[derive(Debug, Clone, PartialEq)] -pub enum SummaryOperatorEvidence { - /// Exact query-time arithmetic over two independently realized operands. - Binary(SummaryOperatorResourceEvidence), - /// Query-time or maintenance-time plain-value work. For `Sort`/`Limit`, - /// providers report the actual comparison/heap work and working set here; - /// the estimator charges it at query multiplicity. - ValueOperation(SummaryOperatorResourceEvidence), - Merge(SummaryOperatorResourceEvidence), - Subtract(SummaryOperatorResourceEvidence), - Delete { - resource: SummaryOperatorResourceEvidence, - events_per_second: f64, - routing_fanout: u64, - }, - Readout(SummaryOperatorResourceEvidence), -} - -impl SummaryOperatorEvidence { - pub(super) fn resource(&self) -> &SummaryOperatorResourceEvidence { - match self { - Self::Binary(resource) - | Self::ValueOperation(resource) - | Self::Merge(resource) - | Self::Subtract(resource) - | Self::Delete { resource, .. } - | Self::Readout(resource) => resource, - } - } - - #[cfg(test)] - pub(super) fn resource_mut(&mut self) -> &mut SummaryOperatorResourceEvidence { - match self { - Self::Binary(resource) - | Self::ValueOperation(resource) - | Self::Merge(resource) - | Self::Subtract(resource) - | Self::Delete { resource, .. } - | Self::Readout(resource) => resource, - } - } -} - -/// Non-aggregation work for a retained pre-ASAP sub-DAG over the comparison -/// horizon. Bootstrap/source I/O belongs exclusively to the owning aggregate, -/// and summary insertion belongs exclusively to its insert evidence. -#[derive(Debug, Clone, PartialEq)] -pub struct RetainedSubDAGEvidence { - pub physical_id: String, - /// Logical output edge consumed by the parent summary operator. - pub output: EdgeStatistics, - pub preprocessing_cpu_ops_over_horizon: f64, - /// Execution workspace, excluding the separately declared output buffer. - pub working_memory_bytes: u64, - pub output_buffer_bytes: u64, -} - -/// Physical evidence bound to the selected DAG's `Rc` identity. A copied, -/// structurally equal node is not silently treated as the same deployment. -#[derive(Debug, Clone, Default)] -pub struct SummaryNodeEvidence { - pub(super) aggregations: HashMap<*const SummaryNode, SummaryAggregateEvidence>, - pub(super) joins: HashMap<*const SummaryNode, SummaryJoinEvidence>, - pub(super) operations: HashMap<*const SummaryNode, SummaryOperatorEvidence>, - pub(super) operation_state_owners: HashMap<*const SummaryNode, *const SummaryNode>, - pub(super) retained_queries: HashMap<*const SummaryNode, RetainedSubDAGEvidence>, -} - -impl SummaryNodeEvidence { - pub fn insert_aggregation( - &mut self, - node: &Rc, - evidence: SummaryAggregateEvidence, - ) { - self.aggregations.insert(Rc::as_ptr(node), evidence); - } - - pub fn insert_join(&mut self, node: &Rc, evidence: SummaryJoinEvidence) { - self.joins.insert(Rc::as_ptr(node), evidence); - } - - pub fn insert_operation(&mut self, node: &Rc, evidence: SummaryOperatorEvidence) { - self.operations.insert(Rc::as_ptr(node), evidence); - } - - /// Bind a stateful operation (currently `SummaryDelete`) to the exact - /// aggregation deployment whose active interval it follows. - pub fn insert_state_operation( - &mut self, - node: &Rc, - state: &Rc, - evidence: SummaryOperatorEvidence, - ) { - self.operations.insert(Rc::as_ptr(node), evidence); - self.operation_state_owners - .insert(Rc::as_ptr(node), Rc::as_ptr(state)); - } - - pub fn insert_retained_query( - &mut self, - node: &Rc, - evidence: RetainedSubDAGEvidence, - ) { - self.retained_queries.insert(Rc::as_ptr(node), evidence); - } - - pub(super) fn aggregation(&self, node: &SummaryNode) -> Option { - self.aggregations.get(&(node as *const _)).cloned() - } -} - -pub(super) fn summary_operation_evidence<'a>( - node: &SummaryNode, - evidence: &'a SummaryNodeEvidence, -) -> Result<&'a SummaryOperatorEvidence, AnalyticalCostError> { - let operation = evidence - .operations - .get(&(node as *const _)) - .ok_or(AnalyticalCostError::MissingOrStale("summary operation"))?; - let matches = matches!( - (&node.expr, operation), - ( - SummaryExpr::BinaryOp { .. }, - SummaryOperatorEvidence::Binary(_) - ) | ( - SummaryExpr::ValueOperation { .. }, - SummaryOperatorEvidence::ValueOperation(_) - ) | ( - SummaryExpr::SummaryMerge { .. }, - SummaryOperatorEvidence::Merge(_) - ) | ( - SummaryExpr::SummarySubtract { .. }, - SummaryOperatorEvidence::Subtract(_) - ) | ( - SummaryExpr::SummaryDelete { .. }, - SummaryOperatorEvidence::Delete { .. } - ) | ( - SummaryExpr::SummaryEstimate { .. }, - SummaryOperatorEvidence::Readout(_) - ) - ); - if matches { - Ok(operation) - } else { - Err(AnalyticalCostError::InconsistentOperatorStatistics( - "summary operation evidence kind does not match SummaryExpr", - )) - } -} - -/// Evidence for recomputing the raw target over the full comparison horizon. -/// Planning-time dimensions describe the initial snapshot. Each scheduled -/// evaluation adds arrivals since planning time; `physical_dag` is therefore -/// a once-counted DAG whose edge statistics already aggregate all evaluations. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct RawInputEvidence { - pub planning_time_input_rows: u64, - pub planning_time_input_bytes: u64, - pub planning_time_source_scan_bytes: u64, - /// Decoded logical bytes added to operator edges by one arriving row. - pub arriving_logical_row_bytes: u64, - /// Physical storage bytes read for one arriving row. Kept separate from - /// logical width so compression and encoding are not silently conflated. - pub arriving_source_row_bytes: u64, - pub ingestion_rate_per_second: f64, - pub physical_dag: EvidenceBackedPhysicalDAG, -} - -/// One complete provider-enumerated physical implementation of the selected -/// summary DAG. The identifier is stable provenance; concrete -/// framework selection is performed by ranking these complete alternatives. -#[derive(Debug, Clone)] -pub struct SummaryPhysicalPlanAlternative { - pub physical_plan_id: String, - pub node_evidence: SummaryNodeEvidence, -} - -/// Apply arrival semantics to both workload-derived and directly bound evidence. -pub(super) fn validate_arrival_rate( - arrival: DataArrival, - rate: f64, -) -> Result<(), AnalyticalCostError> { - if !rate.is_finite() || rate < 0.0 { - return Err(AnalyticalCostError::InvalidIngestionRate(rate)); - } - match arrival { - DataArrival::AtRest if rate != 0.0 => Err(AnalyticalCostError::ComparisonScopeMismatch( - "at-rest ingestion rate", - )), - DataArrival::AtRest | DataArrival::ContinuouslyIngesting => Ok(()), - other => Err(AnalyticalCostError::UnsupportedDataArrival(other)), - } -} diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs deleted file mode 100644 index bc1b7ddaf..000000000 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs +++ /dev/null @@ -1,52 +0,0 @@ -//! Analytical resource cost for at-rest and incrementally maintained summary deployments. -//! -//! The canonical workload and lifecycle types own deployment semantics. This -//! module only adds physical evidence absent from those schemas: state size, -//! window counts, and per-operation CPU measurements or complexity estimates. - -use std::collections::{HashMap, HashSet}; -use std::rc::Rc; - -use asap_types::post_asap::{ - BoundExpr, ErrorMetric, ExactKind, FieldDataType, GuaranteeSource, ProbabilityExpr, - ResultGuarantee, SketchAlgorithm, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, -}; -use asap_types::pre_asap::{ - agg_intent::AggIntent, CompareOpKind, InfoMatcher, Predicate, QueryExpr, Source, -}; -use asap_types::types::AccuracyTarget; -use asap_types::workload::{DataArrival, DataWorkload, QueryRecurrence, RepeatedDemand}; -use serde::{Deserialize, Serialize}; - -use crate::accuracy::{AccuracyModel, DefaultAccuracyModel}; -use crate::analytical_cost::ExecutionMultiplicity; -#[cfg(test)] -use crate::analytical_cost::PhysicalNodeEvidence; -use crate::analytical_cost::{ - estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDAG, PhysicalDAGNode, - PhysicalOperator, ResourceCalibration, ResourceEstimate, -}; -use crate::cost_model::{ - CompleteSummaryCandidateEstimate, Cost, CostModel, CostedSummaryDeployment, DefaultCostModel, -}; -#[cfg(test)] -use crate::physical_operator_statistics::UnaryEdgeStatistics; -use crate::physical_operator_statistics::{ComparisonScope, EdgeStatistics, OperatorStatistics}; -use crate::recurrence::CostRate; -use crate::replacement::{Replacement, ReplacementSubDAG, TargetSubDAG}; -use crate::summary_maintenance_lifecycle::{ - evaluation_schedule, maintenance_mode, SummaryMaintenanceCapabilities, - SummaryMaintenanceLifecycleCostInputs, -}; - -pub const SUMMARY_MAINTENANCE_COST_MODEL_VERSION: &str = "summary-maintenance-resource-v2"; - -mod estimator; -mod evidence; -mod model; -mod window; - -pub use evidence::*; -pub use model::*; -pub use window::*; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs deleted file mode 100644 index 331b7e1a5..000000000 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs +++ /dev/null @@ -1,3651 +0,0 @@ -use super::*; -/// Adapter that supplies the existing lifecycle planner with analytical -/// summary costs across at-rest and continuously ingesting workloads. The -/// planner's existing lifecycle enums and legality checks remain authoritative. -#[derive(Debug, Clone)] -pub struct SummaryMaintenanceCostModel { - pub node_evidence: SummaryNodeEvidence, - pub calibration: ResourceCalibration, - pub capabilities: SummaryMaintenanceCapabilities, - target_comparisons: HashMap<*const QueryExpr, SummaryTargetComparison>, - candidate_comparisons: HashMap, - physical_plan_alternatives: - HashMap>, - window_framework_candidates: - HashMap>, -} - -type CandidateComparisonKey = (*const QueryExpr, *const SummaryNode); - -#[derive(Debug, Clone)] -struct BoundCandidateIdentity { - _target: Rc, - _root: Rc, -} - -#[derive(Debug, Clone)] -struct SummaryTargetComparison { - _target: Rc, - scope: ComparisonScope, - raw: RawInputEvidence, -} - -pub(super) type LogicalSourceSelection = (Source, Vec, Vec); - -pub(super) fn deduplicate_source_selections( - values: Vec, -) -> Vec { - values.into_iter().fold(Vec::new(), |mut unique, value| { - if !unique.contains(&value) { - unique.push(value); - } - unique - }) -} - -fn info_source(selector: &[InfoMatcher]) -> Result { - let mut metric: Option<&str> = None; - for matcher in selector - .iter() - .filter(|matcher| matcher.label == "__name__") - { - if matcher.op != CompareOpKind::Eq || metric.is_some_and(|value| value != matcher.value) { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } - metric = Some(&matcher.value); - } - Ok(Source::TimeSeries { - metric: metric.unwrap_or("target_info").into(), - }) -} - -pub(super) fn query_source_selections( - query: &QueryExpr, - out: &mut Vec, -) -> Result<(), AnalyticalCostError> { - use QueryExpr::*; - match query { - Scan { - source, predicates, .. - } => out.push((source.clone(), predicates.clone(), vec![])), - PromqlVectorFromScalar(child) | PromqlScalarFromVector(child) => { - query_source_selections(child, out)? - } - PromqlInfoEnrich { selector, child } => { - query_source_selections(child, out)?; - out.push((info_source(selector)?, vec![], selector.clone())); - } - PromqlRelabel { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | PromqlSeriesSample { child, .. } - | Sort { child, .. } - | Limit { child, .. } => query_source_selections(child, out)?, - Concat { children, .. } => { - for child in children { - query_source_selections(child, out)?; - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - query_source_selections(left, out)?; - query_source_selections(right, out)?; - } - BinaryOp { lhs, rhs, .. } => { - query_source_selections(lhs, out)?; - query_source_selections(rhs, out)?; - } - PromqlScalarBridge(_) - | EvalTimestamp - | CurrentTimestamp - | Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} - } - Ok(()) -} - -fn validate_query_scope( - target: &QueryExpr, - scope: &ComparisonScope, -) -> Result<(), AnalyticalCostError> { - let mut actual = Vec::new(); - query_source_selections(target, &mut actual)?; - let actual = deduplicate_source_selections(actual); - let mut declared: Vec<_> = scope - .sources - .iter() - .map(|coverage| { - ( - coverage.source.clone(), - coverage.predicates.clone(), - coverage.info_matchers.clone(), - ) - }) - .collect(); - for selection in actual { - let Some(index) = declared.iter().position(|value| value == &selection) else { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "raw target source lineage", - )); - }; - declared.swap_remove(index); - } - if !declared.is_empty() { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "raw target source lineage", - )); - } - Ok(()) -} - -fn validate_physical_scope_coverage( - physical: &EvidenceBackedPhysicalDAG, - scope: &ComparisonScope, -) -> Result<(), AnalyticalCostError> { - let nodes = reachable_physical_nodes(physical)?; - let mut covered = HashSet::new(); - for node in nodes - .into_iter() - .filter(|node| node.operator == PhysicalOperator::Scan) - { - let coverage = node - .source_coverage - .as_ref() - .ok_or_else(|| AnalyticalCostError::MissingScanSourceCoverage(node.id.clone()))?; - let Some(index) = scope.sources.iter().position(|value| value == coverage) else { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "physical source coverage", - )); - }; - covered.insert(index); - } - if covered.len() != scope.sources.len() { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "physical source coverage", - )); - } - Ok(()) -} - -fn reachable_physical_nodes( - physical: &EvidenceBackedPhysicalDAG, -) -> Result, AnalyticalCostError> { - let by_id: HashMap<_, _> = physical - .nodes - .iter() - .map(|node| (node.id.as_str(), node)) - .collect(); - if by_id.len() != physical.nodes.len() { - return Err(AnalyticalCostError::InvalidPhysicalDAG("duplicate node id")); - } - fn visit<'a>( - id: &'a str, - by_id: &HashMap<&'a str, &'a PhysicalDAGNode>, - visiting: &mut HashSet<&'a str>, - visited: &mut HashSet<&'a str>, - nodes: &mut Vec<&'a PhysicalDAGNode>, - ) -> Result<(), AnalyticalCostError> { - if visited.contains(id) { - return Ok(()); - } - if !visiting.insert(id) { - return Err(AnalyticalCostError::InvalidPhysicalDAG("cycle")); - } - let node = by_id - .get(id) - .copied() - .ok_or(AnalyticalCostError::InvalidPhysicalDAG("missing node"))?; - for child in &node.children { - visit(child, by_id, visiting, visited, nodes)?; - } - visiting.remove(id); - visited.insert(id); - nodes.push(node); - Ok(()) - } - let mut nodes = Vec::new(); - visit( - physical.root.as_str(), - &by_id, - &mut HashSet::new(), - &mut HashSet::new(), - &mut nodes, - )?; - Ok(nodes) -} - -fn validate_raw_snapshot_dimensions( - raw: &RawInputEvidence, - scope: &ComparisonScope, -) -> Result<(), AnalyticalCostError> { - validate_arrival_rate(scope.data_arrival, raw.ingestion_rate_per_second)?; - if scope.sources.len() != 1 { - return Err(AnalyticalCostError::MissingComparisonScope( - "single-source raw evolution", - )); - } - if !raw.ingestion_rate_per_second.is_finite() || raw.ingestion_rate_per_second < 0.0 { - return Err(AnalyticalCostError::InvalidIngestionRate( - raw.ingestion_rate_per_second, - )); - } - let bootstrap_is_consistent = if raw.planning_time_input_rows == 0 { - raw.planning_time_input_bytes == 0 && raw.planning_time_source_scan_bytes == 0 - } else { - raw.planning_time_input_bytes > 0 && raw.planning_time_source_scan_bytes > 0 - }; - if !bootstrap_is_consistent - || (raw.ingestion_rate_per_second > 0.0 - && (raw.arriving_logical_row_bytes == 0 || raw.arriving_source_row_bytes == 0)) - { - return Err(AnalyticalCostError::InconsistentBootstrapEvidence); - } - let mut rows = 0_u64; - let mut bytes = 0_u64; - let mut scan = 0_u64; - for offset in evaluation_offsets_ms(scope)? { - let arrivals = (raw.ingestion_rate_per_second * offset as f64 / 1_000.0).ceil(); - if !arrivals.is_finite() || arrivals < 0.0 || arrivals > u64::MAX as f64 { - return Err(AnalyticalCostError::Overflow); - } - let arrivals = arrivals as u64; - rows = rows - .checked_add(raw.planning_time_input_rows) - .and_then(|value| value.checked_add(arrivals)) - .ok_or(AnalyticalCostError::Overflow)?; - bytes = bytes - .checked_add(raw.planning_time_input_bytes) - .and_then(|value| { - arrivals - .checked_mul(raw.arriving_logical_row_bytes) - .and_then(|arriving| value.checked_add(arriving)) - }) - .ok_or(AnalyticalCostError::Overflow)?; - scan = scan - .checked_add(raw.planning_time_source_scan_bytes) - .and_then(|value| { - arrivals - .checked_mul(raw.arriving_source_row_bytes) - .and_then(|arriving| value.checked_add(arriving)) - }) - .ok_or(AnalyticalCostError::Overflow)?; - } - let reachable = reachable_physical_nodes(&raw.physical_dag)?; - if reachable - .iter() - .any(|node| node.execution != ExecutionMultiplicity::Once) - { - return Err(AnalyticalCostError::InvalidPhysicalDAG( - "streaming raw horizon evidence must use once-counted aggregate statistics", - )); - } - let expected = EdgeStatistics { rows, bytes }; - let mut scan_count = 0; - for scan_node in reachable - .into_iter() - .filter(|node| node.operator == PhysicalOperator::Scan) - { - scan_count += 1; - let evidence = raw - .physical_dag - .evidence - .get(&scan_node.id) - .ok_or_else(|| AnalyticalCostError::MissingOperatorStatistics(scan_node.id.clone()))?; - let statistics = &evidence.statistics; - let OperatorStatistics::Scan { - edges, - source_read_bytes, - } = statistics - else { - return Err(AnalyticalCostError::InvalidOperatorStatistics { - node: scan_node.id.clone(), - reason: "raw scan evidence uses the wrong statistics variant", - }); - }; - if edges.input != expected || edges.output != expected || *source_read_bytes != scan { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "raw source evolution", - )); - } - } - if scan_count == 0 { - return Err(AnalyticalCostError::MissingComparisonScope("raw scan")); - } - Ok(()) -} - -pub(super) fn ephemeral_rows_over_horizon( - inputs: SummaryMaintenanceInputs, - scope: &ComparisonScope, -) -> Result { - evaluation_offsets_ms(scope)? - .into_iter() - .try_fold(0_u64, |total, offset| { - let arrivals = (inputs.ingestion_rate_per_second * offset as f64 / 1_000.0).ceil(); - if !arrivals.is_finite() || arrivals < 0.0 || arrivals > u64::MAX as f64 { - return Err(AnalyticalCostError::Overflow); - } - total - .checked_add(inputs.initial_input_rows) - .and_then(|value| value.checked_add(arrivals as u64)) - .ok_or(AnalyticalCostError::Overflow) - }) -} - -pub(super) fn ephemeral_scan_bytes_over_horizon( - inputs: SummaryMaintenanceInputs, - raw: &RawInputEvidence, - scope: &ComparisonScope, -) -> Result { - if inputs.initial_source_scan_bytes == 0 { - return Ok(0); - } - evaluation_offsets_ms(scope)? - .into_iter() - .try_fold(0_u64, |total, offset| { - let arrivals = (inputs.ingestion_rate_per_second * offset as f64 / 1_000.0).ceil(); - if !arrivals.is_finite() || arrivals < 0.0 || arrivals > u64::MAX as f64 { - return Err(AnalyticalCostError::Overflow); - } - total - .checked_add(inputs.initial_source_scan_bytes) - .and_then(|value| { - (arrivals as u64) - .checked_mul(raw.arriving_source_row_bytes) - .and_then(|arriving| value.checked_add(arriving)) - }) - .ok_or(AnalyticalCostError::Overflow) - }) -} - -pub(super) fn evaluation_offsets_ms( - scope: &ComparisonScope, -) -> Result, AnalyticalCostError> { - let count = scope.validate()?; - match &scope.recurrence { - QueryRecurrence::OneTime { - invocations, - execute_at, - } => Ok(vec![ - execute_at.map_or(0, |at| at - .0 - .saturating_sub(scope.planning_time.0)); - *invocations as usize - ]), - QueryRecurrence::Repeated(RepeatedDemand::FixedInterval(interval)) - | QueryRecurrence::Repeated(RepeatedDemand::FixedIntervalAt { interval, .. }) => { - Ok((1..=count).map(|n| n * u64::from(interval.0)).collect()) - } - QueryRecurrence::Repeated(RepeatedDemand::Scheduled(schedule)) => Ok(schedule - .iter() - .filter(|at| { - at.0 >= scope.planning_time.0 - && at.0 <= scope.planning_time.0.saturating_add(scope.horizon.0) - }) - .map(|at| at.0 - scope.planning_time.0) - .collect()), - QueryRecurrence::Repeated(RepeatedDemand::EstimatedRate(_)) => Ok((1..=count) - .map(|n| scope.horizon.0.saturating_mul(n) / count) - .collect()), - QueryRecurrence::Unknown => Err(AnalyticalCostError::InvalidRecurrence), - } -} - -impl SummaryMaintenanceCostModel { - pub fn new( - calibration: ResourceCalibration, - capabilities: SummaryMaintenanceCapabilities, - ) -> Self { - Self { - node_evidence: SummaryNodeEvidence::default(), - calibration, - capabilities, - target_comparisons: HashMap::new(), - candidate_comparisons: HashMap::new(), - physical_plan_alternatives: HashMap::new(), - window_framework_candidates: HashMap::new(), - } - } - - /// Bind one candidate and its raw baseline to the same target-specific - /// comparison context. Rebinding a target to different evidence is - /// rejected rather than silently replacing the canonical context. - pub fn bind_candidate_comparison( - &mut self, - target: &Rc, - root: &Rc, - scope: ComparisonScope, - raw: RawInputEvidence, - ) -> Result<(), AnalyticalCostError> { - scope.validate()?; - validate_query_scope(target, &scope)?; - validate_physical_scope_coverage(&raw.physical_dag, &scope)?; - validate_raw_snapshot_dimensions(&raw, &scope)?; - estimate_physical_dag( - &raw.physical_dag.nodes, - &raw.physical_dag.root, - &scope, - &raw.physical_dag, - )?; - let target_ptr = Rc::as_ptr(target); - if let Some(existing) = self.target_comparisons.get(&target_ptr) { - if existing.scope != scope || existing.raw != raw { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "target comparison", - )); - } - } - // Commit only after every validation above succeeds. Shared nodes do - // not carry one owning target; context identity is `(target, root)`. - self.target_comparisons - .entry(target_ptr) - .or_insert(SummaryTargetComparison { - _target: Rc::clone(target), - scope, - raw, - }); - self.candidate_comparisons.insert( - (target_ptr, Rc::as_ptr(root)), - BoundCandidateIdentity { - _target: Rc::clone(target), - _root: Rc::clone(root), - }, - ); - Ok(()) - } - - /// Add one complete physical implementation for an already-bound logical - /// candidate. Duplicate or empty provider identities are rejected. - pub fn bind_physical_plan_alternative( - &mut self, - target: &Rc, - root: &Rc, - alternative: SummaryPhysicalPlanAlternative, - ) -> Result<(), AnalyticalCostError> { - let key = (Rc::as_ptr(target), Rc::as_ptr(root)); - if !self.candidate_comparisons.contains_key(&key) { - return Err(AnalyticalCostError::MissingOrStale( - "candidate comparison binding", - )); - } - if alternative.physical_plan_id.trim().is_empty() { - return Err(AnalyticalCostError::MissingOrZero("physical_plan_id")); - } - if self.window_framework_candidates.contains_key(&key) { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "physical alternative binding mode", - )); - } - let alternatives = self.physical_plan_alternatives.entry(key).or_default(); - if alternatives - .iter() - .any(|existing| existing.physical_plan_id == alternative.physical_plan_id) - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "physical plan identity", - )); - } - alternatives.push(alternative); - Ok(()) - } - - /// Add one complete abstract window assignment to Planner candidate search. - /// - /// The provider may bind multiple executor-feasible implementations for - /// the same framework assignment; their stable identities and complete - /// evidence keep the implementations distinct during ranking. - pub fn bind_window_framework_candidate( - &mut self, - target: &Rc, - root: &Rc, - candidate: SummaryWindowFrameworkCandidate, - ) -> Result<(), AnalyticalCostError> { - let key = (Rc::as_ptr(target), Rc::as_ptr(root)); - if !self.candidate_comparisons.contains_key(&key) { - return Err(AnalyticalCostError::MissingOrStale( - "candidate comparison binding", - )); - } - if candidate.physical_plan_id.trim().is_empty() { - return Err(AnalyticalCostError::MissingOrZero("physical_plan_id")); - } - if self.physical_plan_alternatives.contains_key(&key) { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "physical alternative binding mode", - )); - } - if candidate.assignments.is_empty() { - return Err(AnalyticalCostError::MissingOrZero( - "window framework assignments", - )); - } - let mut assigned = HashSet::new(); - if candidate - .assignments - .iter() - .any(|assignment| !assigned.insert(Rc::as_ptr(&assignment.summary))) - { - return Err(AnalyticalCostError::MissingOrZero( - "unique window framework assignments", - )); - } - if assigned != summary_aggregation_identities(root) { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "window framework assignments", - )); - } - let candidates = self.window_framework_candidates.entry(key).or_default(); - if candidates - .iter() - .any(|existing| existing.physical_plan_id == candidate.physical_plan_id) - { - return Err(AnalyticalCostError::ComparisonScopeMismatch( - "window framework candidate", - )); - } - candidates.push(candidate); - Ok(()) - } - - fn comparison_context( - &self, - root: &SummaryNode, - target: Option<&QueryExpr>, - horizon: Option, - expected_reads: Option, - ) -> Option<(CandidateComparisonKey, &SummaryTargetComparison)> { - let root_ptr = root as *const _; - let target_ptr = match target { - Some(target) => target as *const _, - None => { - let mut targets = self - .candidate_comparisons - .keys() - .filter_map(|(target, candidate)| (*candidate == root_ptr).then_some(*target)); - let only = targets.next()?; - if targets.next().is_some() { - return None; - } - only - } - }; - let key = (target_ptr, root_ptr); - if !self.candidate_comparisons.contains_key(&key) { - return None; - } - let comparison = self.target_comparisons.get(&target_ptr)?; - if horizon.map(|value| value.0 * 1_000.0) != Some(comparison.scope.horizon.0 as f64) - || expected_reads != Some(comparison.scope.validate().ok()? as f64) - { - return None; - } - Some((key, comparison)) - } - - fn complete_cost_with_evidence( - &self, - root: &SummaryNode, - deployments: &[CostedSummaryDeployment<'_>], - comparison: &SummaryTargetComparison, - evidence: &SummaryNodeEvidence, - window_frameworks: &[Option], - ) -> Option { - self.calibrated( - estimate_heterogeneous_summary( - root, - deployments, - evidence, - &comparison.scope, - &comparison.raw, - window_frameworks, - ) - .ok()?, - ) - } - - fn canonical_inputs(&self, summary: &SummaryNode) -> Option { - let evidence = self.node_evidence.aggregation(summary)?; - evidence.inputs.validate().ok()?; - Some(evidence) - } - - fn calibrated(&self, estimate: ResourceEstimate) -> Option { - estimate.calibrated_cost(&self.calibration).ok().map(Cost) - } - - fn lifecycle_inputs( - &self, - summary: &SummaryNode, - horizon: Option, - ) -> Option { - let evidence = self.canonical_inputs(summary)?; - let inputs = evidence.inputs; - let insert = validated_operator_cpu("insert_cpu_ops", evidence.insert_cpu_ops).ok()?; - let build = self.calibrated(ResourceEstimate::new( - inputs.initial_input_rows as f64 * inputs.bootstrap_window_count as f64 * insert, - 0, - inputs.initial_source_scan_bytes, - ))?; - let maintenance = self.calibrated(ResourceEstimate::new( - inputs.active_window_count as f64 * insert, - 0, - 0, - ))?; - let retained = inputs - .active_window_count - .checked_add(inputs.retained_window_count)? - .checked_mul(inputs.physical_summary_count)? - .checked_mul(inputs.state_bytes_per_summary)?; - let retention_total = self.calibrated(ResourceEstimate::new(0.0, retained, 0))?; - let horizon_seconds = horizon.filter(|value| value.0 > 0.0)?.0; - Some(SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(build), - maintenance_cost_per_update: Some(maintenance), - // Readout is a separate physical operator in the complete DAG. - // A state-only candidate therefore does not fabricate readout - // evidence merely to keep a lifecycle alternative selectable. - summary_read_cost: Some(Cost::ZERO), - retention_cost_rate: Some(CostRate(retention_total.0 / horizon_seconds)), - // Releasing memory has no modeled CPU or I/O. This is not an - // implicit expiration/rebuild policy; those require an explicit - // SummaryDelete or future authoritative lifecycle evidence. - retirement_cost: Some(Cost::ZERO), - }) - } -} - -impl CostModel for SummaryMaintenanceCostModel { - fn candidate_cost( - &self, - candidate: &ReplacementSubDAG, - _target: &TargetSubDAG<'_>, - ) -> Option { - match &candidate.replacement { - Replacement::ExactComposition(_) => None, - // Lifecycle selection supplies a complete override. If it cannot, - // the candidate remains unavailable rather than receiving this - // trait's structural fallback. - Replacement::Summary(_) => None, - Replacement::Rewrite(_) => None, - } - } - - fn rank_candidates( - &self, - intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - DefaultCostModel.rank_candidates(intent, candidates) - } - - fn estimate_cost(&self, _candidate: &ReplacementSubDAG, _target: &TargetSubDAG<'_>) -> f64 { - f64::INFINITY - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - SummaryMaintenanceLifecycleCostInputs::default() - } - - fn summary_maintenance_lifecycle_cost_inputs_for_horizon( - &self, - summary: &SummaryNode, - horizon: Option, - ) -> SummaryMaintenanceLifecycleCostInputs { - self.lifecycle_inputs(summary, horizon).unwrap_or_default() - } - - fn summary_maintenance_capabilities( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - self.capabilities - } - - fn complete_summary_candidate_cost( - &self, - root: &SummaryNode, - target: Option<&QueryExpr>, - deployments: &[CostedSummaryDeployment<'_>], - horizon: Option, - expected_reads: Option, - required_accuracy: &[AccuracyTarget], - ) -> Option { - self.complete_summary_candidate_estimate( - root, - target, - deployments, - horizon, - expected_reads, - required_accuracy, - ) - .map(|estimate| estimate.cost) - } - - fn complete_summary_candidate_estimate( - &self, - root: &SummaryNode, - target: Option<&QueryExpr>, - deployments: &[CostedSummaryDeployment<'_>], - horizon: Option, - expected_reads: Option, - required_accuracy: &[AccuracyTarget], - ) -> Option { - let (key, comparison) = self.comparison_context(root, target, horizon, expected_reads)?; - if let Some(candidates) = self.window_framework_candidates.get(&key) { - return candidates - .iter() - .filter_map(|candidate| { - if candidate.assignments.len() != deployments.len() { - return None; - } - if !candidate - .accuracy - .matches_assignments(&candidate.assignments) - { - return None; - } - let window_frameworks = deployments - .iter() - .map(|deployment| { - candidate - .assignments - .iter() - .find(|assignment| { - std::ptr::eq(assignment.summary.as_ref(), deployment.summary) - }) - .map(|assignment| assignment.framework.clone()) - }) - .collect::>>()?; - let uses_exponential_histogram = window_frameworks.iter().any(|framework| { - matches!( - framework, - Some(SummaryWindowFramework::ExponentialHistogram) - ) - }); - let window_accuracy_guarantee = candidate.accuracy.end_to_end_guarantee( - uses_exponential_histogram, - root.guarantee.as_ref(), - )?; - if !required_accuracy.iter().all(|target| { - DefaultAccuracyModel.satisfies(&window_accuracy_guarantee, target) - }) { - return None; - } - self.complete_cost_with_evidence( - root, - deployments, - comparison, - &candidate.node_evidence, - &window_frameworks, - ) - .map(|cost| CompleteSummaryCandidateEstimate { - cost, - physical_plan_id: Some(candidate.physical_plan_id.clone()), - window_frameworks, - window_accuracy_guarantee: Some(window_accuracy_guarantee), - }) - }) - .min_by(|left, right| left.cost.0.total_cmp(&right.cost.0)); - } - if let Some(alternatives) = self.physical_plan_alternatives.get(&key) { - return alternatives - .iter() - .filter_map(|alternative| { - let frameworks = vec![None; deployments.len()]; - self.complete_cost_with_evidence( - root, - deployments, - comparison, - &alternative.node_evidence, - &frameworks, - ) - .map(|cost| CompleteSummaryCandidateEstimate { - cost, - physical_plan_id: Some(alternative.physical_plan_id.clone()), - window_frameworks: frameworks, - window_accuracy_guarantee: None, - }) - }) - .min_by(|left, right| left.cost.0.total_cmp(&right.cost.0)); - } - let frameworks = vec![None; deployments.len()]; - self.complete_cost_with_evidence( - root, - deployments, - comparison, - &self.node_evidence, - &frameworks, - ) - .map(|cost| CompleteSummaryCandidateEstimate { - cost, - physical_plan_id: None, - window_frameworks: frameworks, - window_accuracy_guarantee: None, - }) - } - - fn complete_summary_candidate_estimate_covers_lifecycle_costs(&self) -> bool { - true - } - - fn raw_query_recompute_cost(&self, target: &QueryExpr) -> Option { - let _ = target; - None - } - - fn raw_query_recompute_total_cost( - &self, - target: &QueryExpr, - expected_reads: f64, - ) -> Option { - let target_ptr = target as *const _; - let comparison = self.target_comparisons.get(&target_ptr)?; - let evaluations = comparison.scope.validate().ok()?; - if expected_reads != evaluations as f64 { - return None; - } - self.calibrated( - estimate_physical_dag( - &comparison.raw.physical_dag.nodes, - &comparison.raw.physical_dag.root, - &comparison.scope, - &comparison.raw.physical_dag, - ) - .ok()?, - ) - } -} - -use super::estimator::*; -#[cfg(test)] -mod tests { - use std::rc::Rc; - - use asap_types::post_asap::{ - EvaluationSchedule, ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, - OutputRepresentation, Schema, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, - }; - use asap_types::pre_asap::{ - agg_intent::AggIntent, ColumnRef, DataType, QueryExpr, Reduction, Source, - }; - use asap_types::workload::{ - DataWorkload, Evidence, EvidenceSource, Predictability, Query, QueryLanguage, - QueryRecurrence, QueryRequirements, QueryTimeScope, QueryWorkload, QueryWorkloadEntry, - Rate, RepeatedDemand, RepeatingEntry, RepetitionInterval, TimeSelection, - }; - - use super::*; - use crate::recurrence::Horizon; - use crate::summary_maintenance_lifecycle::{ - assemble_selected_dag_with_summary_maintenance_lifecycles, - global_selection_with_summary_maintenance_lifecycles, plan_summary_maintenance_lifecycles, - SummaryMaintenanceLifecycleCapabilities, WorkloadDemand, - }; - - fn estimate_test( - root: &SummaryNode, - guarantee: &SummaryMaintenanceLifecycleGuarantee, - inputs: SummaryMaintenanceInputs, - cpu: SummaryOperationCpuEvidence, - ) -> Result { - estimate_incremental_summary_maintenance(root, guarantee, inputs, cpu, &streaming_scope()) - } - - fn estimate_join_test( - root: &SummaryNode, - guarantee: &SummaryMaintenanceLifecycleGuarantee, - inputs: SummaryMaintenanceInputs, - cpu: SummaryOperationCpuEvidence, - join: Option, - ) -> Result { - estimate_incremental_summary_maintenance_with_join( - root, - guarantee, - inputs, - cpu, - join, - &streaming_scope(), - ) - } - - fn scope_for( - data: &DataWorkload, - query: &QueryWorkloadEntry, - planning_time_ms: u64, - horizon_ms: u64, - ) -> ComparisonScope { - ComparisonScope::from_workload( - data, - query, - asap_types::workload::TimestampMs(planning_time_ms), - asap_types::workload::DurationMs(horizon_ms), - vec![crate::physical_operator_statistics::SourceCoverage { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - source_snapshot_id: "stream-start".into(), - predicates: vec![], - info_matchers: vec![], - }], - ) - .unwrap() - } - - fn physical() -> SummaryPhysicalInputEvidence { - SummaryPhysicalInputEvidence { - initial_input_bytes: 640, - initial_source_scan_bytes: 640, - active_window_count: 2, - bootstrap_window_count: 1, - retained_window_count: 3, - physical_summary_count: 2, - state_bytes_per_summary: 100, - } - } - - fn query() -> QueryWorkloadEntry { - QueryWorkloadEntry { - query: Query("streaming count".into()), - requirements: QueryRequirements::default(), - predictability: Predictability::Unknown, - recurrence: QueryRecurrence::Repeated(RepeatedDemand::FixedInterval( - RepetitionInterval(1_000), - )), - time_selection: TimeSelection { - scope: QueryTimeScope::Unknown, - lookback: None, - as_of: None, - }, - } - } - - fn continuous_guarantee() -> SummaryMaintenanceLifecycleGuarantee { - SummaryMaintenanceLifecycleGuarantee { - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle::ContinuouslyMaintained, - summary_maintenance_mode: SummaryMaintenanceMode::Incremental, - evaluation_schedule: EvaluationSchedule::PerUpdate, - output_representation: OutputRepresentation::SummaryState, - } - } - - /// A fixed snapshot needs cardinality evidence, but no stream-rate evidence. - #[test] - fn at_rest_workload_adapter_builds_once_without_arrivals() { - let mut data = streaming_data_workload(); - data.arrival = DataArrival::AtRest; - data.ingestion_rate = Evidence::default(); - data.input_cardinality = Evidence { - value: Some(10), - source: EvidenceSource::Declared, - ..Default::default() - }; - let scope = scope_for(&data, &query(), 0, 5_000); - let inputs = SummaryMaintenanceInputs::from_workload(physical(), &data, &scope).unwrap(); - assert_eq!(inputs.initial_input_rows, 10); - assert_eq!(inputs.ingestion_rate_per_second, 0.0); - let guarantee = SummaryMaintenanceLifecycleGuarantee { - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle::Shared { - retention: asap_types::workload::DurationMs(5_000), - }, - summary_maintenance_mode: SummaryMaintenanceMode::DirectBuild, - evaluation_schedule: EvaluationSchedule::OnRead, - output_representation: OutputRepresentation::SummaryState, - }; - assert_eq!( - lifecycle_row_counts(inputs, &guarantee, &scope).unwrap(), - (10, 0, 5_000) - ); - } - - /// Arrival semantics cannot be overridden by missing or contradictory rate evidence. - #[test] - fn workload_adapter_checks_arrival_scope_and_rate_evidence() { - let mut data = streaming_data_workload(); - data.input_cardinality = Evidence { - value: Some(10), - source: EvidenceSource::Declared, - ..Default::default() - }; - let mut scope = scope_for(&data, &query(), 0, 5_000); - data.ingestion_rate = Evidence::default(); - assert_eq!( - SummaryMaintenanceInputs::from_workload(physical(), &data, &scope), - Err(AnalyticalCostError::MissingOrStale("ingestion_rate")) - ); - data.arrival = DataArrival::AtRest; - assert_eq!( - SummaryMaintenanceInputs::from_workload(physical(), &data, &scope), - Err(AnalyticalCostError::ComparisonScopeMismatch("data arrival")) - ); - scope.data_arrival = DataArrival::AtRest; - for rate in [1.0, -1.0, f64::INFINITY, f64::NAN] { - data.ingestion_rate = Evidence { - value: Some(Rate(rate)), - source: EvidenceSource::Declared, - ..Default::default() - }; - assert!(SummaryMaintenanceInputs::from_workload(physical(), &data, &scope).is_err()); - } - data.ingestion_rate = Evidence::default(); - data.input_cardinality = Evidence::default(); - assert_eq!( - SummaryMaintenanceInputs::from_workload(physical(), &data, &scope), - Err(AnalyticalCostError::MissingOrStale("input_cardinality")) - ); - } - - /// The real lifecycle planner costs a fixed snapshot with the same node evidence API. - #[test] - fn lifecycle_planner_selects_fully_costed_at_rest_summary() { - let workload = streaming_workload(); - let mut data = streaming_data_workload(); - data.arrival = DataArrival::AtRest; - data.ingestion_rate = Evidence::default(); - data.input_cardinality = Evidence { - value: Some(10), - source: EvidenceSource::Declared, - ..Default::default() - }; - let mut scope = streaming_scope(); - scope.data_arrival = DataArrival::AtRest; - let inputs = SummaryMaintenanceInputs::from_workload(physical(), &data, &scope).unwrap(); - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let mut provider = streaming_model(); - bind_aggregations(&mut provider, &target, &root, inputs, streaming_cpu()); - let mut model = streaming_model(); - model.node_evidence = provider.node_evidence; - let mut raw = streaming_raw(); - raw.ingestion_rate_per_second = 0.0; - let edge = EdgeStatistics { - rows: 50, - bytes: 3_200, - }; - raw.physical_dag - .evidence - .get_mut("raw-scan") - .unwrap() - .statistics = OperatorStatistics::Scan { - source_read_bytes: 3_200, - edges: UnaryEdgeStatistics { - input: edge, - output: edge, - promql: None, - }, - }; - model - .bind_candidate_comparison(&target, &root, scope.clone(), raw.clone()) - .unwrap(); - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &data, &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert!(plan.summary_total_cost.is_some()); - assert!(!plan.selected_raw_recompute); - assert_eq!( - plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap() - .summary_maintenance_mode, - SummaryMaintenanceMode::DirectBuild - ); - - // Directly supplied raw evidence must not bypass the workload invariant. - raw.ingestion_rate_per_second = 1.0; - assert_eq!( - streaming_model().bind_candidate_comparison(&target, &root, scope, raw), - Err(AnalyticalCostError::ComparisonScopeMismatch( - "at-rest ingestion rate" - )) - ); - // Nor may a provider hide arrivals on a summary edge. - for aggregation in model.node_evidence.aggregations.values_mut() { - aggregation.inputs.ingestion_rate_per_second = 1.0; - } - let invalid = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &data, &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(invalid.summary_total_cost, None); - } - - #[test] - fn workload_adapter_derives_updates_and_reads_over_one_horizon() { - let data = DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - ingestion_rate: Evidence { - value: Some(Rate(2.0)), - source: EvidenceSource::Observed, - observed_at_ms: Some(100), - valid_for_ms: Some(10_000), - }, - input_cardinality: Evidence { - value: Some(10), - source: EvidenceSource::Observed, - observed_at_ms: Some(100), - valid_for_ms: Some(10_000), - }, - ..DataWorkload::default() - }; - - let scope = scope_for(&data, &query(), 100, 5_000); - let inputs = SummaryMaintenanceInputs::from_workload(physical(), &data, &scope).unwrap(); - assert_eq!(inputs.initial_input_rows, 10); - assert_eq!( - lifecycle_row_counts(inputs, &continuous_guarantee(), &scope) - .unwrap() - .1, - 10 - ); - assert_eq!(scope.validate().unwrap(), 5); - } - - #[test] - fn pure_streaming_can_bootstrap_from_an_empty_state() { - let data = DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - ingestion_rate: Evidence { - value: Some(Rate(2.0)), - source: EvidenceSource::Declared, - observed_at_ms: None, - valid_for_ms: None, - }, - input_cardinality: Evidence { - value: Some(0), - source: EvidenceSource::Declared, - observed_at_ms: None, - valid_for_ms: None, - }, - ..DataWorkload::default() - }; - let mut empty = physical(); - empty.initial_input_bytes = 0; - empty.initial_source_scan_bytes = 0; - let scope = scope_for(&data, &query(), 0, 5_000); - let inputs = SummaryMaintenanceInputs::from_workload(empty, &data, &scope).unwrap(); - let estimate = estimate_test( - &summary_with_operations(false, false, false), - &continuous_guarantee(), - inputs, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }, - ) - .unwrap(); - // 10 arrivals * 2 active windows * 2 insert ops + 5 reads * 2 summaries. - assert_eq!(estimate.cpu_ops(), 50.0); - assert_eq!(estimate.scan_bytes(), 0); - } - - #[test] - fn bootstrap_rows_and_bytes_must_be_present_together() { - let mut inputs = SummaryMaintenanceInputs { - initial_input_rows: 0, - initial_input_bytes: 8, - initial_source_scan_bytes: 0, - ingestion_rate_per_second: 1.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 1, - physical_summary_count: 1, - state_bytes_per_summary: 8, - }; - assert_eq!( - inputs.validate(), - Err(AnalyticalCostError::InconsistentBootstrapEvidence) - ); - inputs.initial_input_rows = 1; - inputs.initial_input_bytes = 0; - assert_eq!( - inputs.validate(), - Err(AnalyticalCostError::InconsistentBootstrapEvidence) - ); - } - - #[test] - fn no_completed_windows_is_a_valid_streaming_deployment() { - let mut inputs = streaming_inputs(); - inputs.retained_window_count = 0; - assert!(inputs.validate().is_ok()); - } - - #[test] - fn bootstrap_rows_are_routed_to_declared_window_assignments() { - let mut inputs = streaming_inputs(); - inputs.ingestion_rate_per_second = 0.0; - inputs.bootstrap_window_count = 3; - let estimate = estimate_test( - &summary_with_operations(false, false, false), - &continuous_guarantee(), - inputs, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }, - ) - .unwrap(); - // 10 bootstrap rows * 3 windows * 2 insert ops + 5 reads * 2 summaries. - assert_eq!(estimate.cpu_ops(), 70.0); - } - - #[test] - fn lifecycle_output_must_remain_summary_state() { - let mut guarantee = continuous_guarantee(); - guarantee.output_representation = OutputRepresentation::FinalizedValue; - assert_eq!( - estimate_test( - &summary_with_operations(false, false, false), - &guarantee, - streaming_inputs(), - streaming_cpu(), - ), - Err(AnalyticalCostError::IncompatibleLifecycleGuarantee) - ); - } - - #[test] - fn existing_lifecycle_planner_selects_a_fully_costed_streaming_alternative() { - let inputs = SummaryMaintenanceInputs { - initial_input_rows: 10, - initial_input_bytes: 640, - initial_source_scan_bytes: 640, - ingestion_rate_per_second: 2.0, - active_window_count: 2, - bootstrap_window_count: 1, - retained_window_count: 3, - physical_summary_count: 2, - state_bytes_per_summary: 100, - }; - let mut model = streaming_model(); - let workload = QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: None, - repeating_queries: Some(vec![RepeatingEntry { - query: Query("streaming count".into()), - demand: RepeatedDemand::FixedInterval(RepetitionInterval(1_000)), - requirements: QueryRequirements::default(), - predictability: Predictability::Predictable { known_at: None }, - time_selection: TimeSelection::default(), - }]), - }; - let root = summary_with_operations(false, false, false); - let target = streaming_sum_query(); - bind_aggregations(&mut model, &target, &root, inputs, streaming_cpu()); - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - let selected = plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap(); - assert_eq!( - selected.summary_maintenance_mode, - SummaryMaintenanceMode::Incremental - ); - assert!(matches!( - selected.summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::Shared { .. } - | SummaryMaintenanceLifecycle::ContinuouslyMaintained - )); - assert!(plan.summary_total_cost.is_some()); - assert_eq!(model.raw_query_recompute_cost(&target), None); - } - - #[test] - fn complete_streaming_cost_can_select_an_ephemeral_direct_build() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: false, - }, - &model, - ) - .unwrap(); - - assert!(plan.summary_total_cost.is_some()); - assert!(matches!( - plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .map(|guarantee| &guarantee.summary_maintenance_lifecycle), - Some(SummaryMaintenanceLifecycle::Ephemeral) - )); - } - - #[test] - fn complete_streaming_cost_ranks_provider_owned_physical_plans() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - - let mut high_retention = model.node_evidence.clone(); - for aggregate in high_retention.aggregations.values_mut() { - aggregate.inputs.retained_window_count = 20; - } - let mut low_retention = model.node_evidence.clone(); - for aggregate in low_retention.aggregations.values_mut() { - aggregate.inputs.retained_window_count = 2; - } - for alternative in [ - SummaryPhysicalPlanAlternative { - physical_plan_id: "high-retention-layout".into(), - node_evidence: high_retention, - }, - SummaryPhysicalPlanAlternative { - physical_plan_id: "low-retention-layout".into(), - node_evidence: low_retention, - }, - ] { - model - .bind_physical_plan_alternative(&target, &root, alternative) - .unwrap(); - } - - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: false, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }, - &model, - ) - .unwrap(); - - assert_eq!( - plan.selected_window_implementation_id.as_deref(), - Some("low-retention-layout") - ); - assert_eq!( - crate::summary_maintenance_dag_export::export_summary_maintenance_plan(&plan) - .selected_window_implementation_id - .as_deref(), - Some("low-retention-layout") - ); - } - - #[test] - fn global_selection_compares_streaming_summary_and_raw_over_one_horizon() { - let target = streaming_sum_query(); - let space = crate::replacement::search_workload(vec![("q", Rc::clone(&target))]); - let workload = streaming_workload(); - let mut model = streaming_model(); - for group in space.target_subdag_candidates() { - for candidate in &group.candidates { - if let Replacement::Summary(root) = &candidate.replacement { - bind_aggregations( - &mut model, - &group.target, - root, - streaming_inputs(), - streaming_cpu(), - ); - } - } - } - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &space.roots[0].1, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap() - .unwrap(); - assert!(!plan.selected_raw_recompute); - assert_eq!(plan.raw_recompute_total_cost, Some(Cost(5_264.0))); - - let mut missing_baseline = model.clone(); - missing_baseline - .target_comparisons - .get_mut(&Rc::as_ptr(&space.roots[0].1)) - .unwrap() - .raw - .physical_dag - .evidence - .clear(); - let unavailable = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &missing_baseline, - ) - .unwrap(); - assert!(unavailable - .for_target(&space.roots[0].1) - .unwrap() - .chosen - .is_none()); - - let mut raw_cheaper = model; - for evidence in raw_cheaper.node_evidence.aggregations.values_mut() { - evidence.insert_cpu_ops = 10_000.0; - } - let cheap_selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &raw_cheaper, - ) - .unwrap(); - assert!(cheap_selection - .for_target(&space.roots[0].1) - .unwrap() - .chosen - .is_none()); - let cheap_plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &cheap_selection, - &space.roots[0].1, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &raw_cheaper, - ) - .unwrap() - .unwrap(); - assert!(cheap_plan.selected_raw_recompute); - assert_eq!(cheap_plan.raw_recompute_total_cost, Some(Cost(5_264.0))); - } - - #[test] - fn raw_evolution_is_bound_to_the_requested_target() { - let target_a = streaming_sum_query(); - let target_b = streaming_sum_query(); - let root_a = summary_with_operations(false, false, false); - let root_b = summary_with_operations(false, false, false); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target_a, - &root_a, - streaming_inputs(), - streaming_cpu(), - ); - let mut faster = streaming_inputs(); - // Candidate-local intermediate cardinality is not the raw target's - // planning-time cardinality and must not constrain its baseline. - faster.initial_input_rows = 7; - faster.initial_input_bytes = 448; - faster.initial_source_scan_bytes = 448; - faster.ingestion_rate_per_second = 4.0; - bind_aggregations(&mut model, &target_b, &root_b, faster, streaming_cpu()); - model - .target_comparisons - .get_mut(&Rc::as_ptr(&target_b)) - .unwrap() - .raw = { - let mut raw = streaming_raw(); - raw.ingestion_rate_per_second = 4.0; - let statistics = &mut raw - .physical_dag - .evidence - .get_mut("raw-scan") - .unwrap() - .statistics; - let OperatorStatistics::Scan { - edges, - source_read_bytes, - } = statistics - else { - unreachable!() - }; - *source_read_bytes = 7_040; - edges.input = EdgeStatistics { - rows: 110, - bytes: 7_040, - }; - edges.output = edges.input; - raw - }; - - let a = model.raw_query_recompute_total_cost(&target_a, 5.0); - let b = model.raw_query_recompute_total_cost(&target_b, 5.0); - assert_eq!(a, Some(Cost(5_264.0))); - assert!(b.unwrap().0 > a.unwrap().0); - assert_eq!(model.raw_query_recompute_total_cost(&target_a, 5.0), a); - } - - #[test] - fn raw_validation_uses_reachable_nodes_and_allows_repeated_source_scans() { - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let scope = streaming_scope(); - let mut raw = streaming_raw(); - let first_scan = raw.physical_dag.nodes[0].clone(); - let mut second_scan = first_scan.clone(); - second_scan.id = "raw-scan-2".into(); - let mut unreachable = first_scan.clone(); - unreachable.id = "unreachable-per-evaluation".into(); - unreachable.execution = ExecutionMultiplicity::PerEvaluation; - raw.physical_dag.nodes = vec![ - first_scan, - second_scan, - unreachable, - PhysicalDAGNode { - id: "raw-concat".into(), - operator: PhysicalOperator::Concat, - children: vec!["raw-scan".into(), "raw-scan-2".into()], - source_coverage: None, - output_buffer_bytes: 0, - retained_bytes: 0, - execution: ExecutionMultiplicity::Once, - }, - ]; - raw.physical_dag.root = "raw-concat".into(); - let scan_evidence = raw.physical_dag.evidence["raw-scan"].clone(); - let mut second_scan_evidence = scan_evidence.clone(); - second_scan_evidence.physical_id = "raw-scan-2".into(); - raw.physical_dag - .evidence - .insert("raw-scan-2".into(), second_scan_evidence); - let mut unreachable_evidence = scan_evidence; - unreachable_evidence.physical_id = "unreachable-per-evaluation".into(); - raw.physical_dag - .evidence - .insert("unreachable-per-evaluation".into(), unreachable_evidence); - raw.physical_dag.evidence.insert( - "raw-concat".into(), - PhysicalNodeEvidence { - physical_id: "raw-concat".into(), - statistics: OperatorStatistics::Concat { - inputs: vec![ - EdgeStatistics { - rows: 80, - bytes: 5_120, - }, - EdgeStatistics { - rows: 80, - bytes: 5_120, - }, - ], - output: EdgeStatistics { - rows: 160, - bytes: 10_240, - }, - promql: None, - }, - output_buffer_bytes: 0, - }, - ); - let mut model = streaming_model(); - assert!(model - .bind_candidate_comparison(&target, &root, scope, raw) - .is_ok()); - } - - #[test] - fn comparison_binding_is_transactional_and_shared_nodes_allow_two_targets() { - let target_a = streaming_sum_query(); - let target_b = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let mut model = streaming_model(); - let mut wrong_scope = streaming_scope(); - wrong_scope.sources[0].source = Source::TimeSeries { - metric: "wrong".into(), - }; - assert_eq!( - model.bind_candidate_comparison(&target_a, &root, wrong_scope, streaming_raw(),), - Err(AnalyticalCostError::ComparisonScopeMismatch( - "raw target source lineage" - )) - ); - assert!(model.target_comparisons.is_empty()); - assert!(model.candidate_comparisons.is_empty()); - - model - .bind_candidate_comparison(&target_a, &root, streaming_scope(), streaming_raw()) - .unwrap(); - model - .bind_candidate_comparison(&target_b, &root, streaming_scope(), streaming_raw()) - .unwrap(); - assert_eq!(model.candidate_comparisons.len(), 2); - } - - #[test] - fn target_scope_rejects_extra_sources_and_tracks_info_matchers() { - let target = streaming_sum_query(); - let mut extra = streaming_scope(); - extra - .sources - .push(crate::physical_operator_statistics::SourceCoverage { - source: Source::TimeSeries { - metric: "unused".into(), - }, - source_snapshot_id: "stream-start".into(), - predicates: vec![], - info_matchers: vec![], - }); - assert_eq!( - validate_query_scope(&target, &extra), - Err(AnalyticalCostError::ComparisonScopeMismatch( - "raw target source lineage" - )) - ); - - let selector = vec![InfoMatcher { - label: "job".into(), - op: CompareOpKind::Eq, - value: "api".into(), - }]; - let info_target = QueryExpr::PromqlInfoEnrich { - selector: selector.clone(), - child: target, - }; - let mut info_scope = streaming_scope(); - info_scope - .sources - .push(crate::physical_operator_statistics::SourceCoverage { - source: Source::TimeSeries { - metric: "target_info".into(), - }, - source_snapshot_id: "info-start".into(), - predicates: vec![], - info_matchers: selector, - }); - validate_query_scope(&info_target, &info_scope).unwrap(); - info_scope.sources[1].info_matchers[0].value = "worker".into(); - assert_eq!( - validate_query_scope(&info_target, &info_scope), - Err(AnalyticalCostError::ComparisonScopeMismatch( - "raw target source lineage" - )) - ); - } - - #[test] - fn delete_owner_must_be_the_unique_state_reachable_from_its_input() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, true); - let mut cpu = streaming_cpu(); - cpu.delete_cpu_ops = Some(1.0); - cpu.delete_events_per_second = Some(1.0); - cpu.delete_routing_fanout = Some(1); - let mut model = streaming_model(); - model.capabilities.delete = true; - bind_aggregations(&mut model, &target, &root, streaming_inputs(), cpu); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let delete_ptr = Rc::as_ptr(summary_input); - let unrelated = summary_with_operations(false, false, false); - let unrelated_agg = evidence_nodes(&unrelated).0[0] as *const _; - model - .node_evidence - .operation_state_owners - .insert(delete_ptr, unrelated_agg); - - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.summary_total_cost, None); - } - - #[test] - fn summary_edge_and_io_evidence_fail_closed() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_join(); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - let join = evidence_nodes(&root).1[0]; - model.node_evidence.joins.insert( - join as *const _, - SummaryJoinEvidence { - physical_id: "join-edge".into(), - inputs: vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); - let bad_edge = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(bad_edge.summary_total_cost, None); - - model - .node_evidence - .joins - .get_mut(&(join as *const _)) - .unwrap() - .inputs = vec![test_edge(), test_edge()]; - model - .node_evidence - .operations - .get_mut(&Rc::as_ptr(&root)) - .unwrap() - .resource_mut() - .io_bytes_per_execution = None; - let missing_io = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(missing_io.summary_total_cost, None); - } - - #[test] - fn summary_edges_io_and_physical_identity_fail_closed() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_join(); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - let (_, joins) = evidence_nodes(&root); - model.node_evidence.insert_join( - &Rc::new(joins[0].clone()), - SummaryJoinEvidence { - physical_id: "unused".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); - // Bind the actual join, then make one parent input disagree with its - // child's output. - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "join-edge".into(), - inputs: vec![test_edge(), EdgeStatistics { rows: 2, bytes: 16 }], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 1, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); - let bad_edge = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(bad_edge.summary_total_cost, None); - - model - .node_evidence - .joins - .get_mut(&(joins[0] as *const _)) - .unwrap() - .inputs = vec![test_edge(), test_edge()]; - model - .node_evidence - .operations - .get_mut(&Rc::as_ptr(&root)) - .unwrap() - .resource_mut() - .io_bytes_per_execution = None; - let missing_io = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(missing_io.summary_total_cost, None); - } - - #[test] - fn liveness_does_not_add_disjoint_execution_workspaces() { - let target = streaming_sum_query(); - let root = summary_join(); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - let join = evidence_nodes(&root).1[0]; - model.node_evidence.joins.insert( - join as *const _, - SummaryJoinEvidence { - physical_id: "huge-join".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: u64::MAX, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); - model - .node_evidence - .operations - .get_mut(&Rc::as_ptr(&root)) - .unwrap() - .resource_mut() - .working_memory_bytes = u64::MAX; - assert_eq!( - estimate_transient_liveness(&root, &model.node_evidence), - Ok(u64::MAX) - ); - } - - #[test] - fn conflicting_evidence_cannot_alias_one_provider_physical_identity() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_join(); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - let aggregations = evidence_nodes(&root).0; - let first = aggregations[0] as *const _; - let second = aggregations[1] as *const _; - model - .node_evidence - .aggregations - .get_mut(&first) - .unwrap() - .physical_id = "aliased-state".into(); - let second_evidence = model.node_evidence.aggregations.get_mut(&second).unwrap(); - second_evidence.physical_id = "aliased-state".into(); - second_evidence.insert_cpu_ops = 99.0; - - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.summary_total_cost, None); - } - - #[test] - fn lifecycle_plan_does_not_fall_back_to_partial_agg_cost_for_a_join_root() { - let workload = streaming_workload(); - let root = summary_join(); - let target = streaming_sum_query(); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.deployments.len(), 2); - assert_eq!(plan.summary_total_cost, None); - - let mut costed = model; - let join_node = evidence_nodes(&root).1[0]; - costed.node_evidence.joins.insert( - join_node as *const _, - SummaryJoinEvidence { - physical_id: "costed-join".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 6.0, - working_memory_bytes: 64, - output_buffer_bytes: 64, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); - let costed_plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &costed, - ) - .unwrap(); - assert!(costed_plan.summary_total_cost.is_some()); - } - - #[test] - fn whole_dag_cost_requires_and_uses_each_rc_bound_state_evidence() { - let workload = streaming_workload(); - let root = summary_join(); - let target = streaming_sum_query(); - let (aggregations, joins) = evidence_nodes(&root); - let mut model = streaming_model(); - bind_comparison(&mut model, &target, &root); - model.node_evidence.aggregations.insert( - aggregations[0] as *const _, - SummaryAggregateEvidence { - physical_id: "left-state".into(), - input: test_edge(), - output: test_edge(), - source_coverage_index: Some(0), - bootstrap_read_identity: "left-bootstrap".into(), - inputs: streaming_inputs(), - insert_cpu_ops: streaming_cpu().insert_cpu_ops.unwrap(), - }, - ); - let incomplete = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(incomplete.summary_total_cost, None); - - let mut second_inputs = streaming_inputs(); - second_inputs.state_bytes_per_summary = 250; - let mut second_cpu = streaming_cpu(); - second_cpu.insert_cpu_ops = Some(5.0); - model.node_evidence.aggregations.insert( - aggregations[1] as *const _, - SummaryAggregateEvidence { - physical_id: "right-state".into(), - input: test_edge(), - output: test_edge(), - source_coverage_index: Some(0), - bootstrap_read_identity: "right-bootstrap".into(), - inputs: second_inputs, - insert_cpu_ops: second_cpu.insert_cpu_ops.unwrap(), - }, - ); - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "join".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 6.0, - working_memory_bytes: 64, - output_buffer_bytes: 64, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); - model.node_evidence.operations.insert( - Rc::as_ptr(&root), - SummaryOperatorEvidence::Readout(SummaryOperatorResourceEvidence { - physical_id: "root-readout".into(), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops: 3.0, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }), - ); - let complete = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert!(complete.summary_total_cost.is_some()); - - model - .node_evidence - .operations - .get_mut(&Rc::as_ptr(&root)) - .unwrap() - .resource_mut() - .working_memory_bytes = 128; - let larger_workspace = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - // The join's 64-byte output remains live while the readout's workspace - // is active. The join's execution workspace is released first. - assert_eq!( - larger_workspace.summary_total_cost.unwrap().0 - complete.summary_total_cost.unwrap().0, - 64.0 - ); - - // Equal SourceCoverage does not imply that two independent state - // builds share one physical read. Only a provider-owned read identity - // permits scan de-duplication. - let mut shared_read = model; - for aggregate in shared_read.node_evidence.aggregations.values_mut() { - aggregate.bootstrap_read_identity = "one-physical-read".into(); - } - let shared = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &shared_read, - ) - .unwrap(); - assert_eq!( - larger_workspace.summary_total_cost.unwrap().0 - shared.summary_total_cost.unwrap().0, - 640.0 - ); - } - - #[test] - fn planner_selects_an_abstract_window_framework_from_downstream_evidence() { - let mut workload = streaming_workload(); - workload.repeating_queries.as_mut().unwrap()[0] - .requirements - .accuracy = - asap_types::workload::AccuracyRequirement::Explicit(AccuracyTarget::EpsilonDelta { - epsilon: 0.10, - delta: 0.01, - }); - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let windowed_summary = Rc::clone(summary_input); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - - let mut tumbling = model.node_evidence.clone(); - for aggregate in tumbling.aggregations.values_mut() { - aggregate.inputs.active_window_count = 1; - aggregate.inputs.retained_window_count = 20; - } - let mut sliding = model.node_evidence.clone(); - for aggregate in sliding.aggregations.values_mut() { - aggregate.inputs.active_window_count = 10; - aggregate.inputs.retained_window_count = 10; - } - let mut exponential_histogram = model.node_evidence.clone(); - for aggregate in exponential_histogram.aggregations.values_mut() { - aggregate.inputs.active_window_count = 2; - aggregate.inputs.retained_window_count = 2; - } - for candidate in [ - SummaryWindowFrameworkCandidate { - physical_plan_id: "tumbling-v1".into(), - assignments: vec![SummaryWindowFrameworkAssignment { - summary: Rc::clone(&windowed_summary), - framework: Some(SummaryWindowFramework::Tumbling), - }], - accuracy: SummaryWindowAccuracyEvidence::Exact, - node_evidence: tumbling, - }, - SummaryWindowFrameworkCandidate { - physical_plan_id: "sliding-v1".into(), - assignments: vec![SummaryWindowFrameworkAssignment { - summary: Rc::clone(&windowed_summary), - framework: Some(SummaryWindowFramework::Sliding), - }], - accuracy: SummaryWindowAccuracyEvidence::Exact, - node_evidence: sliding, - }, - SummaryWindowFrameworkCandidate { - physical_plan_id: "eh-v1".into(), - assignments: vec![SummaryWindowFrameworkAssignment { - summary: Rc::clone(&windowed_summary), - framework: Some(SummaryWindowFramework::ExponentialHistogram), - }], - accuracy: SummaryWindowAccuracyEvidence::ExponentialHistogram( - ExponentialHistogramAccuracyEvidence::UniversalGsum { - epsilon: 0.05, - failure_probability: 0.01, - range: ExponentialHistogramQueryRange::MostRecentWindow, - }, - ), - node_evidence: exponential_histogram, - }, - ] { - model - .bind_window_framework_candidate(&target, &root, candidate) - .unwrap(); - } - // Framework candidates are authoritative. Selection must not depend - // on duplicating one arbitrary implementation into the legacy global - // evidence map. - model.node_evidence = SummaryNodeEvidence::default(); - - let plan = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: false, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }, - &model, - ) - .unwrap(); - - assert_eq!( - plan.deployments[0].selected_window_framework, - Some(SummaryWindowFramework::ExponentialHistogram) - ); - assert_eq!( - plan.selected_window_implementation_id.as_deref(), - Some("eh-v1") - ); - let guarantee = plan.window_accuracy_guarantee.as_ref().unwrap(); - assert_eq!(guarantee.metric, ErrorMetric::RelativeValue); - assert!((guarantee.bound.evaluate().unwrap() - 0.05).abs() < f64::EPSILON); - let exported = - crate::summary_maintenance_dag_export::export_summary_maintenance_plan(&plan); - assert_eq!( - exported.deployments[0].selected_window_framework, - Some(SummaryWindowFramework::ExponentialHistogram) - ); - assert_eq!( - exported.selected_window_implementation_id.as_deref(), - Some("eh-v1") - ); - assert_eq!( - exported.window_accuracy_guarantee.unwrap().metric, - ErrorMetric::RelativeValue - ); - - workload.repeating_queries.as_mut().unwrap()[0] - .requirements - .accuracy = - asap_types::workload::AccuracyRequirement::Explicit(AccuracyTarget::Epsilon(0.01)); - let stricter = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: false, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }, - &model, - ) - .unwrap(); - assert_ne!( - stricter.deployments[0].selected_window_framework, - Some(SummaryWindowFramework::ExponentialHistogram) - ); - assert!(stricter.window_accuracy_guarantee.unwrap().is_exact()); - } - - #[test] - fn window_framework_candidates_require_unique_nonempty_planner_primitives() { - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let windowed_summary = Rc::clone(summary_input); - let mut model = streaming_model(); - bind_comparison(&mut model, &target, &root); - - let empty = model.bind_window_framework_candidate( - &target, - &root, - SummaryWindowFrameworkCandidate { - physical_plan_id: "empty-assignments".into(), - assignments: vec![], - accuracy: SummaryWindowAccuracyEvidence::Exact, - node_evidence: model.node_evidence.clone(), - }, - ); - assert!(matches!(empty, Err(AnalyticalCostError::MissingOrZero(_)))); - - let candidate = SummaryWindowFrameworkCandidate { - physical_plan_id: "tumbling-v1".into(), - assignments: vec![SummaryWindowFrameworkAssignment { - summary: windowed_summary, - framework: Some(SummaryWindowFramework::Tumbling), - }], - accuracy: SummaryWindowAccuracyEvidence::Exact, - node_evidence: model.node_evidence.clone(), - }; - model - .bind_window_framework_candidate(&target, &root, candidate.clone()) - .unwrap(); - assert!(matches!( - model.bind_window_framework_candidate(&target, &root, candidate), - Err(AnalyticalCostError::ComparisonScopeMismatch(_)) - )); - } - - #[test] - fn one_physical_identity_cannot_alias_different_window_frameworks() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_join(); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let SummaryExpr::SummaryJoin { outer, inner, .. } = &summary_input.expr else { - unreachable!(); - }; - let aggregation_nodes = [Rc::clone(outer), Rc::clone(inner)]; - let (aggregations, joins) = evidence_nodes(&root); - model.node_evidence.joins.insert( - joins[0] as *const _, - SummaryJoinEvidence { - physical_id: "joined-readout".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 1.0, - working_memory_bytes: 8, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - ); - - let mut shared_aggregation = - model.node_evidence.aggregations[&(aggregations[0] as *const _)].clone(); - shared_aggregation.physical_id = "shared-window-state".into(); - model - .node_evidence - .aggregations - .insert(aggregations[0] as *const _, shared_aggregation.clone()); - model - .node_evidence - .aggregations - .insert(aggregations[1] as *const _, shared_aggregation); - - let retained_children: Vec<_> = aggregation_nodes - .iter() - .map(|aggregate| match &aggregate.expr { - SummaryExpr::SummaryAgg { child, .. } => Rc::clone(child), - _ => unreachable!(), - }) - .collect(); - let mut shared_retained = - model.node_evidence.retained_queries[&Rc::as_ptr(&retained_children[0])].clone(); - shared_retained.physical_id = "shared-retained-input".into(); - for child in &retained_children { - model - .node_evidence - .retained_queries - .insert(Rc::as_ptr(child), shared_retained.clone()); - } - - let candidate = SummaryWindowFrameworkCandidate { - physical_plan_id: "mixed-framework-join".into(), - assignments: vec![ - SummaryWindowFrameworkAssignment { - summary: Rc::clone(&aggregation_nodes[0]), - framework: Some(SummaryWindowFramework::Tumbling), - }, - SummaryWindowFrameworkAssignment { - summary: Rc::clone(&aggregation_nodes[1]), - framework: Some(SummaryWindowFramework::Sliding), - }, - ], - accuracy: SummaryWindowAccuracyEvidence::Exact, - node_evidence: model.node_evidence.clone(), - }; - model - .bind_window_framework_candidate(&target, &root, candidate) - .unwrap(); - - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.summary_total_cost, None); - } - - #[test] - fn promsketch_eh_accuracy_composes_registered_full_and_subwindow_bounds() { - let full = SummaryWindowAccuracyEvidence::ExponentialHistogram( - ExponentialHistogramAccuracyEvidence::KllRank { - eh_epsilon: 0.01, - kll_epsilon: 0.02, - failure_probability: 0.01, - range: ExponentialHistogramQueryRange::MostRecentWindow, - }, - ) - .guarantee(true) - .unwrap(); - assert_eq!(full.metric, ErrorMetric::Rank); - assert!((full.bound.evaluate().unwrap() - 0.04).abs() < f64::EPSILON); - - let subwindow = SummaryWindowAccuracyEvidence::ExponentialHistogram( - ExponentialHistogramAccuracyEvidence::KllRank { - eh_epsilon: 0.01, - kll_epsilon: 0.02, - failure_probability: 0.01, - range: ExponentialHistogramQueryRange::SubWindow { - suffix_rows: 100, - query_rows: 25, - }, - }, - ) - .guarantee(true) - .unwrap(); - assert!((subwindow.bound.evaluate().unwrap() - 0.10).abs() < f64::EPSILON); - - let gsum = SummaryWindowAccuracyEvidence::ExponentialHistogram( - ExponentialHistogramAccuracyEvidence::UniversalGsum { - epsilon: 0.05, - failure_probability: 0.30, - range: ExponentialHistogramQueryRange::SubWindow { - suffix_rows: 100, - query_rows: 25, - }, - }, - ) - .guarantee(true) - .unwrap(); - assert_eq!(gsum.metric, ErrorMetric::RelativeValue); - assert!((gsum.bound.evaluate().unwrap() - 0.20).abs() < f64::EPSILON); - } - - #[test] - fn eh_accuracy_rejects_negative_components_and_mismatched_summary_guarantees() { - let evidence = SummaryWindowAccuracyEvidence::ExponentialHistogram( - ExponentialHistogramAccuracyEvidence::KllRank { - eh_epsilon: -0.01, - kll_epsilon: 0.03, - failure_probability: 0.01, - range: ExponentialHistogramQueryRange::MostRecentWindow, - }, - ); - assert!(evidence.guarantee(true).is_none()); - - let evidence = SummaryWindowAccuracyEvidence::ExponentialHistogram( - ExponentialHistogramAccuracyEvidence::KllRank { - eh_epsilon: 0.01, - kll_epsilon: 0.02, - failure_probability: 0.01, - range: ExponentialHistogramQueryRange::MostRecentWindow, - }, - ); - let actual_summary = ResultGuarantee { - metric: ErrorMetric::Rank, - bound: BoundExpr::Constant { value: 0.03 }, - failure_probability: ProbabilityExpr::Constant { value: 0.01 }, - provenance: vec![], - }; - assert!(evidence - .end_to_end_guarantee(true, Some(&actual_summary)) - .is_none()); - } - - #[test] - fn exponential_histogram_without_registered_accuracy_composition_fails_closed() { - let mut workload = streaming_workload(); - workload.repeating_queries.as_mut().unwrap()[0] - .requirements - .accuracy = - asap_types::workload::AccuracyRequirement::Explicit(AccuracyTarget::Epsilon(1.0)); - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - unreachable!(); - }; - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - model - .bind_window_framework_candidate( - &target, - &root, - SummaryWindowFrameworkCandidate { - physical_plan_id: "invalid-eh".into(), - assignments: vec![SummaryWindowFrameworkAssignment { - summary: Rc::clone(summary_input), - framework: Some(SummaryWindowFramework::ExponentialHistogram), - }], - accuracy: SummaryWindowAccuracyEvidence::Exact, - node_evidence: model.node_evidence.clone(), - }, - ) - .unwrap(); - - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(plan.summary_total_cost, None); - } - - #[test] - fn whole_dag_fails_closed_for_missing_retained_work_or_false_source_lineage() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - model.node_evidence.retained_queries.clear(); - let missing_retained = plan_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(missing_retained.summary_total_cost, None); - - bind_comparison(&mut model, &target, &root); - model - .target_comparisons - .get_mut(&Rc::as_ptr(&target)) - .unwrap() - .scope - .sources[0] - .source = Source::TimeSeries { - metric: "other_metric".into(), - }; - let false_lineage = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &model, - ) - .unwrap(); - assert_eq!(false_lineage.summary_total_cost, None); - } - - #[test] - fn aggregate_recurses_into_child_operations_and_state_only_needs_no_readout() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let estimated = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &estimated.expr else { - unreachable!(); - }; - let state_only = Rc::clone(summary_input); - let mut no_readout_cpu = streaming_cpu(); - no_readout_cpu.readout_cpu_ops = None; - let mut state_model = streaming_model(); - bind_aggregations( - &mut state_model, - &target, - &state_only, - streaming_inputs(), - no_readout_cpu, - ); - let state_plan = plan_summary_maintenance_lifecycles( - state_only, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &state_model, - ) - .unwrap(); - assert!(state_plan.summary_total_cost.is_some()); - - let child_readout = summary_with_operations(true, false, false); - let SummaryExpr::SummaryEstimate { - summary_input: child, - .. - } = &child_readout.expr - else { - unreachable!(); - }; - let child = Rc::clone(child); - let nested = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Wildcard), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::PerSubpopulationInstance, - filter: None, - }, - schema: estimated.schema.clone(), - guarantee: None, - }); - let mut nested_cpu = streaming_cpu(); - nested_cpu.merge_cpu_ops = Some(1.0); - let mut nested_model = streaming_model(); - bind_aggregations( - &mut nested_model, - &target, - &nested, - streaming_inputs(), - nested_cpu, - ); - nested_model - .node_evidence - .operations - .retain(|_, operation| { - operation.resource().cpu_ops != 1.0 - || operation.resource().working_memory_bytes == 0 - }); - let nested_plan = plan_summary_maintenance_lifecycles( - nested, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &nested_model, - ) - .unwrap(); - assert_eq!(nested_plan.summary_total_cost, None); - } - - #[test] - fn mixed_arrival_fails_closed_until_backlog_and_stream_are_separate() { - let data = DataWorkload { - arrival: DataArrival::Mixed, - ..DataWorkload::default() - }; - let mut scope = streaming_scope(); - scope.data_arrival = DataArrival::Mixed; - assert_eq!( - SummaryMaintenanceInputs::from_workload(physical(), &data, &scope), - Err(AnalyticalCostError::UnsupportedDataArrival( - DataArrival::Mixed - )) - ); - } - - #[test] - fn direct_read_costs_build_updates_windows_and_recurrence() { - let estimate = estimate_test( - &summary_with_operations(false, false, false), - &continuous_guarantee(), - SummaryMaintenanceInputs { - initial_input_rows: 10, - initial_input_bytes: 640, - initial_source_scan_bytes: 640, - ingestion_rate_per_second: 2.0, - active_window_count: 2, - bootstrap_window_count: 1, - retained_window_count: 3, - physical_summary_count: 2, - state_bytes_per_summary: 100, - }, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(3.0), - ..SummaryOperationCpuEvidence::default() - }, - ) - .unwrap(); - // 10 bootstrap + 10 arrivals into two active windows; two states read 5 times. - assert_eq!(estimate.cpu_ops(), 90.0); - assert_eq!(estimate.peak_memory_bytes(), 1_000); - assert_eq!(estimate.scan_bytes(), 640); - } - - #[test] - fn operations_use_update_or_read_multiplicity_and_shared_state_once() { - let estimate = estimate_test( - &summary_with_operations(true, true, true), - &continuous_guarantee(), - SummaryMaintenanceInputs { - initial_input_rows: 1, - initial_input_bytes: 8, - initial_source_scan_bytes: 8, - ingestion_rate_per_second: 4.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 2, - physical_summary_count: 2, - state_bytes_per_summary: 10, - }, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(1.0), - merge_cpu_ops: Some(2.0), - subtract_cpu_ops: Some(3.0), - delete_cpu_ops: Some(5.0), - delete_events_per_second: Some(4.0), - delete_routing_fanout: Some(2), - readout_cpu_ops: Some(7.0), - }, - ) - .unwrap(); - assert_eq!(estimate.cpu_ops(), 21.0 + 20.0 + 30.0 + 200.0 + 70.0); - // Three persistent windows plus one transient result, for two instances. - assert_eq!(estimate.peak_memory_bytes(), 80); - } - - #[test] - fn lifecycle_mode_and_schedule_must_match_existing_planner_semantics() { - let mut guarantee = continuous_guarantee(); - guarantee.evaluation_schedule = EvaluationSchedule::OnRead; - assert_eq!( - estimate_test( - &summary_with_operations(false, false, false), - &guarantee, - SummaryMaintenanceInputs { - initial_input_rows: 1, - initial_input_bytes: 8, - initial_source_scan_bytes: 8, - ingestion_rate_per_second: 1.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 1, - physical_summary_count: 1, - state_bytes_per_summary: 8, - }, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }, - ), - Err(AnalyticalCostError::IncompatibleLifecycleGuarantee) - ); - } - - #[test] - fn missing_cost_for_an_operation_in_the_dag_fails_closed() { - assert_eq!( - estimate_test( - &summary_with_operations(true, false, false), - &continuous_guarantee(), - SummaryMaintenanceInputs { - initial_input_rows: 1, - initial_input_bytes: 8, - initial_source_scan_bytes: 8, - ingestion_rate_per_second: 1.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 1, - physical_summary_count: 1, - state_bytes_per_summary: 8, - }, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }, - ), - Err(AnalyticalCostError::MissingOrStale("merge_cpu_ops")) - ); - } - - #[test] - fn direct_build_mode_is_not_mispriced_as_incremental_maintenance() { - let mut guarantee = continuous_guarantee(); - guarantee.summary_maintenance_lifecycle = SummaryMaintenanceLifecycle::Ephemeral; - guarantee.summary_maintenance_mode = SummaryMaintenanceMode::DirectBuild; - guarantee.evaluation_schedule = EvaluationSchedule::OneShot; - assert_eq!( - estimate_test( - &summary_with_operations(false, false, false), - &guarantee, - SummaryMaintenanceInputs { - initial_input_rows: 1, - initial_input_bytes: 8, - initial_source_scan_bytes: 8, - ingestion_rate_per_second: 1.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 1, - physical_summary_count: 1, - state_bytes_per_summary: 8, - }, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }, - ), - Err(AnalyticalCostError::IncompatibleLifecycleGuarantee) - ); - } - - #[test] - fn prepared_maintenance_charges_only_its_active_interval() { - let guarantee = SummaryMaintenanceLifecycleGuarantee { - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle::Prepared { - activate_at: asap_types::workload::TimestampMs(1_000), - retire_at: asap_types::workload::TimestampMs(6_000), - }, - summary_maintenance_mode: SummaryMaintenanceMode::Incremental, - evaluation_schedule: EvaluationSchedule::PerUpdate, - output_representation: OutputRepresentation::SummaryState, - }; - let estimate = estimate_test( - &summary_with_operations(false, false, false), - &guarantee, - SummaryMaintenanceInputs { - initial_input_rows: 10, - initial_input_bytes: 80, - initial_source_scan_bytes: 80, - ingestion_rate_per_second: 2.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 1, - physical_summary_count: 1, - state_bytes_per_summary: 8, - }, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }, - ) - .unwrap(); - // Two pre-activation arrivals join the bootstrap; eight more are - // maintained through the horizon; five reads are served. - assert_eq!(estimate.cpu_ops(), 25.0); - } - - #[test] - fn shared_retention_is_not_the_comparison_horizon() { - let guarantee = SummaryMaintenanceLifecycleGuarantee { - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle::Shared { - retention: asap_types::workload::DurationMs(999), - }, - summary_maintenance_mode: SummaryMaintenanceMode::Incremental, - evaluation_schedule: EvaluationSchedule::PerUpdate, - output_representation: OutputRepresentation::SummaryState, - }; - assert!(estimate_test( - &summary_with_operations(false, false, false), - &guarantee, - SummaryMaintenanceInputs { - initial_input_rows: 1, - initial_input_bytes: 8, - initial_source_scan_bytes: 8, - ingestion_rate_per_second: 1.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 1, - physical_summary_count: 1, - state_bytes_per_summary: 8, - }, - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }, - ) - .is_ok()); - } - - #[test] - fn lifecycle_retention_rate_integrates_to_one_peak_capacity_charge() { - let target = streaming_sum_query(); - let root = summary_with_operations(false, false, false); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - let aggregation = evidence_nodes(&root).0[0]; - let inputs = model - .lifecycle_inputs(aggregation, Some(Horizon(5.0))) - .unwrap(); - let integrated = inputs.retention_cost_rate.unwrap().0 * 5.0; - // (2 active + 3 retained) * 2 states * 100 bytes, calibrated once. - assert_eq!(integrated, 1_000.0); - } - - #[test] - fn summary_join_requires_cardinality_and_working_memory_evidence() { - let joined = summary_join(); - let inputs = SummaryMaintenanceInputs { - initial_input_rows: 1, - initial_input_bytes: 8, - initial_source_scan_bytes: 8, - ingestion_rate_per_second: 1.0, - active_window_count: 1, - bootstrap_window_count: 1, - retained_window_count: 1, - physical_summary_count: 1, - state_bytes_per_summary: 8, - }; - let cpu = SummaryOperationCpuEvidence { - insert_cpu_ops: Some(1.0), - readout_cpu_ops: Some(1.0), - ..SummaryOperationCpuEvidence::default() - }; - assert_eq!( - estimate_join_test(&joined, &continuous_guarantee(), inputs, cpu, None,), - Err(AnalyticalCostError::MissingOrStale("summary_join")) - ); - let estimate = estimate_join_test( - &joined, - &continuous_guarantee(), - inputs, - cpu, - Some(SummaryJoinEvidence { - physical_id: "diagnostic-join".into(), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops_per_execution: 12.0, - working_memory_bytes: 32, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }), - ) - .unwrap(); - assert_eq!(estimate.cpu_ops(), 77.0); - assert_eq!(estimate.peak_memory_bytes(), 64); // 4 persistent states + join memory. - } - - fn summary_with_operations(merge: bool, subtract: bool, delete: bool) -> Rc { - let state_type = FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count); - let schema = Schema::lifted(vec![Field::new("count", state_type.clone(), false)], None); - let leaf = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - })), - schema: schema.clone(), - guarantee: None, - }); - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: leaf, - family: state_type, - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Wildcard), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::PerSubpopulationInstance, - filter: None, - }, - schema: schema.clone(), - guarantee: None, - }); - let mut root = Rc::clone(&agg); - if merge { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&agg), Rc::clone(&agg)], - }, - schema: schema.clone(), - guarantee: None, - }); - } - if subtract { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummarySubtract { - left: Rc::clone(&root), - right: Rc::clone(&agg), - }, - schema: schema.clone(), - guarantee: None, - }); - } - if delete { - root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryDelete { - summary_input: root, - key: ColumnRef::Wildcard, - }, - schema: schema.clone(), - guarantee: None, - }); - } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: root, - query: asap_types::post_asap::SketchStatistic::PointCount { - key: ColumnRef::Wildcard, - value: None, - }, - }, - schema, - guarantee: Some(ResultGuarantee::exact("exact count readout")), - }) - } - - fn summary_join() -> Rc { - let left = summary_with_operations(false, false, false); - let right = summary_with_operations(false, false, false); - let SummaryExpr::SummaryEstimate { - summary_input: left, - .. - } = &left.expr - else { - unreachable!() - }; - let SummaryExpr::SummaryEstimate { - summary_input: right, - .. - } = &right.expr - else { - unreachable!() - }; - let schema = left.schema.clone(); - let join = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryJoin { - outer: Rc::clone(left), - inner: Rc::clone(right), - key: ColumnRef::Wildcard, - family: FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count), - }, - schema: schema.clone(), - guarantee: None, - }); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: join, - query: asap_types::post_asap::SketchStatistic::PointCount { - key: ColumnRef::Wildcard, - value: None, - }, - }, - schema, - guarantee: None, - }) - } - - fn summary_binary() -> Rc { - let operand = summary_with_operations(false, false, false); - Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: asap_types::post_asap::ExecutionTiming::QueryTime, - lhs: Rc::clone(&operand), - rhs: operand, - operator: asap_types::post_asap::BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - }, - }, - schema: Schema::lifted( - vec![Field::new( - "value", - FieldDataType::Plain(DataType::Float64), - false, - )], - None, - ), - guarantee: Some(ResultGuarantee::exact("test binary")), - }) - } - - #[test] - fn exact_binary_is_costable_with_explicit_physical_evidence() { - let workload = streaming_workload(); - let target = streaming_sum_query(); - let root = summary_binary(); - let mut model = streaming_model(); - bind_aggregations( - &mut model, - &target, - &root, - streaming_inputs(), - streaming_cpu(), - ); - - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &streaming_data_workload(), &[0]), - 0, - Some(Horizon(5.0)), - SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: false, - }, - &model, - ) - .expect("binary physical evidence should produce a complete cost"); - assert!(plan.summary_total_cost.is_some()); - } - - fn streaming_sum_query() -> Rc { - let scan = Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - }); - Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: scan, - }) - } - - fn streaming_workload() -> QueryWorkload { - QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: None, - repeating_queries: Some(vec![RepeatingEntry { - query: Query("sum(metrics)".into()), - demand: RepeatedDemand::FixedInterval(RepetitionInterval(1_000)), - requirements: QueryRequirements::default(), - predictability: Predictability::Predictable { known_at: None }, - time_selection: TimeSelection::default(), - }]), - } - } - - fn streaming_data_workload() -> DataWorkload { - DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - data_ingestion_interval: Evidence { - value: Some(asap_types::workload::DurationMs(1_000)), - ..Default::default() - }, - ingestion_rate: Evidence { - value: Some(Rate(2.0)), - source: EvidenceSource::Declared, - observed_at_ms: None, - valid_for_ms: None, - }, - ..Default::default() - } - } - - fn streaming_model() -> SummaryMaintenanceCostModel { - SummaryMaintenanceCostModel::new( - ResourceCalibration { - cost_per_cpu_op: 1.0, - cost_per_scan_byte: 1.0, - cost_per_retained_byte: 1.0, - version: "test".into(), - }, - SummaryMaintenanceCapabilities { - incremental_update: true, - merge: false, - delete: false, - }, - ) - } - - fn streaming_raw() -> RawInputEvidence { - let scope = streaming_scope(); - let node = PhysicalDAGNode { - id: "raw-scan".into(), - operator: PhysicalOperator::Scan, - children: vec![], - source_coverage: Some(scope.sources[0].clone()), - output_buffer_bytes: 0, - retained_bytes: 0, - execution: ExecutionMultiplicity::Once, - }; - let edge = EdgeStatistics { - rows: 80, - bytes: 5_120, - }; - let statistics = OperatorStatistics::Scan { - source_read_bytes: 5_120, - edges: UnaryEdgeStatistics { - input: edge, - output: edge, - promql: None, - }, - }; - RawInputEvidence { - planning_time_input_rows: 10, - planning_time_input_bytes: 640, - planning_time_source_scan_bytes: 640, - arriving_logical_row_bytes: 64, - arriving_source_row_bytes: 64, - ingestion_rate_per_second: 2.0, - physical_dag: EvidenceBackedPhysicalDAG { - nodes: vec![node], - root: "raw-scan".into(), - evidence: HashMap::from([( - "raw-scan".into(), - PhysicalNodeEvidence { - physical_id: "raw-scan".into(), - statistics, - output_buffer_bytes: 0, - }, - )]), - }, - } - } - - fn bind_comparison( - model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, - ) { - model - .bind_candidate_comparison(target, root, streaming_scope(), streaming_raw()) - .unwrap(); - fn retained( - model: &mut SummaryMaintenanceCostModel, - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, - ) { - if !seen.insert(Rc::as_ptr(node)) { - return; - } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => { - model.node_evidence.insert_retained_query( - node, - RetainedSubDAGEvidence { - physical_id: format!("retained-{node:p}"), - output: test_edge(), - preprocessing_cpu_ops_over_horizon: 1.0, - working_memory_bytes: 8, - output_buffer_bytes: 0, - }, - ); - } - SummaryExpr::SummaryAgg { child, .. } - | SummaryExpr::ValueOperation { child, .. } => retained(model, child, seen), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - retained(model, child, seen); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - retained(model, left, seen); - retained(model, right, seen); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - retained(model, summary_input, seen) - } - } - } - retained(model, root, &mut HashSet::new()); - } - - fn streaming_inputs() -> SummaryMaintenanceInputs { - SummaryMaintenanceInputs { - initial_input_rows: 10, - initial_input_bytes: 640, - initial_source_scan_bytes: 640, - ingestion_rate_per_second: 2.0, - active_window_count: 2, - bootstrap_window_count: 1, - retained_window_count: 3, - physical_summary_count: 2, - state_bytes_per_summary: 100, - } - } - - fn test_edge() -> EdgeStatistics { - EdgeStatistics { rows: 1, bytes: 8 } - } - - fn streaming_cpu() -> SummaryOperationCpuEvidence { - SummaryOperationCpuEvidence { - insert_cpu_ops: Some(2.0), - readout_cpu_ops: Some(3.0), - ..SummaryOperationCpuEvidence::default() - } - } - - fn bind_aggregations( - model: &mut SummaryMaintenanceCostModel, - target: &Rc, - root: &Rc, - inputs: SummaryMaintenanceInputs, - cpu: SummaryOperationCpuEvidence, - ) { - bind_comparison(model, target, root); - for node in evidence_nodes(root).0 { - let source_root = matches!( - &node.expr, - SummaryExpr::SummaryAgg { child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)) - ); - let mut node_inputs = inputs; - if !source_root { - node_inputs.initial_input_rows = test_edge().rows; - node_inputs.initial_input_bytes = test_edge().bytes; - node_inputs.initial_source_scan_bytes = 0; - } - model.node_evidence.aggregations.insert( - node as *const _, - SummaryAggregateEvidence { - physical_id: format!("agg-{node:p}"), - input: test_edge(), - output: test_edge(), - source_coverage_index: source_root.then_some(0), - bootstrap_read_identity: if source_root { - "shared-bootstrap".into() - } else { - String::new() - }, - inputs: node_inputs, - insert_cpu_ops: cpu.insert_cpu_ops.unwrap(), - }, - ); - } - fn bind_ops( - model: &mut SummaryMaintenanceCostModel, - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - inputs: SummaryMaintenanceInputs, - cpu: SummaryOperationCpuEvidence, - ) { - if !seen.insert(node as *const _) { - return; - } - let operation = match &node.expr { - SummaryExpr::BinaryOp { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Binary(SummaryOperatorResourceEvidence { - physical_id: format!("binary-{node:p}"), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) - }), - SummaryExpr::ValueOperation { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::ValueOperation(SummaryOperatorResourceEvidence { - physical_id: format!("value-operation-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) - }), - SummaryExpr::SummaryMerge { .. } => cpu.merge_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Merge(SummaryOperatorResourceEvidence { - physical_id: format!("merge-{node:p}"), - inputs: match &node.expr { - SummaryExpr::SummaryMerge { children, .. } => { - vec![test_edge(); children.len()] - } - _ => unreachable!(), - }, - output: test_edge(), - cpu_ops, - working_memory_bytes: inputs.state_bytes_per_summary, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) - }), - SummaryExpr::SummarySubtract { .. } => cpu.subtract_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Subtract(SummaryOperatorResourceEvidence { - physical_id: format!("subtract-{node:p}"), - inputs: vec![test_edge(), test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: inputs.state_bytes_per_summary, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) - }), - SummaryExpr::SummaryDelete { .. } => cpu.delete_cpu_ops.and_then(|cpu_ops| { - Some(SummaryOperatorEvidence::Delete { - resource: SummaryOperatorResourceEvidence { - physical_id: format!("delete-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }, - events_per_second: cpu.delete_events_per_second?, - routing_fanout: cpu.delete_routing_fanout?, - }) - }), - SummaryExpr::SummaryEstimate { .. } => cpu.readout_cpu_ops.map(|cpu_ops| { - SummaryOperatorEvidence::Readout(SummaryOperatorResourceEvidence { - physical_id: format!("readout-{node:p}"), - inputs: vec![test_edge()], - output: test_edge(), - cpu_ops, - working_memory_bytes: 0, - output_buffer_bytes: 0, - executions_per_evaluation: 1, - io_bytes_per_execution: Some(0), - }) - }), - _ => None, - }; - if let Some(operation) = operation { - model - .node_evidence - .operations - .insert(node as *const _, operation); - if let SummaryExpr::SummaryDelete { summary_input, .. } = &node.expr { - fn owning_aggs( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - owners: &mut Vec<*const SummaryNode>, - ) { - if !seen.insert(node as *const _) { - return; - } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - owners.push(node as *const _); - owning_aggs(child, seen, owners); - } - SummaryExpr::ValueOperation { child, .. } => { - owning_aggs(child, seen, owners) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - owning_aggs(child, seen, owners); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - owning_aggs(left, seen, owners); - owning_aggs(right, seen, owners); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - owning_aggs(summary_input, seen, owners); - } - SummaryExpr::KeepPreAsap(_) => {} - } - } - let mut owners = Vec::new(); - owning_aggs(summary_input, &mut HashSet::new(), &mut owners); - owners.sort_unstable(); - owners.dedup(); - if let [owner] = owners.as_slice() { - model - .node_evidence - .operation_state_owners - .insert(node as *const _, *owner); - } - } - } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } - | SummaryExpr::ValueOperation { child, .. } => { - bind_ops(model, child, seen, inputs, cpu) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - bind_ops(model, child, seen, inputs, cpu); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - bind_ops(model, left, seen, inputs, cpu); - bind_ops(model, right, seen, inputs, cpu); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - bind_ops(model, summary_input, seen, inputs, cpu) - } - SummaryExpr::KeepPreAsap(_) => {} - } - } - bind_ops(model, root, &mut HashSet::new(), inputs, cpu); - } - - fn streaming_scope() -> ComparisonScope { - let workload = streaming_workload(); - let entry = workload.entries().next().unwrap(); - ComparisonScope::from_workload( - &DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - data_ingestion_interval: Evidence { - value: Some(asap_types::workload::DurationMs(1_000)), - ..Default::default() - }, - ingestion_rate: Evidence { - value: Some(Rate(2.0)), - source: EvidenceSource::Declared, - observed_at_ms: None, - valid_for_ms: None, - }, - ..Default::default() - }, - &entry, - asap_types::workload::TimestampMs(0), - asap_types::workload::DurationMs(5_000), - vec![crate::physical_operator_statistics::SourceCoverage { - source: Source::TimeSeries { - metric: "metrics".into(), - }, - source_snapshot_id: "stream-start".into(), - predicates: vec![], - info_matchers: vec![], - }], - ) - .unwrap() - } -} diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs deleted file mode 100644 index c3c12c20b..000000000 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/window.rs +++ /dev/null @@ -1,278 +0,0 @@ -use super::*; - -/// One per-state window choice within a complete Planner candidate. -#[derive(Debug, Clone)] -pub struct SummaryWindowFrameworkAssignment { - pub summary: Rc, - /// `None` explicitly means that this state is not window-organized. - pub framework: Option, -} - -/// Cost evidence for one complete abstract window-framework assignment across -/// a summary DAG in Planner search. -/// -/// The provider derives this evidence from a concrete downstream -/// implementation under the current data workload. The stable identity is -/// provenance for the chosen implementation, while deployment placement and -/// runtime configuration remain downstream concerns. -#[derive(Debug, Clone)] -pub struct SummaryWindowFrameworkCandidate { - /// Stable identity of the complete provider implementation whose evidence - /// is bound to this planner-visible framework assignment. - pub physical_plan_id: String, - /// Exactly one assignment for every summary deployment in the DAG. - pub assignments: Vec, - /// Registered end-to-end accuracy composition for this complete window - /// assignment. EH combinations must use one of the specialized proofs; - /// unknown combinations fail closed. - pub accuracy: SummaryWindowAccuracyEvidence, - pub node_evidence: SummaryNodeEvidence, -} - -pub(super) fn summary_aggregation_identities(root: &SummaryNode) -> HashSet<*const SummaryNode> { - fn visit( - node: &SummaryNode, - seen: &mut HashSet<*const SummaryNode>, - out: &mut HashSet<*const SummaryNode>, - ) { - if !seen.insert(node as *const _) { - return; - } - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => { - out.insert(node as *const _); - visit(child, seen, out); - } - SummaryExpr::ValueOperation { child, .. } => visit(child, seen, out), - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - visit(child, seen, out); - } - } - SummaryExpr::SummarySubtract { left, right } - | SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::BinaryOp { - lhs: left, - rhs: right, - .. - } - | SummaryExpr::SummaryJoin { - outer: left, - inner: right, - .. - } => { - visit(left, seen, out); - visit(right, seen, out); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - visit(summary_input, seen, out); - } - } - } - - let mut out = HashSet::new(); - visit(root, &mut HashSet::new(), &mut out); - out -} - -/// Cardinality normalization used by the PromSketch EH bounds. The paper's -/// sub-window error is stated relative to the suffix beginning at the query's -/// left endpoint, so a query-relative bound needs `suffix_rows/query_rows`. -#[derive(Debug, Clone, Copy, PartialEq)] -pub enum ExponentialHistogramQueryRange { - MostRecentWindow, - SubWindow { suffix_rows: u64, query_rows: u64 }, -} - -/// Registered accuracy compositions for Exponential Histogram realizations. -#[derive(Debug, Clone, Copy, PartialEq)] -pub enum ExponentialHistogramAccuracyEvidence { - /// PromSketch EHKLL normalized rank error. - KllRank { - eh_epsilon: f64, - kll_epsilon: f64, - failure_probability: f64, - range: ExponentialHistogramQueryRange, - }, - /// PromSketch EHUniv/GSum relative error. - UniversalGsum { - epsilon: f64, - failure_probability: f64, - range: ExponentialHistogramQueryRange, - }, -} - -#[derive(Debug, Clone, PartialEq)] -pub enum SummaryWindowAccuracyEvidence { - /// The window implementation preserves exact query-time coverage and adds - /// no error. Used for exact tumbling/sliding realizations. - Exact, - ExponentialHistogram(ExponentialHistogramAccuracyEvidence), -} - -impl ExponentialHistogramQueryRange { - fn suffix_to_query_ratio(self) -> Option { - match self { - Self::MostRecentWindow => Some(1.0), - Self::SubWindow { - suffix_rows, - query_rows, - } if query_rows > 0 && suffix_rows >= query_rows => { - Some(suffix_rows as f64 / query_rows as f64) - } - Self::SubWindow { .. } => None, - } - } -} - -impl SummaryWindowAccuracyEvidence { - pub(super) fn matches_assignments( - &self, - assignments: &[SummaryWindowFrameworkAssignment], - ) -> bool { - let eh_summaries: Vec<_> = assignments - .iter() - .filter(|assignment| { - assignment.framework == Some(SummaryWindowFramework::ExponentialHistogram) - }) - .collect(); - match self { - Self::Exact => eh_summaries.is_empty(), - Self::ExponentialHistogram(ExponentialHistogramAccuracyEvidence::KllRank { - .. - }) => { - eh_summaries.len() == 1 - && eh_summaries.iter().all(|assignment| { - matches!( - &assignment.summary.expr, - SummaryExpr::SummaryAgg { - family: FieldDataType::Sketch(kind, _), - .. - } if kind.algorithm() == &SketchAlgorithm::Kll - ) - }) - } - Self::ExponentialHistogram(ExponentialHistogramAccuracyEvidence::UniversalGsum { - .. - }) => { - eh_summaries.len() == 1 - && eh_summaries.iter().all(|assignment| { - matches!( - &assignment.summary.expr, - SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate( - ExactKind::Count | ExactKind::Sum, - _ - ), - .. - } - ) - }) - } - } - } - - /// Compose the two EH combinations proved by PromSketch - /// (doi:10.14778/3742728.3742732). Unknown EH combinations deliberately - /// have no catch-all arm. - pub(super) fn guarantee(&self, uses_exponential_histogram: bool) -> Option { - match self { - Self::Exact if !uses_exponential_histogram => { - Some(ResultGuarantee::exact("exact window coverage")) - } - Self::Exact => None, - Self::ExponentialHistogram(evidence) if uses_exponential_histogram => { - let (metric, bound, failure_probability, rule) = match *evidence { - ExponentialHistogramAccuracyEvidence::KllRank { - eh_epsilon, - kll_epsilon, - failure_probability, - range, - } => { - if !eh_epsilon.is_finite() - || eh_epsilon < 0.0 - || !kll_epsilon.is_finite() - || kll_epsilon < 0.0 - { - return None; - } - ( - ErrorMetric::Rank, - 2.0 * eh_epsilon * range.suffix_to_query_ratio()? + kll_epsilon, - failure_probability, - "promsketch_eh_kll_rank", - ) - } - ExponentialHistogramAccuracyEvidence::UniversalGsum { - epsilon, - failure_probability, - range, - } => { - if !epsilon.is_finite() || epsilon < 0.0 { - return None; - } - ( - ErrorMetric::RelativeValue, - epsilon * range.suffix_to_query_ratio()?, - failure_probability, - "promsketch_eh_universal_gsum", - ) - } - }; - if !bound.is_finite() - || bound < 0.0 - || !failure_probability.is_finite() - || !(0.0..=1.0).contains(&failure_probability) - { - return None; - } - Some(ResultGuarantee { - metric, - bound: BoundExpr::Constant { value: bound }, - failure_probability: ProbabilityExpr::Constant { - value: failure_probability, - }, - provenance: vec![GuaranteeSource::RuntimeObservation { - source: rule.into(), - detail: serde_json::json!({ - "reference": "doi:10.14778/3742728.3742732" - }), - }], - }) - } - Self::ExponentialHistogram(_) => None, - } - } - - pub(super) fn end_to_end_guarantee( - &self, - uses_exponential_histogram: bool, - summary_guarantee: Option<&ResultGuarantee>, - ) -> Option { - let summary = summary_guarantee?; - match self { - Self::Exact if !uses_exponential_histogram => Some(summary.clone()), - Self::ExponentialHistogram(ExponentialHistogramAccuracyEvidence::KllRank { - kll_epsilon, - failure_probability, - .. - }) if uses_exponential_histogram => { - let summary_bound = summary.bound.evaluate()?; - let summary_failure = summary.failure_probability.evaluate()?; - if summary.metric != ErrorMetric::Rank - || summary_bound != *kll_epsilon - || summary_failure != *failure_probability - { - return None; - } - self.guarantee(true) - } - Self::ExponentialHistogram(ExponentialHistogramAccuracyEvidence::UniversalGsum { - .. - }) if uses_exponential_histogram && summary.is_exact() => self.guarantee(true), - _ => None, - } - } -} diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs deleted file mode 100644 index 8e63a4ce1..000000000 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ /dev/null @@ -1,162 +0,0 @@ -//! Serializable DAG export for a materialized summary-maintenance plan. -//! -//! `asap-types::dag_export` owns the crate-neutral post-ASAP DAG shape. This -//! adapter lives in the mapping layer, where summary-maintenance lifecycle -//! alternatives and their typed rejection reasons are available, and emits -//! both views together. - -use std::collections::HashMap; -use std::rc::Rc; - -use serde::Serialize; - -use asap_types::dag_export::{self, SummaryDAG}; -use asap_types::post_asap::{ - PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, -}; - -use crate::summary_maintenance_lifecycle::{ - SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecycleRejection, -}; - -#[derive(Debug, Clone, Serialize)] -pub struct SummaryMaintenanceDAGExport { - pub dag: SummaryDAG, - pub deployments: Vec, - pub horizon_seconds: Option, - pub evaluation_rate_per_second: Option, - pub update_rate_per_second: Option, - pub expected_reads: Option, - pub selected_raw_recompute: bool, - #[serde(skip_serializing_if = "Option::is_none")] - /// Provider implementation key. The legacy JSON field name is retained - /// until the surrounding export receives its own schema-version bump. - #[serde(rename = "selected_physical_plan_id")] - pub selected_window_implementation_id: Option, - pub summary_total_cost: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub window_accuracy_guarantee: Option, - pub raw_recompute_total_cost: Option, -} - -#[derive(Debug, Clone, Serialize)] -pub struct SummaryMaintenanceDeploymentExport { - pub post_asap_node_id: PostAsapNodeId, - #[serde(skip_serializing_if = "Option::is_none")] - pub selected_window_framework: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub selected: Option, - pub alternatives: Vec, -} - -#[derive(Debug, Clone, Serialize)] -pub struct SummaryMaintenanceLifecycleAlternativeExport { - pub lifecycle: SummaryMaintenanceLifecycle, - pub total_cost: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub rejection: Option, - pub assumptions: Vec, -} - -pub type SummaryMaintenanceLifecycleGuaranteeExport = SummaryMaintenanceLifecycleGuarantee; - -pub fn export_summary_maintenance_plan( - plan: &SummaryMaintenanceLifecyclePlan, -) -> SummaryMaintenanceDAGExport { - let deployments: Vec<_> = plan - .deployments - .iter() - .map(|deployment| SummaryMaintenanceDeploymentExport { - post_asap_node_id: deployment.post_asap_node_id, - selected_window_framework: deployment.selected_window_framework.clone(), - selected: deployment - .summary_maintenance_lifecycle_guarantee - .as_ref() - .cloned(), - alternatives: deployment - .alternatives - .iter() - .map(|alternative| SummaryMaintenanceLifecycleAlternativeExport { - lifecycle: alternative.summary_maintenance_lifecycle.clone(), - total_cost: alternative.total_cost.map(|cost| cost.0), - rejection: alternative.rejection.clone(), - assumptions: alternative.assumptions.clone(), - }) - .collect(), - }) - .collect(); - let mut dag = dag_export::export_summary(&plan.root); - let deployment_by_summary: HashMap<_, _> = plan - .deployments - .iter() - .zip(&deployments) - .map(|(deployment, export)| (Rc::as_ptr(&deployment.summary), export)) - .collect(); - let mut next_node_id = 0; - annotate_lifecycle_deployments( - &plan.root, - &mut dag, - &deployment_by_summary, - &mut next_node_id, - ); - - SummaryMaintenanceDAGExport { - dag, - deployments, - horizon_seconds: plan.horizon.map(|horizon| horizon.0), - evaluation_rate_per_second: plan.evaluation_rate.map(|rate| rate.0), - update_rate_per_second: plan.update_rate.map(|rate| rate.0), - expected_reads: plan.expected_reads, - selected_raw_recompute: plan.selected_raw_recompute, - selected_window_implementation_id: plan.selected_window_implementation_id.clone(), - summary_total_cost: plan.summary_total_cost.map(|cost| cost.0), - window_accuracy_guarantee: plan.window_accuracy_guarantee.clone(), - raw_recompute_total_cost: plan.raw_recompute_total_cost.map(|cost| cost.0), - } -} - -/// Walk in the same post-order as `dag_export::export_summary` and attach a -/// deployment directly to every flattened occurrence of its state node. -/// This makes the decision visible to DAG consumers without asking them to -/// reconstruct pointer identity from DAG position. -fn annotate_lifecycle_deployments( - node: &SummaryNode, - dag: &mut SummaryDAG, - deployments: &HashMap<*const SummaryNode, &SummaryMaintenanceDeploymentExport>, - next_node_id: &mut usize, -) { - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { - for child in summary_children(&node.expr) { - annotate_lifecycle_deployments(child, dag, deployments, next_node_id); - } - } - let dag_node = &mut dag.nodes[*next_node_id]; - if let Some(deployment) = deployments.get(&(node as *const SummaryNode)) { - dag_node.detail["summary_maintenance"] = - serde_json::to_value(deployment).expect("lifecycle export is serializable"); - } - *next_node_id += 1; -} - -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { - match expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], - SummaryExpr::SummaryAgg { child, .. } => vec![child], - SummaryExpr::ValueOperation { child, .. } => vec![child], - SummaryExpr::SummaryJoin { outer, inner, .. } - | SummaryExpr::RelationalJoin { - left: outer, - right: inner, - .. - } - | SummaryExpr::SummarySubtract { - left: outer, - right: inner, - } => vec![outer, inner], - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryMerge { children, .. } => children.iter().collect(), - } -} diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs deleted file mode 100644 index cf41efeff..000000000 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ /dev/null @@ -1,3847 +0,0 @@ -//! Workload-aware summary-maintenance lifecycle planning. -//! -//! A **summary-maintenance lifecycle** is the planner policy for when one -//! materialized summary state is created, retained or shared, updated as data -//! arrives, and retired. It is deliberately narrower than the end-to-end data -//! lifecycle and independent of query recurrence. Recurrence says when and how -//! often queries will read the result. The planner converts that demand into -//! expected reads and an evaluation rate, then uses those quantities to compare -//! rebuilding per query with retaining or continuously maintaining state. -//! Recurrence does not itself prescribe a state-maintenance policy. -//! -//! This module enumerates and costs `Ephemeral`, `Prepared`, `Shared`, and -//! `ContinuouslyMaintained` alternatives for every unique `SummaryAgg` in a -//! materialized plan, and for every maintained population (`MaintainPopulation`) -//! that is not an input of a `SummaryAgg`. [`SummaryMaintenanceMode`] is an orthogonal detail of -//! the selected deployment: state is either built directly or updated -//! incrementally. Unknown evidence stays unknown and therefore cannot make a -//! long-lived alternative win. - -use std::collections::{HashMap, HashSet}; -use std::rc::Rc; - -use asap_types::post_asap::{ - compile_post_asap_dag_with_node_ids, share_common_summary_sub_dags, EvaluationSchedule, - ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDAG, - PostAsapDAGValidationError, PostAsapNodeId, ResultGuarantee, SummaryExpr, - SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, - SummaryNode, SummaryWindowFramework, ValueOperation, -}; -use asap_types::pre_asap::QueryExpr; -use asap_types::types::AccuracyTarget; -use asap_types::workload::{ - DataArrival, DataWorkload, Predictability, QueryRecurrence, QueryWorkload, RepeatedDemand, - TimestampMs, WorkloadError, -}; - -use crate::analytical_cost::AnalyticalCostError; -use crate::cost_model::{ - CompleteSummaryCandidateEstimate, Cost, CostModel, CostedSummaryDeployment, -}; -use crate::physical_operator_statistics::evaluations_in_horizon; -use crate::recurrence::{ - CostRate, EvaluationRate, Horizon, RecurrenceError, RecurrenceProfile, UpdateRate, -}; -use crate::replacement::{ - CandidateCostOverrides, CandidateLogicalASAPDAGs, GlobalSelection, RealizationError, - Replacement, -}; - -/// Summary-maintenance lifecycle shapes supported by the target runtime. -/// -/// These independent flags describe the set of lifecycle alternatives the -/// runtime implements, not simultaneous states of one deployment. Multiple -/// flags may be `true` (a runtime can support both ephemeral and prepared -/// state, for example); the planner still selects exactly one mutually -/// exclusive [`SummaryMaintenanceLifecycle`] for each deployment. A supported -/// alternative may still be rejected because workload evidence is missing or -/// its cost is unknown. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct SummaryMaintenanceLifecycleCapabilities { - /// The runtime can build a fresh state for each invocation and retire it - /// after that invocation finishes. - pub supports_ephemeral: bool, - /// The runtime can build state before a predictable execution and retain - /// it until that scheduled execution window ends. - pub supports_prepared: bool, - /// The runtime can retain one state and reuse it across multiple reads. - pub supports_shared: bool, - /// The runtime can keep state current by applying arriving data updates. - pub supports_continuously_maintained: bool, -} - -/// State operations supported by one concrete summary family and -/// representation. -/// -/// This differs from [`SummaryMaintenanceLifecycleCapabilities`]: these flags -/// describe what the summary algorithm itself can do, while lifecycle -/// capabilities describe what deployment policies the target runtime can -/// orchestrate. -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -pub struct SummaryMaintenanceCapabilities { - /// Existing state can incorporate arriving input without a full rebuild. - pub incremental_update: bool, - /// Two independently built states can be combined into one equivalent - /// state. - pub merge: bool, - /// Expired or retracted input can be removed from existing state. - pub delete: bool, -} - -impl SummaryMaintenanceLifecycleCapabilities { - pub const ALL: Self = Self { - supports_ephemeral: true, - supports_prepared: true, - supports_shared: true, - supports_continuously_maintained: true, - }; -} - -impl Default for SummaryMaintenanceLifecycleCapabilities { - fn default() -> Self { - Self::ALL - } -} - -/// Primitive costs for one concrete summary state. Every field is optional: -/// missing statistics produce an uncosted alternative, never a zero. -/// -/// The lifecycle planner combines these state-specific inputs with workload -/// rates, invocation counts, and the optimization horizon. All `Cost` fields -/// are one-time costs unless their name explicitly says otherwise. -#[derive(Debug, Clone, Default, PartialEq)] -pub struct SummaryMaintenanceLifecycleCostInputs { - /// One-time cost to construct the state from its input. - pub build_cost: Option, - /// Cost to incorporate one arriving input update into existing state. - pub maintenance_cost_per_update: Option, - /// Cost of one read or finalization from already-built summary state. - pub summary_read_cost: Option, - /// Cost per second for retaining the state over a lifecycle window. - pub retention_cost_rate: Option, - /// One-time cost to release or retire the state. - pub retirement_cost: Option, -} - -#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize)] -#[serde(rename_all = "snake_case")] -pub enum SummaryMaintenanceLifecycleRejection { - UnsupportedByRuntime, - RequiresPredictableOneTimeQuery, - RequiresMultipleReads, - RequiresHorizon, - RequiresContinuousData, - MissingOrStaleIngestionRate, - SummaryDoesNotSupportIncrementalUpdates, - SummaryDoesNotSupportDeletion, - MissingCostEvidence, -} - -/// One candidate lifecycle policy for a particular summary deployment. -/// -/// `total_cost: None` never means zero: it means the planner lacks enough -/// evidence to cost the candidate. Such a candidate is not selectable and its -/// `rejection` explains why. -#[derive(Debug, Clone, PartialEq)] -pub struct SummaryMaintenanceLifecycleAlternative { - /// State creation, retention, sharing, update, and retirement policy. - pub summary_maintenance_lifecycle: SummaryMaintenanceLifecycle, - /// Complete cost over the requested horizon, when every input is known. - pub total_cost: Option, - /// Why this alternative cannot be selected; `None` means it is legal and - /// fully costed. - pub rejection: Option, - /// Human-readable premises used when deriving and costing the alternative. - pub assumptions: Vec, -} - -impl SummaryMaintenanceLifecycleAlternative { - fn selectable(&self) -> bool { - self.rejection.is_none() && self.total_cost.is_some() - } -} - -/// One unique retained-state deployment. Shared `Rc` nodes are emitted once. -#[derive(Debug, Clone)] -pub struct SummaryMaintenanceDeployment { - /// Identity of this summary in the exported post-ASAP semantic DAG. - /// It is scoped to one plan version and is not a summary definition or - /// summary instance identity. - pub post_asap_node_id: PostAsapNodeId, - /// The unique materialized `SummaryAgg`, or maintained population - /// (`MaintainPopulation`) not consumed by a `SummaryAgg`, represented by - /// this deployment. Cost-model lifecycle hooks receive this node. - pub summary: Rc, - /// Lifecycle, evaluation, and representation commitment selected for this - /// state, or `None` when no alternative is selectable. - pub summary_maintenance_lifecycle_guarantee: Option, - /// Abstract window primitive selected for this state. Concrete runtime - /// implementation, placement, and identity remain downstream decisions. - pub selected_window_framework: Option, - /// Every lifecycle shape considered, including rejected and uncosted ones. - pub alternatives: Vec, -} - -/// Workload-aware lifecycle and window-framework decisions for every unique -/// summary state reachable from one materialized post-ASAP root. -#[derive(Debug, Clone)] -pub struct SummaryMaintenanceLifecyclePlan { - /// Root of the materialized post-ASAP DAG being deployed. - pub root: Rc, - /// One entry per unique reachable `SummaryAgg`, then per unique - /// maintained population outside any `SummaryAgg`'s inputs; shared `Rc` - /// nodes appear only once. - pub deployments: Vec, - /// Caller-supplied optimization horizon used to turn rates into total - /// costs. `None` keeps horizon-dependent alternatives unselectable. - pub horizon: Option, - /// Aggregate recurring query-evaluation rate derived from the workload. - pub evaluation_rate: Option, - /// Fresh source-data ingestion rate, when supplied by the workload. - pub update_rate: Option, - /// Total demand inside the horizon, when every recurrence is known. - pub expected_reads: Option, - /// Whether global costing preferred rebuilding the raw expression over all - /// summary deployments. - pub selected_raw_recompute: bool, - /// Provider-owned identity of the selected complete physical deployment - /// (for example a tumbling, sliding, or exponential-histogram plan). - pub selected_window_implementation_id: Option, - /// Cost of the selected set of summary deployments, when fully known. - pub summary_total_cost: Option, - /// Composed accuracy guarantee supplied by the selected physical window - /// evidence, when the window framework introduces approximation. - pub window_accuracy_guarantee: Option, - /// Cost of evaluating the original expression for the same demand, when - /// fully known. - pub raw_recompute_total_cost: Option, -} - -/// Why a lifecycle plan cannot assign execution timing to its DAG. -#[derive(Debug, thiserror::Error, PartialEq)] -pub enum SummaryMaintenanceTimingError { - #[error(transparent)] - InvalidPostAsapDAG(#[from] ExecutionDataStateError), - #[error("summary {0:?} has no selected lifecycle")] - UnselectedLifecycle(PostAsapNodeId), - /// A maintained population outside any `SummaryAgg`'s inputs has no - /// deployment, so its timing would be guessed. Enumeration always emits - /// one; this arises only for a plan whose root or deployments were edited. - #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] - UnplannedMaintainedState(PostAsapNodeId), - #[error(transparent)] - InvalidPhases(#[from] PostAsapDAGValidationError), -} - -impl SummaryMaintenanceLifecyclePlan { - /// The post-ASAP DAG of [`Self::root`] with every node's timing derived - /// from the selected lifecycles, so physical compilation places it. - /// - /// A retained (non-`Ephemeral`) state outlives one query, so it and every - /// input it consumes run at ingestion time. Every other node runs at query - /// time: readouts and consumers of retained state, and each `Ephemeral` - /// state not consumed by retained state together with its inputs, whose - /// raw data the deployment must supply as a query source. This applies to - /// maintained populations as to `SummaryAgg` states; a population feeding - /// a `SummaryAgg` is one of its inputs. Timings already on the root are - /// ignored. - pub fn execution_timed_dag(&self) -> Result { - let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; - let dag = compiled.dag; - for population in &standalone_populations(&self.root) { - let id = compiled - .node_ids - .node_id(population) - .expect("collected population belongs to the compiled DAG"); - if !self - .deployments - .iter() - .any(|deployment| deployment.post_asap_node_id == id) - { - return Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)); - } - } - let mut pending = Vec::new(); - for deployment in &self.deployments { - let guarantee = deployment - .summary_maintenance_lifecycle_guarantee - .as_ref() - .ok_or(SummaryMaintenanceTimingError::UnselectedLifecycle( - deployment.post_asap_node_id, - ))?; - if guarantee.summary_maintenance_lifecycle != SummaryMaintenanceLifecycle::Ephemeral { - pending.push(deployment.post_asap_node_id); - } - } - let mut ingestion = HashSet::new(); - while let Some(id) = pending.pop() { - if ingestion.insert(id) { - pending.extend( - dag.edges - .iter() - .filter(|edge| edge.consumer == id) - .map(|edge| edge.producer), - ); - } - } - let phases = dag - .nodes - .iter() - .map(|node| { - let timing = if ingestion.contains(&node.id) { - ExecutionTiming::IngestionTime - } else { - ExecutionTiming::QueryTime - }; - (node.id, timing) - }) - .collect(); - Ok(dag.with_execution_phases(&phases)?) - } -} - -/// Explicit association between a materialized target and the normalized -/// workload entries whose demand consumes it. -/// -/// [`QueryWorkload`] remains the source of query demand, while source-data -/// evidence is supplied independently. Indices avoid copying normalized entry -/// definitions while ensuring unrelated entries do not influence a target's -/// lifecycle decision. -#[derive(Debug, Clone, Copy)] -pub struct WorkloadDemand<'a> { - /// Original normalized query workload. - pub workload: &'a QueryWorkload, - /// Independent source-data evidence, when the caller has it. - pub data_workload: Option<&'a DataWorkload>, - /// Indices from [`QueryWorkload::entries`] that consume this target. - pub entry_indices: &'a [usize], -} - -impl<'a> WorkloadDemand<'a> { - /// Bind query demand without source-data evidence. Callers that have a - /// [`DataWorkload`] should use [`Self::new_with_data`] so ingestion facts - /// are not silently discarded. - pub const fn new_without_data(workload: &'a QueryWorkload, entry_indices: &'a [usize]) -> Self { - Self { - workload, - data_workload: None, - entry_indices, - } - } - - pub const fn new_with_data( - workload: &'a QueryWorkload, - data_workload: &'a DataWorkload, - entry_indices: &'a [usize], - ) -> Self { - Self { - workload, - data_workload: Some(data_workload), - entry_indices, - } - } -} - -#[derive(Debug, thiserror::Error)] -pub enum SummaryMaintenanceLifecyclePlanError { - #[error(transparent)] - InvalidWorkload(#[from] WorkloadError), - #[error("optimization horizon must be finite and strictly positive")] - InvalidHorizon, - #[error("workload entry index {index} is out of bounds for {entry_count} entries")] - InvalidWorkloadEntry { index: usize, entry_count: usize }, - #[error("a workload-demand binding must contain at least one entry")] - EmptyWorkloadDemand, - #[error("workload entry index {index} appears more than once in one demand binding")] - DuplicateWorkloadEntry { index: usize }, - #[error(transparent)] - InvalidPostAsapDAG(#[from] ExecutionDataStateError), -} - -#[derive(Debug, thiserror::Error)] -pub enum SummaryMaintenanceLifecycleAssemblyError { - #[error(transparent)] - AssembleDAG(#[from] RealizationError), - #[error(transparent)] - SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), -} - -/// Failure while deriving workload-aware candidate costs before global -/// selection. -#[derive(Debug, thiserror::Error)] -pub enum SummaryMaintenanceLifecycleSelectionError { - #[error(transparent)] - Recurrence(#[from] RecurrenceError), - #[error(transparent)] - SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), -} - -/// Every lifecycle alternative for each unique retained state of one fixed -/// root, before any lifecycle is chosen. -/// -/// Planner selection ([`plan_summary_maintenance_lifecycles`]) and a -/// deployment's explicit choice ([`Self::select`]) both finish from this value, -/// so they produce the same [`SummaryMaintenanceLifecyclePlan`] shape. -pub struct SummaryMaintenanceLifecycleCandidates<'a> { - /// Unselected plan: deployments carry alternatives but no guarantee or - /// window framework. - plan: SummaryMaintenanceLifecyclePlan, - components: Vec, - arrival: DataArrival, - required_accuracy: Vec, - cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, -} - -/// Why an explicit per-state lifecycle choice cannot be bound. -#[derive(Debug, thiserror::Error, PartialEq)] -pub enum SummaryMaintenanceLifecycleChoiceError { - #[error("summary {0:?} is not a deployment of this root")] - UnknownSummary(PostAsapNodeId), - #[error("summary {0:?} is chosen more than once")] - DuplicateChoice(PostAsapNodeId), - #[error("summary {0:?} has no chosen lifecycle")] - MissingChoice(PostAsapNodeId), - #[error("chosen lifecycle is not an enumerated alternative of summary {0:?}")] - NotAnAlternative(PostAsapNodeId), - #[error("chosen lifecycle of summary {post_asap_node_id:?} is rejected: {rejection:?}")] - Rejected { - post_asap_node_id: PostAsapNodeId, - rejection: Option, - }, - #[error("summary states on one maintenance path have different evaluation schedules")] - IncompatibleEvaluationSchedules, - #[error("the cost model supplied no complete estimate for the chosen combination")] - NoCompleteEstimate, -} - -impl SummaryMaintenanceLifecycleCandidates<'_> { - /// One entry per unique retained state (see - /// [`SummaryMaintenanceLifecyclePlan::deployments`]), with every - /// alternative and its rejection; no lifecycle or window framework is - /// selected. - pub fn deployments(&self) -> &[SummaryMaintenanceDeployment] { - &self.plan.deployments - } - - /// Guarantee that binding `lifecycle` would attach under this workload's - /// data arrival, so a caller can price an alternative before choosing it. - pub fn guarantee( - &self, - lifecycle: &SummaryMaintenanceLifecycle, - ) -> SummaryMaintenanceLifecycleGuarantee { - lifecycle_guarantee(lifecycle, self.arrival) - } - - fn context(&self) -> CompleteCostContext<'_> { - CompleteCostContext { - root: &self.plan.root, - components: &self.components, - cost_model: self.cost_model, - comparison_target: self.comparison_target, - horizon: self.plan.horizon, - expected_reads: self.plan.expected_reads, - required_accuracy: &self.required_accuracy, - } - } - - fn finish( - mut self, - estimate: Option, - ) -> SummaryMaintenanceLifecyclePlan { - if let Some(estimate) = estimate { - self.plan.summary_total_cost = Some(estimate.cost); - self.plan.selected_window_implementation_id = estimate.physical_plan_id; - self.plan.window_accuracy_guarantee = estimate.window_accuracy_guarantee; - } - self.plan - } - - /// Planner's choice: the cheapest complete combination of eligible - /// alternatives. - fn select_cheapest(mut self) -> SummaryMaintenanceLifecyclePlan { - let estimate = select_complete_lifecycle_combination( - &self.plan.root, - &mut self.plan.deployments, - &self.components, - self.arrival, - self.cost_model, - self.comparison_target, - self.plan.horizon, - self.plan.expected_reads, - &self.required_accuracy, - ); - self.finish(estimate) - } - - /// Bind one caller-chosen lifecycle per summary state. Each choice must be - /// an alternative Planner itself could select; the complete estimate is - /// then obtained exactly as for Planner selection, so window framework and - /// cost are the model's and unknown cost is never replaced by zero. - pub fn select( - mut self, - choices: &[(PostAsapNodeId, SummaryMaintenanceLifecycle)], - ) -> Result { - use SummaryMaintenanceLifecycleChoiceError as E; - let deployments = &self.plan.deployments; - let mut chosen: Vec> = - vec![None; deployments.len()]; - let context = self.context(); - for (id, lifecycle) in choices { - let index = deployments - .iter() - .position(|deployment| deployment.post_asap_node_id == *id) - .ok_or(E::UnknownSummary(*id))?; - if chosen[index].is_some() { - return Err(E::DuplicateChoice(*id)); - } - let alternative = deployments[index] - .alternatives - .iter() - .find(|alternative| alternative.summary_maintenance_lifecycle == *lifecycle) - .ok_or(E::NotAnAlternative(*id))?; - if !context.eligible(alternative) { - return Err(E::Rejected { - post_asap_node_id: *id, - rejection: alternative.rejection.clone(), - }); - } - chosen[index] = Some(alternative); - } - let selected = chosen - .into_iter() - .enumerate() - .map(|(index, alternative)| { - let alternative = - alternative.ok_or(E::MissingChoice(deployments[index].post_asap_node_id))?; - Ok(( - index, - lifecycle_guarantee(&alternative.summary_maintenance_lifecycle, self.arrival), - // Reached only for costed alternatives or when the - // complete hook is authoritative, matching Planner search. - alternative.total_cost.unwrap_or(Cost::ZERO), - )) - }) - .collect::, E>>()?; - if selected.is_empty() { - return Ok(self.finish(None)); - } - if !context.schedules_compatible(&selected) { - return Err(E::IncompatibleEvaluationSchedules); - } - let estimate = context - .estimate(deployments, &selected) - .ok_or(E::NoCompleteEstimate)?; - let guarantees = selected - .into_iter() - .map(|(index, guarantee, _)| (index, guarantee)) - .collect(); - apply_selection(&mut self.plan.deployments, guarantees, &estimate); - Ok(self.finish(Some(estimate))) - } -} - -/// Workload-wide evidence derived specifically for summary-maintenance -/// lifecycle enumeration and costing. -/// -/// This is not another workload input model. [`QueryWorkload`] and its -/// normalized entries remain the source of truth. Unlike one -/// [`asap_types::workload::QueryWorkloadEntry`], these values aggregate all -/// entries at a particular planning time and optional horizon. It also cannot -/// reuse [`crate::recurrence::RecurrenceProfile`], which describes recurrence -/// for one candidate target and counts consumers rather than invocations. -#[derive(Debug)] -struct SummaryMaintenanceWorkloadFacts { - required_accuracy: Vec, - /// Total one-time and recurring reads inside the horizon. `None` means a - /// recurrence or horizon was unknown, not zero reads. - reads: Option, - /// Sum of declared invocations across all one-time workload entries. - one_time_invocations: u64, - /// Sum of usable recurring query rates in evaluations per second. - evaluation_rate: Option, - /// Fresh workload-level ingestion rate in updates per second. - update_rate: Option, - /// Whether the workload's source data is static, arriving, mixed, or - /// unknown. - arrival: DataArrival, - /// Earliest known activation and latest scheduled execution across - /// predictable one-time entries. `None` means no valid preparation window. - prepared_window: Option<(TimestampMs, TimestampMs)>, - /// Whether every bound consumer is a predictable one-time query suitable - /// for prepared state. - prepared_eligible: bool, - /// Whether maintaining the selected moving time scope requires deleting - /// expired input from summary state. - requires_deletion: bool, -} - -/// Validate a materialized plan, enumerate lifecycle alternatives for each -/// unique summary state, and select the cheapest legal alternative whose cost -/// is fully known. -pub fn plan_summary_maintenance_lifecycles( - root: Rc, - demand: WorkloadDemand<'_>, - now_ms: u64, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &dyn CostModel, -) -> Result { - Ok(enumerate_summary_maintenance_lifecycles( - root, - demand, - now_ms, - horizon, - capabilities, - cost_model, - )? - .select_cheapest()) -} - -/// Validate a materialized plan and enumerate lifecycle alternatives for each -/// unique summary state without choosing one. A deployment that prices the -/// alternatives itself binds its choice with -/// [`SummaryMaintenanceLifecycleCandidates::select`]. -pub fn enumerate_summary_maintenance_lifecycles<'a>( - root: Rc, - demand: WorkloadDemand<'_>, - now_ms: u64, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &'a dyn CostModel, -) -> Result, SummaryMaintenanceLifecyclePlanError> { - enumerate_with_profile( - root, - demand, - now_ms, - horizon, - capabilities, - cost_model, - None, - None, - ) -} - -/// Internal candidate-costing form. The workload binding supplies temporal -/// eligibility and data-arrival facts; `profile` supplies effective uses after -/// DAG path multiplicity has been propagated by `CandidateLogicalASAPDAGs`. -#[expect(clippy::too_many_arguments, reason = "internal bound planning context")] -fn enumerate_with_profile<'a>( - root: Rc, - demand: WorkloadDemand<'_>, - now_ms: u64, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &'a dyn CostModel, - profile: Option, - comparison_target: Option<&'a QueryExpr>, -) -> Result, SummaryMaintenanceLifecyclePlanError> { - demand.workload.validate()?; - if let Some(data) = demand.data_workload { - data.validate()?; - } - if horizon.is_some_and(|h| !h.0.is_finite() || h.0 <= 0.0) { - return Err(SummaryMaintenanceLifecyclePlanError::InvalidHorizon); - } - let mut facts = workload_facts( - demand.workload, - demand.data_workload, - demand.entry_indices, - now_ms, - horizon, - )?; - if let Some(profile) = profile { - facts.one_time_invocations = u64::try_from(profile.one_shot_consumers).unwrap_or(u64::MAX); - facts.evaluation_rate = profile.evaluation_rate; - facts.update_rate = profile.update_rate; - facts.reads = match (profile.evaluation_rate, horizon) { - (Some(rate), Some(horizon)) => { - Some(profile.one_shot_consumers as f64 + rate.0 * horizon.0) - } - (Some(_), None) => None, - (None, _) if profile.one_shot_consumers > 0 => Some(profile.one_shot_consumers as f64), - // Preserve unknown recurrence from the normalized workload. An - // empty profile does not prove that the target is never read. - (None, _) => facts.reads, - }; - } - let mut summaries = Vec::new(); - collect_states( - &root, - &mut HashSet::new(), - &mut summaries, - StateKind::SummaryAgg, - ); - summaries.extend(standalone_populations(&root)); - let node_ids = compile_post_asap_dag_with_node_ids(&root)?.node_ids; - let components = summary_state_components(&summaries); - let deployments: Vec = summaries - .into_iter() - .map(|summary| { - let alternatives = alternatives_for( - &facts, - horizon, - capabilities, - cost_model.summary_maintenance_capabilities(&summary), - cost_model.summary_maintenance_lifecycle_cost_inputs_for_horizon(&summary, horizon), - ); - SummaryMaintenanceDeployment { - post_asap_node_id: node_ids - .node_id(&summary) - .expect("collected summary belongs to the compiled DAG"), - summary, - summary_maintenance_lifecycle_guarantee: None, - selected_window_framework: None, - alternatives, - } - }) - .collect(); - let selected_raw_recompute = matches!(root.expr, SummaryExpr::KeepPreAsap(_)); - Ok(SummaryMaintenanceLifecycleCandidates { - plan: SummaryMaintenanceLifecyclePlan { - root, - deployments, - horizon, - evaluation_rate: facts.evaluation_rate, - update_rate: facts.update_rate, - expected_reads: facts.reads, - selected_raw_recompute, - selected_window_implementation_id: None, - summary_total_cost: None, - window_accuracy_guarantee: None, - raw_recompute_total_cost: None, - }, - components, - arrival: facts.arrival, - required_accuracy: facts.required_accuracy, - cost_model, - comparison_target, - }) -} - -/// Rank semantic summary siblings using the cheapest legal -/// summary-maintenance lifecycle for each candidate before final global -/// selection. The candidate space stays compact; only cost overrides are -/// attached, so shared `Rc` identity and exact-composition commitments remain -/// the responsibility of `GlobalSelection`. -/// -/// Summary candidates of different targets whose outermost `SummaryAgg` is -/// structurally identical (for example p50 and p99 over one KLL) form a class. -/// When [`shared_state_cost`] can cost that state once against the union of -/// the targets' entries, each member is offered an equal split of it instead -/// of its independent cost. If selection then leaves any member of a class on -/// another choice, that class reverts to independent costs and selection runs -/// once more. -pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( - space: &'a CandidateLogicalASAPDAGs, - demand: WorkloadDemand<'_>, - now_ms: u64, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &dyn CostModel, -) -> Result, SummaryMaintenanceLifecycleSelectionError> { - let WorkloadDemand { - workload, - data_workload, - entry_indices: root_workload_entries, - } = demand; - let profiles = space.recurrence_profiles_from_workload( - workload, - data_workload, - root_workload_entries, - now_ms, - horizon, - )?; - let bindings = space.workload_entries_by_target(workload, root_workload_entries)?; - let mut costs = CandidateCostOverrides::default(); - // Finalized summary candidates, as sharing-class members. - let mut members = Vec::new(); - for group in space.target_subdag_candidates() { - let Some(entry_indices) = bindings.get(&Rc::as_ptr(&group.target)) else { - continue; - }; - for candidate in &group.candidates { - let Replacement::Summary(summary) = &candidate.replacement else { - continue; - }; - costs.finalize_target(&group.target); - let plan = enumerate_with_profile( - Rc::clone(summary), - WorkloadDemand { - workload, - data_workload, - entry_indices, - }, - now_ms, - horizon, - capabilities, - cost_model, - Some(profiles.for_target(&group.target)), - Some(&group.target), - )? - .select_cheapest(); - let raw = plan - .expected_reads - .and_then(|reads| cost_model.raw_query_recompute_total_cost(&group.target, reads)); - // Final comparison is atomic: without the raw side, no summary - // override is published even when that summary alone is costed. - if let Some(raw) = raw { - costs.insert_raw(&group.target, raw); - if !plan.deployments.is_empty() { - if let Some(total) = plan.summary_total_cost { - costs.insert(&group.target, candidate, total); - } - members.push((group, candidate, Rc::clone(summary))); - } - } - } - } - - // Intern every member once; members whose outermost state (the - // `SummaryAgg` every other state of the candidate feeds) interns to the - // same node share it. Classes are kept in first-member order. - let interned = share_common_summary_sub_dags( - members - .iter() - .enumerate() - .map(|(index, (_, _, summary))| (index, Rc::clone(summary))) - .collect(), - ); - let mut classes: Vec<(Rc, Vec)> = Vec::new(); - for (index, root) in interned { - let states = summary_states(&root); - let Some(state) = states - .iter() - .find(|state| summary_states(state).len() == states.len()) - else { - continue; - }; - if !standalone_populations(&root).is_empty() { - continue; - } - match classes.iter_mut().find(|(s, _)| Rc::ptr_eq(s, state)) { - Some((_, class)) => class.push(index), - None => classes.push((Rc::clone(state), vec![index])), - } - } - let mut shared = Vec::new(); - for (state, class) in classes { - let mut targets: Vec<&Rc> = Vec::new(); - for &index in &class { - let target = &members[index].0.target; - if !targets.iter().any(|t| Rc::ptr_eq(t, target)) { - targets.push(target); - } - } - if targets.len() < 2 { - continue; - } - let mut entries: Vec = targets - .iter() - .flat_map(|target| bindings[&Rc::as_ptr(target)].iter().copied()) - .collect(); - entries.sort_unstable(); - entries.dedup(); - let Some(cost) = shared_state_cost( - &state, - WorkloadDemand { - workload, - data_workload, - entry_indices: &entries, - }, - now_ms, - horizon, - capabilities, - cost_model, - )? - else { - continue; - }; - shared.push((class, Cost(cost.0 / targets.len() as f64))); - } - - let with_shared = |kept: &[(Vec, Cost)]| { - let mut costs = costs.clone(); - for (class, split) in kept { - for &index in class { - let (group, candidate, _) = &members[index]; - costs.insert(&group.target, candidate, *split); - } - } - costs - }; - let selection = space.global_selection_with_candidate_costs( - cost_model, - &profiles, - horizon, - &with_shared(&shared), - )?; - let before = shared.len(); - shared.retain(|(class, _)| { - class.iter().all(|&index| { - let target = &members[index].0.target; - let chosen = selection.for_target(target).and_then(|s| s.chosen); - class.iter().any(|&other| { - Rc::ptr_eq(&members[other].0.target, target) - && chosen.is_some_and(|chosen| std::ptr::eq(chosen, members[other].1)) - }) - }) - }); - if shared.len() == before { - return Ok(selection); - } - Ok(space.global_selection_with_candidate_costs( - cost_model, - &profiles, - horizon, - &with_shared(&shared), - )?) -} - -/// Cost of one `SummaryAgg` state maintained once for every entry in -/// `demand`, or `None` when no lifecycle alternative is selectable for it. -/// No comparison target is supplied: the state serves several queries. -pub(crate) fn shared_state_cost( - state: &Rc, - demand: WorkloadDemand<'_>, - now_ms: u64, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &dyn CostModel, -) -> Result, SummaryMaintenanceLifecyclePlanError> { - Ok(enumerate_with_profile( - Rc::clone(state), - demand, - now_ms, - horizon, - capabilities, - cost_model, - None, - None, - )? - .select_cheapest() - .summary_total_cost) -} - -/// Every unique `SummaryAgg` reachable from `root`. -pub(crate) fn summary_states(root: &Rc) -> Vec> { - let mut states = Vec::new(); - collect_states( - root, - &mut HashSet::new(), - &mut states, - StateKind::SummaryAgg, - ); - states -} - -/// Assemble a globally selected phase-valid DAG and attach workload-aware -/// summary maintenance decisions. This does not create or maintain runtime state. -pub fn assemble_selected_dag_with_summary_maintenance_lifecycles( - selection: &GlobalSelection<'_>, - target: &Rc, - demand: WorkloadDemand<'_>, - now_ms: u64, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &dyn CostModel, -) -> Result, SummaryMaintenanceLifecycleAssemblyError> { - selection - .assemble_selected_dag(target)? - .map(|root| { - plan_assembled_dag( - root, - target, - demand, - now_ms, - horizon, - capabilities, - cost_model, - ) - }) - .transpose() -} - -/// The lifecycle half of -/// [`assemble_selected_dag_with_summary_maintenance_lifecycles`], for a root -/// the caller already assembled (and possibly interned across queries). -pub(crate) fn plan_assembled_dag( - root: Rc, - target: &Rc, - demand: WorkloadDemand<'_>, - now_ms: u64, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - cost_model: &dyn CostModel, -) -> Result { - let mut plan = enumerate_with_profile( - root, - demand, - now_ms, - horizon, - capabilities, - cost_model, - None, - Some(target), - )? - .select_cheapest(); - plan.raw_recompute_total_cost = plan - .expected_reads - .and_then(|reads| cost_model.raw_query_recompute_total_cost(target, reads)); - if !plan.selected_raw_recompute - && plan.raw_recompute_total_cost.is_none_or(|raw| { - plan.summary_total_cost - .is_none_or(|summary| raw.0 <= summary.0) - }) - { - plan.root = crate::replacement::keep_pre_asap(target)?; - plan.deployments.clear(); - plan.selected_raw_recompute = true; - plan.selected_window_implementation_id = None; - plan.summary_total_cost = None; - plan.window_accuracy_guarantee = None; - } - Ok(plan) -} - -fn workload_facts( - workload: &QueryWorkload, - data_workload: Option<&DataWorkload>, - workload_entry_indices: &[usize], - now_ms: u64, - horizon: Option, -) -> Result { - let mut one_time_invocations = 0u64; - let mut recurring_reads = 0.0; - let mut recurring_known = true; - let mut evaluation_rate = 0.0; - let mut has_evaluation_rate = false; - let mut prepared_start: Option = None; - let mut prepared_end: Option = None; - let mut prepared_eligible = true; - let mut requires_deletion = false; - let mut required_accuracy = Vec::new(); - - let entries: Vec<_> = workload.entries().collect(); - if workload_entry_indices.is_empty() { - return Err(SummaryMaintenanceLifecyclePlanError::EmptyWorkloadDemand); - } - let mut seen_indices = HashSet::new(); - for &index in workload_entry_indices { - if !seen_indices.insert(index) { - return Err(SummaryMaintenanceLifecyclePlanError::DuplicateWorkloadEntry { index }); - } - let entry = entries.get(index).ok_or( - SummaryMaintenanceLifecyclePlanError::InvalidWorkloadEntry { - index, - entry_count: entries.len(), - }, - )?; - required_accuracy.push(entry.requirements.accuracy.target()); - requires_deletion |= entry.time_selection.lookback.is_some() - && entry.time_selection.as_of.is_none() - && matches!( - entry.time_selection.scope, - asap_types::workload::QueryTimeScope::RealTime - | asap_types::workload::QueryTimeScope::Mixed - ); - match &entry.recurrence { - QueryRecurrence::OneTime { - invocations, - execute_at, - } => { - one_time_invocations = one_time_invocations.saturating_add(*invocations); - let covered = if let ( - Predictability::Predictable { - known_at: Some(known), - }, - Some(execute), - ) = (&entry.predictability, execute_at) - { - if known < execute && now_ms < execute.0 { - let activate = TimestampMs(known.0.max(now_ms)); - prepared_start = - Some(prepared_start.map_or(activate, |old| old.min(activate))); - prepared_end = Some(prepared_end.map_or(*execute, |old| old.max(*execute))); - true - } else { - false - } - } else { - false - }; - prepared_eligible &= covered; - } - QueryRecurrence::Repeated(RepeatedDemand::FixedInterval(interval)) - | QueryRecurrence::Repeated(RepeatedDemand::FixedIntervalAt { interval, .. }) => { - prepared_eligible = false; - let rate = 1000.0 / f64::from(interval.0); - evaluation_rate += rate; - has_evaluation_rate = true; - if let Some(h) = horizon { - recurring_reads += h.0 * rate; - } else { - recurring_known = false; - } - } - QueryRecurrence::Repeated(RepeatedDemand::Scheduled(schedule)) => { - prepared_eligible = false; - if let Some(h) = horizon { - let end_ms = now_ms.saturating_add((h.0 * 1000.0) as u64); - let reads_in_horizon = schedule - .iter() - .filter(|at| at.0 >= now_ms && at.0 <= end_ms) - .count() as f64; - recurring_reads += reads_in_horizon; - evaluation_rate += reads_in_horizon / h.0; - has_evaluation_rate = true; - } else { - recurring_known = false; - } - } - QueryRecurrence::Repeated(RepeatedDemand::EstimatedRate(estimate)) => { - prepared_eligible = false; - if !estimate.is_fresh_at(now_ms) { - recurring_known = false; - continue; - } - let rate = estimate.expected_rate.0; - evaluation_rate += rate; - has_evaluation_rate = true; - if let Some(h) = horizon { - recurring_reads += h.0 * rate; - } else { - recurring_known = false; - } - } - QueryRecurrence::Unknown => { - prepared_eligible = false; - recurring_known = false; - } - } - } - - let data = data_workload; - let arrival = data.map_or(DataArrival::Unknown, |data| data.arrival); - let update_rate = data - .and_then(|data| data.ingestion_rate.value_at(now_ms)) - .map(|rate| UpdateRate(rate.0)); - let reads = if let Some(horizon) = horizon { - let horizon_ms = horizon.0 * 1_000.0; - if !horizon_ms.is_finite() - || horizon_ms <= 0.0 - || horizon_ms > u64::MAX as f64 - || horizon_ms.fract() != 0.0 - { - None - } else { - workload_entry_indices - .iter() - .try_fold(0_u64, |total, index| { - let entry = entries.get(*index)?; - match evaluations_in_horizon(&entry.recurrence, now_ms, horizon_ms as u64) { - Ok(count) => total.checked_add(count), - Err(AnalyticalCostError::NoEvaluationsInHorizon) => Some(total), - Err(_) => None, - } - }) - .map(|count| count as f64) - } - } else { - recurring_known.then_some(one_time_invocations as f64 + recurring_reads) - }; - Ok(SummaryMaintenanceWorkloadFacts { - required_accuracy, - reads, - one_time_invocations, - evaluation_rate: has_evaluation_rate.then_some(EvaluationRate(evaluation_rate)), - update_rate, - arrival, - prepared_window: prepared_start.zip(prepared_end), - prepared_eligible, - requires_deletion, - }) -} - -fn alternatives_for( - facts: &SummaryMaintenanceWorkloadFacts, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - summary_capabilities: SummaryMaintenanceCapabilities, - costs: SummaryMaintenanceLifecycleCostInputs, -) -> Vec { - let alternatives = vec![ - ephemeral(facts, capabilities, &costs), - prepared(facts, capabilities, summary_capabilities, &costs), - shared(facts, horizon, capabilities, summary_capabilities, &costs), - continuous(facts, horizon, capabilities, summary_capabilities, &costs), - ]; - alternatives -} - -fn ephemeral( - facts: &SummaryMaintenanceWorkloadFacts, - capabilities: SummaryMaintenanceLifecycleCapabilities, - costs: &SummaryMaintenanceLifecycleCostInputs, -) -> SummaryMaintenanceLifecycleAlternative { - let lifecycle = SummaryMaintenanceLifecycle::Ephemeral; - if !capabilities.supports_ephemeral { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime, - ); - } - let total_cost = zip_costs(&[ - costs.build_cost, - costs.summary_read_cost, - costs.retirement_cost, - ]) - .zip(facts.reads) - .map(|(per_read, reads)| Cost(per_read * reads)); - costed_or_unknown( - lifecycle, - total_cost, - vec!["state is rebuilt per invocation".into()], - ) -} - -fn prepared( - facts: &SummaryMaintenanceWorkloadFacts, - capabilities: SummaryMaintenanceLifecycleCapabilities, - summary_capabilities: SummaryMaintenanceCapabilities, - costs: &SummaryMaintenanceLifecycleCostInputs, -) -> SummaryMaintenanceLifecycleAlternative { - if !facts.prepared_eligible { - return rejected( - SummaryMaintenanceLifecycle::Prepared { - activate_at: TimestampMs(0), - retire_at: TimestampMs(0), - }, - SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery, - ); - } - let Some((activate_at, retire_at)) = facts.prepared_window else { - return rejected( - SummaryMaintenanceLifecycle::Prepared { - activate_at: TimestampMs(0), - retire_at: TimestampMs(0), - }, - SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery, - ); - }; - let lifecycle = SummaryMaintenanceLifecycle::Prepared { - activate_at, - retire_at, - }; - if !capabilities.supports_prepared { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime, - ); - } - if let Some(rejection) = maintenance_capability_rejection(facts, summary_capabilities) { - return rejected(lifecycle, rejection); - } - let seconds = retire_at.0.saturating_sub(activate_at.0) as f64 / 1000.0; - let maintenance = maintenance_cost(facts, costs, seconds); - let total_cost = match ( - costs.build_cost, - costs.summary_read_cost, - costs.retention_cost_rate, - costs.retirement_cost, - maintenance, - ) { - (Some(build), Some(read), Some(retention), Some(retire), Some(maintenance)) => Some(Cost( - build.0 - + read.0 * facts.one_time_invocations as f64 - + retention.0 * seconds - + retire.0 - + maintenance, - )), - _ => None, - }; - costed_or_unknown( - lifecycle, - total_cost, - vec!["activation and retirement come from the declared schedule".into()], - ) -} - -fn shared( - facts: &SummaryMaintenanceWorkloadFacts, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - summary_capabilities: SummaryMaintenanceCapabilities, - costs: &SummaryMaintenanceLifecycleCostInputs, -) -> SummaryMaintenanceLifecycleAlternative { - let lifecycle = SummaryMaintenanceLifecycle::Shared { - retention: asap_types::workload::DurationMs(horizon.map_or(0, |h| (h.0 * 1000.0) as u64)), - }; - if !capabilities.supports_shared { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime, - ); - } - if let Some(rejection) = maintenance_capability_rejection(facts, summary_capabilities) { - return rejected(lifecycle, rejection); - } - if facts.reads.is_none_or(|reads| reads <= 1.0) { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::RequiresMultipleReads, - ); - } - let Some(horizon) = horizon else { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::RequiresHorizon, - ); - }; - let total_cost = retained_cost(facts, costs, horizon.0); - costed_or_unknown( - lifecycle, - total_cost, - vec!["one state is shared across reads".into()], - ) -} - -fn continuous( - facts: &SummaryMaintenanceWorkloadFacts, - horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, - summary_capabilities: SummaryMaintenanceCapabilities, - costs: &SummaryMaintenanceLifecycleCostInputs, -) -> SummaryMaintenanceLifecycleAlternative { - let lifecycle = SummaryMaintenanceLifecycle::ContinuouslyMaintained; - if !capabilities.supports_continuously_maintained { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime, - ); - } - if !matches!( - facts.arrival, - DataArrival::ContinuouslyIngesting | DataArrival::Mixed - ) { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::RequiresContinuousData, - ); - } - if facts.update_rate.is_none() { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::MissingOrStaleIngestionRate, - ); - } - if let Some(rejection) = maintenance_capability_rejection(facts, summary_capabilities) { - return rejected(lifecycle, rejection); - } - let Some(horizon) = horizon else { - return rejected( - lifecycle, - SummaryMaintenanceLifecycleRejection::RequiresHorizon, - ); - }; - let total_cost = retained_cost(facts, costs, horizon.0); - costed_or_unknown( - lifecycle, - total_cost, - vec!["updates are applied for the optimization horizon".into()], - ) -} - -fn maintenance_capability_rejection( - facts: &SummaryMaintenanceWorkloadFacts, - capabilities: SummaryMaintenanceCapabilities, -) -> Option { - if matches!( - facts.arrival, - DataArrival::ContinuouslyIngesting | DataArrival::Mixed - ) && !capabilities.incremental_update - { - Some(SummaryMaintenanceLifecycleRejection::SummaryDoesNotSupportIncrementalUpdates) - } else if matches!( - facts.arrival, - DataArrival::ContinuouslyIngesting | DataArrival::Mixed - ) && facts.requires_deletion - && !capabilities.delete - { - Some(SummaryMaintenanceLifecycleRejection::SummaryDoesNotSupportDeletion) - } else { - None - } -} - -fn retained_cost( - facts: &SummaryMaintenanceWorkloadFacts, - costs: &SummaryMaintenanceLifecycleCostInputs, - seconds: f64, -) -> Option { - let reads = facts.reads?; - let maintenance = maintenance_cost(facts, costs, seconds)?; - Some(Cost( - costs.build_cost?.0 - + maintenance - + reads * costs.summary_read_cost?.0 - + seconds * costs.retention_cost_rate?.0 - + costs.retirement_cost?.0, - )) -} - -fn maintenance_cost( - facts: &SummaryMaintenanceWorkloadFacts, - costs: &SummaryMaintenanceLifecycleCostInputs, - seconds: f64, -) -> Option { - match facts.arrival { - DataArrival::AtRest => Some(0.0), - DataArrival::ContinuouslyIngesting | DataArrival::Mixed => { - Some(seconds * facts.update_rate?.0 * costs.maintenance_cost_per_update?.0) - } - DataArrival::Unknown => None, - } -} - -fn zip_costs(costs: &[Option]) -> Option { - costs - .iter() - .try_fold(0.0, |sum, cost| Some(sum + cost.as_ref()?.0)) -} - -fn costed_or_unknown( - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle, - total_cost: Option, - assumptions: Vec, -) -> SummaryMaintenanceLifecycleAlternative { - SummaryMaintenanceLifecycleAlternative { - summary_maintenance_lifecycle, - total_cost, - rejection: total_cost - .is_none() - .then_some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence), - assumptions, - } -} - -fn rejected( - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle, - rejection: SummaryMaintenanceLifecycleRejection, -) -> SummaryMaintenanceLifecycleAlternative { - SummaryMaintenanceLifecycleAlternative { - summary_maintenance_lifecycle, - total_cost: None, - rejection: Some(rejection), - assumptions: Vec::new(), - } -} - -#[derive(Clone, Copy, PartialEq)] -enum StateKind { - SummaryAgg, - Population, -} - -/// Collect every unique node of `kind` reachable from `node`. -fn collect_states( - node: &Rc, - seen: &mut HashSet<*const SummaryNode>, - output: &mut Vec>, - kind: StateKind, -) { - if !seen.insert(Rc::as_ptr(node)) { - return; - } - match &node.expr { - SummaryExpr::SummaryAgg { child, .. } => { - if kind == StateKind::SummaryAgg { - output.push(Rc::clone(node)); - } - collect_states(child, seen, output, kind); - } - SummaryExpr::ValueOperation { - child, operation, .. - } => { - if kind == StateKind::Population - && matches!(operation, ValueOperation::MaintainPopulation { .. }) - { - output.push(Rc::clone(node)); - } - collect_states(child, seen, output, kind) - } - SummaryExpr::SummaryJoin { outer, inner, .. } - | SummaryExpr::RelationalJoin { - left: outer, - right: inner, - .. - } - | SummaryExpr::BinaryOp { - lhs: outer, - rhs: inner, - .. - } - | SummaryExpr::SummarySubtract { - left: outer, - right: inner, - } => { - collect_states(outer, seen, output, kind); - collect_states(inner, seen, output, kind); - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - collect_states(summary_input, seen, output, kind) - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - collect_states(child, seen, output, kind); - } - } - SummaryExpr::KeepPreAsap(_) => {} - } -} - -/// Maintained populations that are not an input of any `SummaryAgg`. A -/// population feeding summary state is on that state's maintenance path, so -/// that state's lifecycle times it, even when a readout also reads it directly. -fn standalone_populations(root: &Rc) -> Vec> { - let mut summaries = Vec::new(); - collect_states( - root, - &mut HashSet::new(), - &mut summaries, - StateKind::SummaryAgg, - ); - let mut nested = Vec::new(); - let mut seen = HashSet::new(); - for summary in &summaries { - collect_states(summary, &mut seen, &mut nested, StateKind::Population); - } - let nested: HashSet<_> = nested.iter().map(Rc::as_ptr).collect(); - let mut populations = Vec::new(); - collect_states( - root, - &mut HashSet::new(), - &mut populations, - StateKind::Population, - ); - populations.retain(|population| !nested.contains(&Rc::as_ptr(population))); - populations -} - -pub(crate) fn evaluation_schedule( - lifecycle: &SummaryMaintenanceLifecycle, - arrival: DataArrival, -) -> EvaluationSchedule { - match lifecycle { - SummaryMaintenanceLifecycle::Ephemeral => EvaluationSchedule::OneShot, - SummaryMaintenanceLifecycle::Prepared { .. } - | SummaryMaintenanceLifecycle::Shared { .. } - if matches!( - arrival, - DataArrival::ContinuouslyIngesting | DataArrival::Mixed - ) => - { - EvaluationSchedule::PerUpdate - } - SummaryMaintenanceLifecycle::Prepared { .. } => EvaluationSchedule::OneShot, - SummaryMaintenanceLifecycle::Shared { .. } => EvaluationSchedule::OnRead, - SummaryMaintenanceLifecycle::ContinuouslyMaintained => EvaluationSchedule::PerUpdate, - } -} - -/// Summary states composed on one maintenance path must be produced on the -/// same schedule. Return a component id for each collected state. -fn summary_state_components(summaries: &[Rc]) -> Vec { - let indices: HashMap<_, _> = summaries - .iter() - .enumerate() - .map(|(index, summary)| (Rc::as_ptr(summary), index)) - .collect(); - let mut parents: Vec<_> = (0..summaries.len()).collect(); - - fn find(parents: &mut [usize], index: usize) -> usize { - if parents[index] != index { - parents[index] = find(parents, parents[index]); - } - parents[index] - } - - for (parent_index, summary) in summaries.iter().enumerate() { - let SummaryExpr::SummaryAgg { child, .. } = &summary.expr else { - continue; - }; - if !matches!( - child.expr, - SummaryExpr::SummaryAgg { .. } - | SummaryExpr::SummaryJoin { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryMerge { .. } - ) { - continue; - } - let mut descendants = Vec::new(); - collect_states( - child, - &mut HashSet::new(), - &mut descendants, - StateKind::SummaryAgg, - ); - for descendant in descendants { - let child_index = indices[&Rc::as_ptr(&descendant)]; - let parent_root = find(&mut parents, parent_index); - let child_root = find(&mut parents, child_index); - parents[child_root] = parent_root; - } - } - (0..parents.len()) - .map(|index| find(&mut parents, index)) - .collect() -} - -/// Inputs shared by every complete lifecycle-combination evaluation of one -/// root, whether Planner searches combinations or a caller supplies one. -struct CompleteCostContext<'a> { - root: &'a SummaryNode, - components: &'a [usize], - cost_model: &'a dyn CostModel, - comparison_target: Option<&'a QueryExpr>, - horizon: Option, - expected_reads: Option, - required_accuracy: &'a [AccuracyTarget], -} - -impl CompleteCostContext<'_> { - /// Planner's own admission rule for one alternative. Uncosted alternatives - /// are admitted only when the complete-candidate hook is authoritative. - fn eligible(&self, alternative: &SummaryMaintenanceLifecycleAlternative) -> bool { - alternative.selectable() - || (self - .cost_model - .complete_summary_candidate_estimate_covers_lifecycle_costs() - && alternative.rejection - == Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence)) - } - - /// `selected` holds one entry per deployment, in deployment order. - fn schedules_compatible( - &self, - selected: &[(usize, SummaryMaintenanceLifecycleGuarantee, Cost)], - ) -> bool { - !selected.iter().enumerate().any(|(left, (_, a, _))| { - selected.iter().enumerate().any(|(right, (_, b, _))| { - self.components[left] == self.components[right] - && a.evaluation_schedule != b.evaluation_schedule - }) - }) - } - - fn estimate( - &self, - deployments: &[SummaryMaintenanceDeployment], - selected: &[(usize, SummaryMaintenanceLifecycleGuarantee, Cost)], - ) -> Option { - if !self.schedules_compatible(selected) { - return None; - } - let costed: Vec<_> = selected - .iter() - .map(|(index, guarantee, cost)| CostedSummaryDeployment { - summary: &deployments[*index].summary, - guarantee, - selected_cost: *cost, - }) - .collect(); - let estimate = self.cost_model.complete_summary_candidate_estimate( - self.root, - self.comparison_target, - &costed, - self.horizon, - self.expected_reads, - self.required_accuracy, - )?; - (estimate.window_frameworks.len() == deployments.len()).then_some(estimate) - } -} - -fn lifecycle_guarantee( - lifecycle: &SummaryMaintenanceLifecycle, - arrival: DataArrival, -) -> SummaryMaintenanceLifecycleGuarantee { - SummaryMaintenanceLifecycleGuarantee { - summary_maintenance_mode: maintenance_mode(lifecycle, arrival), - evaluation_schedule: evaluation_schedule(lifecycle, arrival), - summary_maintenance_lifecycle: lifecycle.clone(), - output_representation: OutputRepresentation::SummaryState, - } -} - -fn apply_selection( - deployments: &mut [SummaryMaintenanceDeployment], - guarantees: Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, - estimate: &CompleteSummaryCandidateEstimate, -) { - for (index, guarantee) in guarantees { - deployments[index].summary_maintenance_lifecycle_guarantee = Some(guarantee); - } - for (deployment, framework) in deployments - .iter_mut() - .zip(estimate.window_frameworks.iter().cloned()) - { - deployment.selected_window_framework = framework; - } -} - -#[expect(clippy::too_many_arguments, reason = "complete combination context")] -fn select_complete_lifecycle_combination( - root: &SummaryNode, - deployments: &mut [SummaryMaintenanceDeployment], - components: &[usize], - arrival: DataArrival, - cost_model: &dyn CostModel, - comparison_target: Option<&QueryExpr>, - horizon: Option, - expected_reads: Option, - required_accuracy: &[AccuracyTarget], -) -> Option { - const MAX_COMPLETE_LIFECYCLE_COMBINATIONS: usize = 4_096; - if deployments.is_empty() { - return None; - } - let context = CompleteCostContext { - root, - components, - cost_model, - comparison_target, - horizon, - expected_reads, - required_accuracy, - }; - // The whole-candidate hook is intentionally arbitrary, so partial costs - // cannot soundly prune the search. Bound exhaustive enumeration and fail - // closed instead of allowing an adversarial DAG to consume exponential - // planner time. - let combinations = deployments - .iter() - .try_fold(1_usize, |product, deployment| { - let selectable = deployment - .alternatives - .iter() - .filter(|alternative| context.eligible(alternative)) - .count(); - product.checked_mul(selectable) - })?; - if combinations == 0 || combinations > MAX_COMPLETE_LIFECYCLE_COMBINATIONS { - return None; - } - type Best = Option<( - CompleteSummaryCandidateEstimate, - Vec<(usize, SummaryMaintenanceLifecycleGuarantee)>, - )>; - fn visit( - index: usize, - context: &CompleteCostContext<'_>, - deployments: &[SummaryMaintenanceDeployment], - arrival: DataArrival, - selected: &mut Vec<(usize, SummaryMaintenanceLifecycleGuarantee, Cost)>, - best: &mut Best, - ) { - if index == deployments.len() { - let Some(estimate) = context.estimate(deployments, selected) else { - return; - }; - if best - .as_ref() - .is_none_or(|(best_estimate, _)| estimate.cost.0 < best_estimate.cost.0) - { - *best = Some(( - estimate, - selected - .iter() - .map(|(index, guarantee, _)| (*index, guarantee.clone())) - .collect(), - )); - } - return; - } - for alternative in deployments[index] - .alternatives - .iter() - .filter(|alternative| context.eligible(alternative)) - { - selected.push(( - index, - lifecycle_guarantee(&alternative.summary_maintenance_lifecycle, arrival), - alternative.total_cost.unwrap_or(Cost::ZERO), - )); - visit(index + 1, context, deployments, arrival, selected, best); - selected.pop(); - } - } - - let mut best = None; - visit( - 0, - &context, - deployments, - arrival, - &mut Vec::new(), - &mut best, - ); - let (estimate, guarantees) = best?; - apply_selection(deployments, guarantees, &estimate); - Some(estimate) -} - -pub(crate) fn maintenance_mode( - lifecycle: &SummaryMaintenanceLifecycle, - arrival: DataArrival, -) -> SummaryMaintenanceMode { - match lifecycle { - SummaryMaintenanceLifecycle::Ephemeral => SummaryMaintenanceMode::DirectBuild, - SummaryMaintenanceLifecycle::ContinuouslyMaintained => SummaryMaintenanceMode::Incremental, - SummaryMaintenanceLifecycle::Prepared { .. } - | SummaryMaintenanceLifecycle::Shared { .. } => match arrival { - DataArrival::ContinuouslyIngesting | DataArrival::Mixed => { - SummaryMaintenanceMode::Incremental - } - DataArrival::AtRest | DataArrival::Unknown => SummaryMaintenanceMode::DirectBuild, - }, - } -} - -#[cfg(test)] -mod tests { - // Independent data evidence must be validated at both planning boundaries. - #[test] - fn rejects_invalid_parallel_data_evidence() { - let query = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let space = crate::replacement::search_workload(vec![("q", quantile_query())]); - for rate in [1.0, -1.0, f64::NAN, f64::INFINITY] { - let mut data = at_rest(); - data.ingestion_rate.value = Some(Rate(rate)); - assert!(space - .recurrence_profiles_from_workload(&query, Some(&data), &[0], 0, None) - .is_err()); - assert!(plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data(&query, &data, &[0]), - 0, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &crate::cost_model::DefaultCostModel, - ) - .is_err()); - } - } - use super::*; - use asap_types::post_asap::{ - ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, PostAsapOperatorPayload, - ResultGuarantee, Schema, SketchAlgorithm, - }; - use asap_types::pre_asap::AggIntent; - use asap_types::pre_asap::{ColumnRef, DataType, QueryExpr, Reduction, Source}; - use asap_types::types::AccuracyTarget; - use asap_types::workload::{ - BatchEntry, DataWorkload, DurationMs, Evidence, EvidenceSource, Predictability, Query, - QueryLanguage, QueryRequirements, Rate, RepeatingEntry, RepetitionInterval, TimeSelection, - }; - - struct UnitCosts; - - impl CostModel for UnitCosts { - fn rank_candidates( - &self, - _intent: &asap_types::pre_asap::AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(Cost(10.0)), - maintenance_cost_per_update: Some(Cost(1.0)), - summary_read_cost: Some(Cost(1.0)), - retention_cost_rate: Some(CostRate(0.1)), - retirement_cost: Some(Cost(1.0)), - } - } - - fn summary_maintenance_capabilities( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - SummaryMaintenanceCapabilities { - incremental_update: true, - merge: true, - delete: true, - } - } - } - - struct RawCheaper; - - impl CostModel for RawCheaper { - fn rank_candidates( - &self, - _intent: &asap_types::pre_asap::AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) - } - - fn summary_maintenance_capabilities( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - UnitCosts.summary_maintenance_capabilities(summary) - } - - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { - Some(Cost(1.0)) - } - } - - struct NoDelete; - - impl CostModel for NoDelete { - fn rank_candidates( - &self, - _intent: &asap_types::pre_asap::AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - UnitCosts.summary_maintenance_lifecycle_cost_inputs(summary) - } - - fn summary_maintenance_capabilities( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - SummaryMaintenanceCapabilities { - incremental_update: true, - merge: true, - delete: false, - } - } - } - - struct SummaryMaintenancePrefersDdSketch; - - impl CostModel for SummaryMaintenancePrefersDdSketch { - fn raw_query_recompute_total_cost( - &self, - _target: &QueryExpr, - _expected_reads: f64, - ) -> Option { - Some(Cost(1_000.0)) - } - - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - // Preserve semantic mapping's KLL-first order. The lifecycle - // total below must be what changes the final choice. - candidates.to_vec() - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - let build = match sketch_algorithm(summary) { - Some(SketchAlgorithm::Kll) => 100.0, - Some(SketchAlgorithm::DDSketch) => 1.0, - _ => 10.0, - }; - SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(Cost(build)), - maintenance_cost_per_update: Some(Cost(1.0)), - summary_read_cost: Some(Cost(1.0)), - retention_cost_rate: Some(CostRate(0.1)), - retirement_cost: Some(Cost(1.0)), - } - } - } - - struct IncompatibleNestedCosts; - - struct WholeCandidatePrefersContinuous; - - impl CostModel for WholeCandidatePrefersContinuous { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn complete_summary_candidate_cost( - &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, - deployments: &[CostedSummaryDeployment<'_>], - _horizon: Option, - _expected_reads: Option, - _required_accuracy: &[AccuracyTarget], - ) -> Option { - Some( - if deployments.iter().all(|deployment| { - matches!( - deployment.guarantee.summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::ContinuouslyMaintained - ) - }) { - Cost(1.0) - } else { - Cost(100.0) - }, - ) - } - } - - impl CostModel for IncompatibleNestedCosts { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - let is_leaf = matches!( - summary.expr, - SummaryExpr::SummaryAgg { ref child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)) - ); - SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(Cost(if is_leaf { 1.0 } else { 100.0 })), - maintenance_cost_per_update: Some(Cost(if is_leaf { 100.0 } else { 0.0 })), - summary_read_cost: Some(Cost::ZERO), - retention_cost_rate: Some(CostRate(0.0)), - retirement_cost: Some(Cost::ZERO), - } - } - - fn summary_maintenance_capabilities( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - SummaryMaintenanceCapabilities { - incremental_update: true, - merge: true, - delete: true, - } - } - } - - fn sketch_algorithm(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_algorithm(summary_input), - SummaryExpr::SummaryAgg { - family: FieldDataType::Sketch(kind, _), - .. - } => Some(kind.algorithm().clone()), - _ => None, - } - } - - fn query_root() -> Rc { - query_root_for("m") - } - - fn query_root_for(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - }) - } - - fn sum_query() -> Rc { - Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: query_root(), - }) - } - - fn quantile_query() -> Rc { - Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Quantile { - col: None, - q: 0.99, - accuracy: AccuracyTarget::Epsilon(0.1), - }], - output_names: vec![], - filters: vec![], - having: None, - child: query_root(), - }) - } - - fn summary() -> Rc { - let child = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(query_root()), - schema: Schema::lifted(vec![], None), - guarantee: Some(ResultGuarantee::exact("raw")), - }); - let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( - "value".into(), - )), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: Some(ResultGuarantee::exact("sum")), - }) - } - - fn nested_summary() -> Rc { - let child = summary(); - let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( - "state".into(), - )), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: Some(ResultGuarantee::exact("nested sum")), - }) - } - - fn batch(predictability: Predictability) -> BatchEntry { - BatchEntry { - query: Query("sum(m)".into()), - requirements: QueryRequirements::default(), - predictability, - invocations: 1, - execute_at: None, - time_selection: TimeSelection::default(), - } - } - - fn workload( - batches: Vec, - repeating: Vec, - _data: DataWorkload, - ) -> QueryWorkload { - QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: (!batches.is_empty()).then_some(batches), - repeating_queries: (!repeating.is_empty()).then_some(repeating), - } - } - - fn at_rest() -> DataWorkload { - DataWorkload { - arrival: DataArrival::AtRest, - data_ingestion_interval: Evidence { - value: Some(DurationMs(1_000)), - ..Default::default() - }, - ..Default::default() - } - } - - fn continuous(observed_at_ms: u64, valid_for_ms: u64) -> DataWorkload { - DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - data_ingestion_interval: Evidence { - value: Some(DurationMs(1_000)), - ..Default::default() - }, - ingestion_rate: Evidence { - value: Some(Rate(1.0)), - source: EvidenceSource::Observed, - observed_at_ms: Some(observed_at_ms), - valid_for_ms: Some(valid_for_ms), - }, - ..Default::default() - } - } - - fn repeating() -> RepeatingEntry { - RepeatingEntry { - query: Query("sum(m)".into()), - demand: RepeatedDemand::FixedInterval(RepetitionInterval(1_000)), - requirements: QueryRequirements::default(), - predictability: Predictability::Predictable { known_at: None }, - time_selection: TimeSelection::default(), - } - } - - fn selected_summary_maintenance_lifecycle( - deployment: &SummaryMaintenanceDeployment, - ) -> Option<&SummaryMaintenanceLifecycle> { - deployment - .summary_maintenance_lifecycle_guarantee - .as_ref() - .map(|guarantee| &guarantee.summary_maintenance_lifecycle) - } - - #[test] - fn fixed_interval_reads_use_the_physical_horizon_multiplicity() { - let mut query = repeating(); - query.demand = RepeatedDemand::FixedInterval(RepetitionInterval(600)); - let workload = workload(vec![], vec![query], at_rest()); - - let facts = - workload_facts(&workload, Some(&at_rest()), &[0], 0, Some(Horizon(1.0))).unwrap(); - - assert_eq!(facts.reads, Some(1.0)); - } - - #[test] - fn unpredictable_one_time_at_rest_selects_ephemeral() { - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data( - &workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()), - &[0], - ), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!(plan.deployments.len(), 1); - assert_eq!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(&SummaryMaintenanceLifecycle::Ephemeral) - ); - let guarantee = plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap(); - assert_eq!(guarantee.evaluation_schedule, EvaluationSchedule::OneShot); - assert_eq!( - guarantee.summary_maintenance_mode, - SummaryMaintenanceMode::DirectBuild - ); - assert_eq!( - guarantee.output_representation, - OutputRepresentation::SummaryState - ); - assert_eq!( - plan.deployments[0].alternatives[0].total_cost, - Some(Cost(12.0)) - ); - } - - #[test] - fn predictable_scheduled_one_time_offers_prepared_state() { - let mut entry = batch(Predictability::Predictable { - known_at: Some(TimestampMs(1_000)), - }); - entry.execute_at = Some(TimestampMs(11_000)); - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data( - &workload(vec![entry], vec![], at_rest()), - &at_rest(), - &[0], - ), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - let prepared = &plan.deployments[0].alternatives[1]; - assert!(prepared.rejection.is_none()); - assert_eq!(prepared.total_cost, Some(Cost(13.0))); - } - - #[test] - fn prepared_state_starts_no_earlier_than_planning_time() { - let mut entry = batch(Predictability::Predictable { - known_at: Some(TimestampMs(1_000)), - }); - entry.execute_at = Some(TimestampMs(11_000)); - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data( - &workload(vec![entry], vec![], at_rest()), - &at_rest(), - &[0], - ), - 6_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - let prepared = &plan.deployments[0].alternatives[1]; - assert_eq!( - prepared.summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::Prepared { - activate_at: TimestampMs(6_000), - retire_at: TimestampMs(11_000), - } - ); - assert_eq!(prepared.total_cost, Some(Cost(12.5))); - } - - #[test] - fn expired_one_time_execution_cannot_select_prepared_state() { - let mut entry = batch(Predictability::Predictable { - known_at: Some(TimestampMs(1_000)), - }); - entry.execute_at = Some(TimestampMs(2_000)); - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data(&workload(vec![entry], vec![], at_rest()), &[0]), - 3_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!( - plan.deployments[0].alternatives[1].rejection, - Some(SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery) - ); - } - - #[test] - fn nested_summary_lifecycles_have_compatible_evaluation_schedules() { - let workload = workload(vec![], vec![repeating()], continuous(1_000, 20_000)); - let plan = plan_summary_maintenance_lifecycles( - nested_summary(), - WorkloadDemand::new_without_data(&workload, &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &IncompatibleNestedCosts, - ) - .unwrap(); - - assert_eq!(plan.deployments.len(), 2); - let schedules: HashSet<_> = plan - .deployments - .iter() - .map(|deployment| { - deployment - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap() - .evaluation_schedule - }) - .collect(); - assert_eq!(schedules.len(), 1); - } - - #[test] - fn repeated_at_rest_selects_shared_without_inventing_updates() { - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data( - &workload(vec![], vec![repeating()], at_rest()), - &at_rest(), - &[0], - ), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(&SummaryMaintenanceLifecycle::Shared { - retention: DurationMs(10_000) - }) - ); - assert_eq!( - plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap() - .summary_maintenance_mode, - SummaryMaintenanceMode::DirectBuild - ); - assert_eq!( - plan.deployments[0].alternatives[3].rejection, - Some(SummaryMaintenanceLifecycleRejection::RequiresContinuousData) - ); - assert_eq!(plan.update_rate, None); - } - - #[test] - fn repeated_continuous_workload_can_select_continuous_maintenance() { - let capabilities = SummaryMaintenanceLifecycleCapabilities { - supports_shared: false, - ..SummaryMaintenanceLifecycleCapabilities::ALL - }; - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data( - &workload(vec![], vec![repeating()], continuous(1_000, 60_000)), - &continuous(1_000, 60_000), - &[0], - ), - 1_000, - Some(Horizon(10.0)), - capabilities, - &UnitCosts, - ) - .unwrap(); - assert_eq!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(&SummaryMaintenanceLifecycle::ContinuouslyMaintained) - ); - assert_eq!( - plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap() - .summary_maintenance_mode, - SummaryMaintenanceMode::Incremental - ); - assert_eq!(plan.evaluation_rate, Some(EvaluationRate(1.0))); - assert_eq!(plan.update_rate, Some(UpdateRate(1.0))); - } - - #[test] - fn stale_ingestion_evidence_cannot_enable_continuous_maintenance() { - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data( - &workload(vec![], vec![repeating()], continuous(1_000, 1_000)), - &continuous(1_000, 1_000), - &[0], - ), - 3_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!( - plan.deployments[0].alternatives[3].rejection, - Some(SummaryMaintenanceLifecycleRejection::MissingOrStaleIngestionRate) - ); - assert_eq!(plan.update_rate, None); - } - - #[test] - fn unknown_costs_do_not_make_a_long_lived_lifecycle_win() { - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data( - &workload(vec![], vec![repeating()], continuous(1_000, 60_000)), - &[0], - ), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &crate::cost_model::DefaultCostModel, - ) - .unwrap(); - assert_eq!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - None - ); - assert!(plan.deployments[0] - .alternatives - .iter() - .all(|alternative| alternative.rejection.is_some())); - } - - #[test] - fn unrelated_workload_entries_do_not_create_reuse_for_a_target() { - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data( - &workload( - vec![batch(Predictability::AdHoc), batch(Predictability::AdHoc)], - vec![], - at_rest(), - ), - &[0], - ), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(&SummaryMaintenanceLifecycle::Ephemeral) - ); - assert_eq!( - plan.deployments[0].alternatives[2].rejection, - Some(SummaryMaintenanceLifecycleRejection::RequiresMultipleReads) - ); - } - - #[test] - fn scheduled_rate_counts_only_executions_inside_the_horizon() { - let mut entry = repeating(); - entry.demand = RepeatedDemand::Scheduled(vec![ - TimestampMs(999), - TimestampMs(5_000), - TimestampMs(20_000), - ]); - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data(&workload(vec![], vec![entry], at_rest()), &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!(plan.evaluation_rate, Some(EvaluationRate(0.1))); - } - - #[test] - fn demand_binding_rejects_empty_and_duplicate_entries() { - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - assert!(matches!( - plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data(&workload, &[]), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ), - Err(SummaryMaintenanceLifecyclePlanError::EmptyWorkloadDemand) - )); - assert!(matches!( - plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data(&workload, &[0, 0]), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ), - Err(SummaryMaintenanceLifecyclePlanError::DuplicateWorkloadEntry { index: 0 }) - )); - } - - #[test] - fn prepared_requires_every_bound_consumer_to_be_scheduled_and_predictable() { - let mut predictable = batch(Predictability::Predictable { - known_at: Some(TimestampMs(1_000)), - }); - predictable.execute_at = Some(TimestampMs(2_000)); - let workload = workload( - vec![predictable, batch(Predictability::AdHoc)], - vec![], - at_rest(), - ); - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_without_data(&workload, &[0, 1]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!( - plan.deployments[0].alternatives[1].rejection, - Some(SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery) - ); - } - - #[test] - fn moving_realtime_maintenance_requires_summary_deletion_support() { - let mut entry = repeating(); - entry.time_selection = TimeSelection { - scope: asap_types::workload::QueryTimeScope::RealTime, - lookback: Some(DurationMs(60_000)), - as_of: None, - }; - let plan = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data( - &workload(vec![], vec![entry], continuous(1_000, 60_000)), - &continuous(1_000, 60_000), - &[0], - ), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &NoDelete, - ) - .unwrap(); - assert_eq!( - plan.deployments[0].alternatives[3].rejection, - Some(SummaryMaintenanceLifecycleRejection::SummaryDoesNotSupportDeletion) - ); - } - - #[test] - fn lifecycle_cost_can_fall_back_to_raw_recomputation() { - let target = sum_query(); - let space = crate::replacement::search_workload(vec![("q", Rc::clone(&target))]); - let selection = space.global_selection(&RawCheaper); - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &space.roots[0].1, - WorkloadDemand::new_without_data(&workload, &[0]), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &RawCheaper, - ) - .unwrap() - .unwrap(); - assert!(plan.selected_raw_recompute); - assert_eq!(plan.raw_recompute_total_cost, Some(Cost(1.0))); - assert_eq!(plan.summary_total_cost, None); - assert!(plan.deployments.is_empty()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); - - let exported = - crate::summary_maintenance_dag_export::export_summary_maintenance_plan(&plan); - assert!(exported.selected_raw_recompute); - assert_eq!(exported.raw_recompute_total_cost, Some(1.0)); - assert_eq!(exported.summary_total_cost, None); - assert!(exported.deployments.is_empty()); - } - - #[test] - fn whole_candidate_cost_is_evaluated_before_selecting_a_lifecycle() { - let root = summary(); - let mut deployments = vec![SummaryMaintenanceDeployment { - post_asap_node_id: PostAsapNodeId(0), - summary: Rc::clone(&root), - summary_maintenance_lifecycle_guarantee: None, - selected_window_framework: None, - alternatives: vec![ - SummaryMaintenanceLifecycleAlternative { - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle::Ephemeral, - total_cost: Some(Cost(1.0)), - rejection: None, - assumptions: vec![], - }, - SummaryMaintenanceLifecycleAlternative { - summary_maintenance_lifecycle: - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - total_cost: Some(Cost(10.0)), - rejection: None, - assumptions: vec![], - }, - ], - }]; - - let total = select_complete_lifecycle_combination( - &root, - &mut deployments, - &[0], - DataArrival::ContinuouslyIngesting, - &WholeCandidatePrefersContinuous, - None, - Some(Horizon(10.0)), - Some(2.0), - &[], - ); - - assert_eq!(total.map(|estimate| estimate.cost), Some(Cost(1.0))); - assert!(matches!( - deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap() - .summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::ContinuouslyMaintained - )); - } - - #[test] - fn complete_lifecycle_enumeration_fails_closed_above_safe_bound() { - let root = summary(); - let alternatives = vec![ - SummaryMaintenanceLifecycleAlternative { - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle::Ephemeral, - total_cost: Some(Cost(1.0)), - rejection: None, - assumptions: vec![], - }, - SummaryMaintenanceLifecycleAlternative { - summary_maintenance_lifecycle: SummaryMaintenanceLifecycle::ContinuouslyMaintained, - total_cost: Some(Cost(2.0)), - rejection: None, - assumptions: vec![], - }, - ]; - let mut deployments: Vec<_> = (0..13) - .map(|summary_index| SummaryMaintenanceDeployment { - post_asap_node_id: PostAsapNodeId(summary_index as u32), - summary: Rc::clone(&root), - summary_maintenance_lifecycle_guarantee: None, - selected_window_framework: None, - alternatives: alternatives.clone(), - }) - .collect(); - assert_eq!( - select_complete_lifecycle_combination( - &root, - &mut deployments, - &(0..13).collect::>(), - DataArrival::ContinuouslyIngesting, - &WholeCandidatePrefersContinuous, - None, - Some(Horizon(10.0)), - Some(2.0), - &[], - ), - None - ); - assert!(deployments - .iter() - .all(|deployment| deployment.summary_maintenance_lifecycle_guarantee.is_none())); - } - - #[test] - fn materialization_falls_back_to_raw_when_raw_cost_is_unavailable() { - let target = quantile_query(); - let space = crate::replacement::search_workload(vec![("q", target)]); - let selection = space.global_selection(&UnitCosts); - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &space.roots[0].1, - WorkloadDemand::new_without_data(&workload, &[0]), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap() - .unwrap(); - assert!(plan.selected_raw_recompute); - assert!(plan.raw_recompute_total_cost.is_none()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); - } - - #[test] - fn unmatched_target_is_reported_as_raw_recomputation() { - let target = query_root(); - let space = crate::replacement::search_workload(vec![("q", Rc::clone(&target))]); - let selection = space.global_selection(&RawCheaper); - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &space.roots[0].1, - WorkloadDemand::new_without_data(&workload, &[0]), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &RawCheaper, - ) - .unwrap() - .unwrap(); - - assert!(plan.selected_raw_recompute); - assert_eq!(plan.raw_recompute_total_cost, Some(Cost(1.0))); - assert_eq!(plan.summary_total_cost, None); - assert!(plan.deployments.is_empty()); - assert!(matches!(plan.root.expr, SummaryExpr::KeepPreAsap(_))); - } - - #[test] - fn lifecycle_cost_reorders_semantic_summary_candidates_before_materialization() { - let target = quantile_query(); - let space = crate::replacement::search_workload(vec![("q", target)]); - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand::new_with_data(&workload, &at_rest(), &[0]), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &SummaryMaintenancePrefersDdSketch, - ) - .unwrap(); - let materialized = selection - .assemble_selected_dag(&space.roots[0].1) - .unwrap() - .unwrap(); - - assert_eq!( - sketch_algorithm(&materialized), - Some(SketchAlgorithm::DDSketch) - ); - } - - #[test] - fn lifecycle_cost_counts_one_shared_summary_node_once() { - let shared = summary(); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&shared), Rc::clone(&shared)], - }, - schema: shared.schema.clone(), - guarantee: None, - }); - let workload = workload( - vec![batch(Predictability::AdHoc), batch(Predictability::AdHoc)], - vec![], - at_rest(), - ); - let horizon = Some(Horizon(10.0)); - let plan = plan_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &at_rest(), &[0, 1]), - 1_000, - horizon, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!(plan.deployments.len(), 1); - assert!(matches!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(SummaryMaintenanceLifecycle::Shared { .. }) - )); - } - - /// A state costs 10 however often it is read. Recomputing p50 raw costs - /// 1 and p99 costs 8. - struct P50PrefersRaw; - - impl CostModel for P50PrefersRaw { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(Cost(10.0)), - maintenance_cost_per_update: Some(Cost::ZERO), - summary_read_cost: Some(Cost::ZERO), - retention_cost_rate: Some(CostRate(0.0)), - retirement_cost: Some(Cost::ZERO), - } - } - - fn summary_maintenance_capabilities( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - UnitCosts.summary_maintenance_capabilities(summary) - } - - fn raw_query_recompute_total_cost( - &self, - target: &QueryExpr, - _expected_reads: f64, - ) -> Option { - match target { - QueryExpr::Aggregate { measures, .. } => match measures[..] { - [AggIntent::Quantile { q: 0.5, .. }] => Some(Cost(1.0)), - _ => Some(Cost(8.0)), - }, - _ => None, - } - } - } - - /// p50 and p99 form a sharing class over one state (5 each), but p50's - /// raw recompute (1) still wins. The class reverts, so p99 is reselected - /// at its independent cost (10) and recomputes raw (8), as it does alone. - /// Checked at selection: the assembled plan's own raw comparison would - /// recompute p99 raw either way. - #[test] - fn sharing_class_reverts_when_a_member_selects_elsewhere() { - let quantile = |q| { - Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Quantile { - col: None, - q, - accuracy: AccuracyTarget::Epsilon(0.1), - }], - // A shared output name keeps p50 and p99 on one state. - output_names: vec!["value".into()], - filters: vec![], - having: None, - child: query_root(), - }) - }; - let workload = workload(vec![], vec![repeating(), repeating()], at_rest()); - // Whether each root selected a summary rather than raw recompute. - let summaries = |space: &CandidateLogicalASAPDAGs<&str>, entries: &[usize]| { - let selection = global_selection_with_summary_maintenance_lifecycles( - space, - WorkloadDemand::new_with_data(&workload, &at_rest(), entries), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &P50PrefersRaw, - ) - .unwrap(); - space - .roots - .iter() - .map(|(_, target)| selection.for_target(target).unwrap().chosen.is_some()) - .collect::>() - }; - - let space = crate::replacement::search_workload(vec![ - ("p50", quantile(0.5)), - ("p99", quantile(0.99)), - ]); - let alone = crate::replacement::search_workload(vec![("p99", quantile(0.99))]); - assert_eq!(summaries(&space, &[0, 1]), vec![false, false]); - assert_eq!(summaries(&alone, &[1]), vec![false]); - } - - #[test] - fn normalized_workload_drives_candidate_logical_asap_dags_recurrence_profiles() { - let root = query_root(); - let space = crate::replacement::search_workload(vec![("dashboard", Rc::clone(&root))]); - let workload = workload(vec![], vec![repeating()], continuous(1_000, 60_000)); - let profiles = space - .recurrence_profiles_from_workload( - &workload, - Some(&continuous(1_000, 60_000)), - &[0], - 1_000, - Some(Horizon(10.0)), - ) - .unwrap(); - // `search_workload` canonicalizes roots through CSE; recurrence - // profiles are keyed by that canonical post-CSE node. - let profile = profiles.for_target(&space.roots[0].1); - assert_eq!(profile.evaluation_rate, Some(EvaluationRate(1.0))); - assert_eq!(profile.update_rate, Some(UpdateRate(1.0))); - assert_eq!(profile.one_shot_consumers, 0); - } - - #[test] - fn recurrence_binding_is_explicit_when_root_order_differs_from_workload_order() { - let repeating_root = query_root_for("dashboard"); - let batch_root = query_root_for("batch"); - let space = crate::replacement::search_workload(vec![ - ("dashboard", repeating_root), - ("batch", batch_root), - ]); - let workload = workload( - vec![batch(Predictability::AdHoc)], - vec![repeating()], - at_rest(), - ); - let profiles = space - .recurrence_profiles_from_workload(&workload, None, &[1, 0], 1_000, Some(Horizon(10.0))) - .unwrap(); - let dashboard = profiles.for_target(&space.roots[0].1); - let batch = profiles.for_target(&space.roots[1].1); - assert_eq!(dashboard.evaluation_rate, Some(EvaluationRate(1.0))); - assert_eq!(dashboard.one_shot_consumers, 0); - assert_eq!(batch.evaluation_rate, None); - assert_eq!(batch.one_shot_consumers, 1); - } - - fn continuous_candidates<'a>( - workload: &QueryWorkload, - data: &DataWorkload, - model: &'a dyn CostModel, - ) -> SummaryMaintenanceLifecycleCandidates<'a> { - enumerate_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data(workload, data, &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities { - supports_shared: false, - ..SummaryMaintenanceLifecycleCapabilities::ALL - }, - model, - ) - .unwrap() - } - - fn choose( - candidates: &SummaryMaintenanceLifecycleCandidates<'_>, - lifecycle: SummaryMaintenanceLifecycle, - ) -> Vec<(PostAsapNodeId, SummaryMaintenanceLifecycle)> { - candidates - .deployments() - .iter() - .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) - .collect() - } - - // Enumeration reports all four lifecycle kinds with their rejections and - // selects nothing. - #[test] - fn enumeration_exposes_every_lifecycle_without_selecting() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let candidates = continuous_candidates(&workload, &data, &UnitCosts); - let [deployment] = candidates.deployments() else { - panic!("one summary state"); - }; - assert_eq!(deployment.summary_maintenance_lifecycle_guarantee, None); - assert_eq!(deployment.selected_window_framework, None); - let outcome: Vec<_> = deployment - .alternatives - .iter() - .map(|alternative| { - ( - &alternative.summary_maintenance_lifecycle, - alternative.rejection.clone(), - alternative.total_cost.is_some(), - ) - }) - .collect(); - assert!(matches!( - outcome.as_slice(), - [ - (SummaryMaintenanceLifecycle::Ephemeral, None, true), - ( - SummaryMaintenanceLifecycle::Prepared { .. }, - Some(SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery), - false - ), - ( - SummaryMaintenanceLifecycle::Shared { .. }, - Some(SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime), - false - ), - ( - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - None, - true - ), - ] - )); - let guarantee = candidates.guarantee(&SummaryMaintenanceLifecycle::ContinuouslyMaintained); - assert_eq!( - guarantee.summary_maintenance_mode, - SummaryMaintenanceMode::Incremental - ); - assert_eq!(guarantee.evaluation_schedule, EvaluationSchedule::PerUpdate); - } - - // Explicitly choosing Planner's own selection reproduces Planner's plan. - #[test] - fn explicit_choice_of_planner_selection_reproduces_planner_plan() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let planned = plan_summary_maintenance_lifecycles( - summary(), - WorkloadDemand::new_with_data(&workload, &data, &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities { - supports_shared: false, - ..SummaryMaintenanceLifecycleCapabilities::ALL - }, - &UnitCosts, - ) - .unwrap(); - let candidates = continuous_candidates(&workload, &data, &UnitCosts); - let choice = choose( - &candidates, - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - ); - let chosen = candidates.select(&choice).unwrap(); - assert_eq!(format!("{chosen:?}"), format!("{planned:?}")); - } - - // A deployment may bind a legal alternative Planner's estimate does not - // prefer; the plan carries that alternative's guarantee and cost. - #[test] - fn explicit_choice_may_bind_a_costlier_legal_alternative() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let candidates = continuous_candidates(&workload, &data, &UnitCosts); - let ephemeral_cost = candidates.deployments()[0].alternatives[0].total_cost; - let choice = choose(&candidates, SummaryMaintenanceLifecycle::Ephemeral); - let plan = candidates.select(&choice).unwrap(); - assert_eq!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(&SummaryMaintenanceLifecycle::Ephemeral) - ); - assert_eq!(plan.summary_total_cost, ephemeral_cost); - } - - // Choices that Planner could not select, or that do not cover exactly the - // enumerated states, are refused rather than bound. - #[test] - fn explicit_choice_rejects_illegal_or_incomplete_choices() { - use SummaryMaintenanceLifecycleChoiceError as E; - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let select = |model: &dyn CostModel, choice: &dyn Fn(PostAsapNodeId) -> Vec<_>| { - let candidates = continuous_candidates(&workload, &data, model); - let id = candidates.deployments()[0].post_asap_node_id; - (id, candidates.select(&choice(id)).unwrap_err()) - }; - let shared = SummaryMaintenanceLifecycle::Shared { - retention: DurationMs(10_000), - }; - let (id, error) = select(&UnitCosts, &|id| vec![(id, shared.clone())]); - assert_eq!( - error, - E::Rejected { - post_asap_node_id: id, - rejection: Some(SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime), - } - ); - let continuous = SummaryMaintenanceLifecycle::ContinuouslyMaintained; - let (id, error) = select(&crate::cost_model::DefaultCostModel, &|id| { - vec![(id, SummaryMaintenanceLifecycle::Ephemeral)] - }); - assert_eq!( - error, - E::Rejected { - post_asap_node_id: id, - rejection: Some(SummaryMaintenanceLifecycleRejection::MissingCostEvidence), - } - ); - let (id, error) = select(&UnitCosts, &|id| { - vec![( - id, - SummaryMaintenanceLifecycle::Shared { - retention: DurationMs(1), - }, - )] - }); - assert_eq!(error, E::NotAnAlternative(id)); - let (id, error) = select(&UnitCosts, &|_| vec![]); - assert_eq!(error, E::MissingChoice(id)); - let (id, error) = select(&UnitCosts, &|id| { - vec![(id, continuous.clone()), (id, continuous.clone())] - }); - assert_eq!(error, E::DuplicateChoice(id)); - let (_, error) = select(&UnitCosts, &|_| { - vec![(PostAsapNodeId(u32::MAX), continuous.clone())] - }); - assert_eq!(error, E::UnknownSummary(PostAsapNodeId(u32::MAX))); - } - - // Nested states on one maintenance path must share an evaluation schedule. - #[test] - fn explicit_choice_rejects_incompatible_nested_schedules() { - let workload = workload(vec![], vec![repeating()], continuous(1_000, 20_000)); - let candidates = enumerate_summary_maintenance_lifecycles( - nested_summary(), - WorkloadDemand::new_with_data(&workload, &continuous(1_000, 20_000), &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &IncompatibleNestedCosts, - ) - .unwrap(); - let [outer, inner] = candidates.deployments() else { - panic!("two summary states"); - }; - let choice = vec![ - ( - outer.post_asap_node_id, - SummaryMaintenanceLifecycle::Ephemeral, - ), - ( - inner.post_asap_node_id, - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - ), - ]; - assert_eq!( - candidates.select(&choice).unwrap_err(), - SummaryMaintenanceLifecycleChoiceError::IncompatibleEvaluationSchedules - ); - } - - // A multi-summary root yields one candidate entry per unique state, with - // a shared `Rc` state listed once. - #[test] - fn enumeration_lists_each_unique_summary_state_once() { - let shared = summary(); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - children: vec![Rc::clone(&shared), Rc::clone(&shared), summary()], - }, - schema: shared.schema.clone(), - guarantee: None, - }); - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let candidates = enumerate_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(&workload, &at_rest(), &[0]), - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - let ids: HashSet<_> = candidates - .deployments() - .iter() - .map(|deployment| deployment.post_asap_node_id) - .collect(); - assert_eq!(candidates.deployments().len(), 2); - assert_eq!(ids.len(), 2); - assert!(candidates - .deployments() - .iter() - .any(|deployment| Rc::ptr_eq(&deployment.summary, &shared))); - } - - fn readout(state: &Rc) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: Rc::clone(state), - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::QueryTime, - }, - schema: Schema { - closed: true, - unique_keys: vec![], - fields: vec![Field { - table: None, - name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), - nullable: false, - }], - time_index: None, - }, - guarantee: Some(ResultGuarantee::exact("sum")), - }) - } - - fn lifecycle_matching( - alternatives: &[SummaryMaintenanceLifecycleAlternative], - kind: fn(&SummaryMaintenanceLifecycle) -> bool, - ) -> SummaryMaintenanceLifecycle { - alternatives - .iter() - .map(|alternative| &alternative.summary_maintenance_lifecycle) - .find(|lifecycle| kind(lifecycle)) - .expect("lifecycle kind is an alternative") - .clone() - } - - /// Bind the lifecycle `choose` picks for every state of `root`, then - /// derive the timed DAG. - fn timed_dag( - root: Rc, - workload: &QueryWorkload, - data: &DataWorkload, - horizon: Option, - choose: impl Fn(&SummaryMaintenanceDeployment) -> SummaryMaintenanceLifecycle, - ) -> PostAsapDAG { - let candidates = enumerate_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data(workload, data, &[0]), - 1_000, - horizon, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - let choice: Vec<_> = candidates - .deployments() - .iter() - .map(|deployment| (deployment.post_asap_node_id, choose(deployment))) - .collect(); - let dag = candidates - .select(&choice) - .unwrap() - .execution_timed_dag() - .unwrap(); - dag.validate().unwrap(); - dag - } - - /// Operator kinds in node-id order, each paired with its timing. - fn timings(dag: &PostAsapDAG) -> Vec<(&'static str, ExecutionTiming)> { - dag.nodes - .iter() - .map(|node| { - let kind = match node.payload { - PostAsapOperatorPayload::Fallback { .. } => "raw", - PostAsapOperatorPayload::SummaryAgg { .. } => "state", - PostAsapOperatorPayload::Value { .. } => "readout", - PostAsapOperatorPayload::Binary { .. } => "binary", - _ => "other", - }; - (kind, node.output_state.timing) - }) - .collect() - } - - const INGEST: ExecutionTiming = ExecutionTiming::IngestionTime; - const QUERY: ExecutionTiming = ExecutionTiming::QueryTime; - - // Every retained lifecycle kind runs its state and inputs at ingestion - // time and its readout at query time. - #[test] - fn retained_lifecycles_time_state_and_inputs_at_ingestion() { - let mut scheduled = batch(Predictability::Predictable { - known_at: Some(TimestampMs(1_000)), - }); - scheduled.execute_at = Some(TimestampMs(11_000)); - type Case = ( - QueryWorkload, - DataWorkload, - Option, - fn(&SummaryMaintenanceLifecycle) -> bool, - ); - let cases: [Case; 3] = [ - ( - workload(vec![], vec![repeating()], continuous(1_000, 60_000)), - continuous(1_000, 60_000), - Some(Horizon(10.0)), - |lifecycle| { - matches!( - lifecycle, - SummaryMaintenanceLifecycle::ContinuouslyMaintained - ) - }, - ), - ( - workload(vec![], vec![repeating()], at_rest()), - at_rest(), - Some(Horizon(10.0)), - |lifecycle| matches!(lifecycle, SummaryMaintenanceLifecycle::Shared { .. }), - ), - ( - workload(vec![scheduled], vec![], at_rest()), - at_rest(), - None, - |lifecycle| matches!(lifecycle, SummaryMaintenanceLifecycle::Prepared { .. }), - ), - ]; - for (workload, data, horizon, kind) in cases { - let dag = timed_dag( - readout(&summary()), - &workload, - &data, - horizon, - |deployment| lifecycle_matching(&deployment.alternatives, kind), - ); - assert_eq!( - timings(&dag), - [("raw", INGEST), ("state", INGEST), ("readout", QUERY)] - ); - } - } - - // An Ephemeral state, its raw input, and its readout all run at query time. - #[test] - fn ephemeral_lifecycle_times_state_and_downstream_at_query() { - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let dag = timed_dag(readout(&summary()), &workload, &at_rest(), None, |_| { - SummaryMaintenanceLifecycle::Ephemeral - }); - assert_eq!( - timings(&dag), - [("raw", QUERY), ("state", QUERY), ("readout", QUERY)] - ); - } - - // One state read by two consumers is one deployment; its timing follows - // that single choice while both consumers run at query time. - #[test] - fn shared_state_is_timed_once_for_all_consumers() { - let state = summary(); - let lhs = readout(&state); - let rhs = Rc::new(lhs.as_ref().clone()); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - lhs, - rhs, - operator: asap_types::post_asap::BinaryOperator { - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - timing: QUERY, - }, - schema: readout(&state).schema.clone(), - guarantee: None, - }); - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let dag = timed_dag(root, &workload, &data, Some(Horizon(10.0)), |deployment| { - assert!(Rc::ptr_eq(&deployment.summary, &state)); - SummaryMaintenanceLifecycle::ContinuouslyMaintained - }); - assert_eq!( - timings(&dag), - [ - ("raw", INGEST), - ("state", INGEST), - ("readout", QUERY), - ("readout", QUERY), - ("binary", QUERY), - ] - ); - } - - // An Ephemeral state consumed by retained state is built on the retained - // state's ingestion path; it is not retained, but cannot run at query time. - #[test] - fn ephemeral_state_feeding_retained_state_runs_at_ingestion() { - let mut scheduled = batch(Predictability::Predictable { - known_at: Some(TimestampMs(1_000)), - }); - scheduled.execute_at = Some(TimestampMs(11_000)); - let root = nested_summary(); - let workload = workload(vec![scheduled], vec![], at_rest()); - let dag = timed_dag( - Rc::clone(&root), - &workload, - &at_rest(), - None, - |deployment| { - if Rc::ptr_eq(&deployment.summary, &root) { - lifecycle_matching(&deployment.alternatives, |lifecycle| { - matches!(lifecycle, SummaryMaintenanceLifecycle::Prepared { .. }) - }) - } else { - SummaryMaintenanceLifecycle::Ephemeral - } - }, - ); - assert_eq!( - timings(&dag), - [("raw", INGEST), ("state", INGEST), ("state", INGEST)] - ); - } - - // Timing is not derived for a state without a selected lifecycle, and a - // raw-recompute plan runs entirely at query time. - #[test] - fn timing_requires_a_selected_lifecycle_for_every_state() { - let workload = workload(vec![batch(Predictability::AdHoc)], vec![], at_rest()); - let data = at_rest(); - let demand = WorkloadDemand::new_with_data(&workload, &data, &[0]); - let plan = plan_summary_maintenance_lifecycles( - readout(&summary()), - demand, - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &crate::cost_model::DefaultCostModel, - ) - .unwrap(); - assert_eq!( - plan.execution_timed_dag().unwrap_err(), - SummaryMaintenanceTimingError::UnselectedLifecycle( - plan.deployments[0].post_asap_node_id - ) - ); - let raw = plan_summary_maintenance_lifecycles( - crate::replacement::keep_pre_asap(&sum_query()).unwrap(), - demand, - 1_000, - None, - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - assert_eq!( - timings(&raw.execution_timed_dag().unwrap()), - [("raw", QUERY)] - ); - } - - /// A strategy-built `sum(a)` over one maintained current-series population. - fn population_readout() -> Rc { - let target = Rc::new(crate::test_support::lower_promql( - "sum(a)", - AccuracyTarget::Exact, - )); - crate::maintained_population::MaintainedPopulationStrategy::new(std::slice::from_ref( - &target, - )) - .candidate(&target) - .unwrap() - } - - fn is_population(node: &SummaryNode) -> bool { - matches!( - node.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { .. }, - .. - } - ) - } - - fn population_timings(dag: &PostAsapDAG) -> Vec<(&'static str, ExecutionTiming)> { - dag.nodes - .iter() - .zip(timings(dag)) - .map(|(node, (kind, timing))| match node.payload { - PostAsapOperatorPayload::Value { - operation: ValueOperation::MaintainPopulation { .. }, - } => ("population", timing), - _ => (kind, timing), - }) - .collect() - } - - // A maintained population is enumerated as retained state, with costs - // from the caller's model for both the maintained and the rebuilt choice. - #[test] - fn enumeration_includes_maintained_population() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let candidates = enumerate_summary_maintenance_lifecycles( - population_readout(), - WorkloadDemand::new_with_data(&workload, &data, &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - let [deployment] = candidates.deployments() else { - panic!("one population state"); - }; - assert!(is_population(&deployment.summary)); - let cost = |lifecycle: SummaryMaintenanceLifecycle| { - deployment - .alternatives - .iter() - .find(|alternative| alternative.summary_maintenance_lifecycle == lifecycle) - .and_then(|alternative| alternative.total_cost) - }; - // Ephemeral: (build 10 + read 1 + retire 1) x 10 reads. Maintained over - // 10 s at 1 update/s: build 10 + updates 10 + reads 10 + retention 1 + retire 1. - assert_eq!( - cost(SummaryMaintenanceLifecycle::Ephemeral), - Some(Cost(120.0)) - ); - assert_eq!( - cost(SummaryMaintenanceLifecycle::ContinuouslyMaintained), - Some(Cost(32.0)) - ); - } - - // Without cost evidence a population's alternatives stay unknown: Planner - // selects none and timing is refused rather than guessed. - #[test] - fn population_without_cost_evidence_stays_unselected() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let plan = plan_summary_maintenance_lifecycles( - population_readout(), - WorkloadDemand::new_with_data(&workload, &data, &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &crate::cost_model::DefaultCostModel, - ) - .unwrap(); - let [deployment] = plan.deployments.as_slice() else { - panic!("one population state"); - }; - assert!(deployment - .alternatives - .iter() - .all(|alternative| alternative.total_cost.is_none())); - assert!(deployment.summary_maintenance_lifecycle_guarantee.is_none()); - assert_eq!( - plan.execution_timed_dag().unwrap_err(), - SummaryMaintenanceTimingError::UnselectedLifecycle(deployment.post_asap_node_id) - ); - } - - // A retained population and its raw input run at ingestion time; an - // Ephemeral population is rebuilt from raw input at query time. - #[test] - fn population_lifecycle_choice_decides_its_timing() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let timed = |lifecycle: SummaryMaintenanceLifecycle| { - population_timings(&timed_dag( - population_readout(), - &workload, - &data, - Some(Horizon(10.0)), - |_| lifecycle.clone(), - )) - }; - assert_eq!( - timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), - [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] - ); - assert_eq!( - timed(SummaryMaintenanceLifecycle::Ephemeral), - [("raw", QUERY), ("population", QUERY), ("readout", QUERY)] - ); - } - - // When retaining is cheaper, Planner's own selection keeps the population - // maintained at ingestion time, as realization strategies placed it before - // population timing became a lifecycle decision. - #[test] - fn planner_selection_retains_population_at_ingestion() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let plan = plan_summary_maintenance_lifecycles( - population_readout(), - WorkloadDemand::new_with_data(&workload, &data, &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - // Shared and ContinuouslyMaintained tie at 32; the first wins. - assert!(matches!( - selected_summary_maintenance_lifecycle(&plan.deployments[0]), - Some(SummaryMaintenanceLifecycle::Shared { .. }) - )); - assert_eq!( - population_timings(&plan.execution_timed_dag().unwrap()), - [("raw", INGEST), ("population", INGEST), ("readout", QUERY)] - ); - } - - // A population feeding summary state is that state's input, not a separate - // deployment: the state's lifecycle times it. - #[test] - fn population_feeding_summary_state_follows_that_state() { - let SummaryExpr::ValueOperation { - child: population, .. - } = &population_readout().expr - else { - unreachable!() - }; - let state = summary(); - let SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - .. - } = &state.expr - else { - unreachable!() - }; - let state = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(population), - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: None, - }, - ..state.as_ref().clone() - }); - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let timed = |lifecycle: SummaryMaintenanceLifecycle| { - population_timings(&timed_dag( - readout(&state), - &workload, - &data, - Some(Horizon(10.0)), - |deployment| { - assert!(Rc::ptr_eq(&deployment.summary, &state)); - lifecycle.clone() - }, - )) - }; - assert_eq!( - timed(SummaryMaintenanceLifecycle::ContinuouslyMaintained), - [ - ("raw", INGEST), - ("population", INGEST), - ("state", INGEST), - ("readout", QUERY) - ] - ); - assert_eq!( - timed(SummaryMaintenanceLifecycle::Ephemeral), - [ - ("raw", QUERY), - ("population", QUERY), - ("state", QUERY), - ("readout", QUERY) - ] - ); - } - - // A population both read directly and consumed by summary state is that - // state's input in either traversal order: not a separate deployment, and - // timed by the state's lifecycle. - #[test] - fn shared_population_follows_its_summary_consumer() { - let direct = population_readout(); - let SummaryExpr::ValueOperation { - child: population, .. - } = &direct.expr - else { - unreachable!() - }; - let state = summary(); - let SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - .. - } = &state.expr - else { - unreachable!() - }; - let state = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(population), - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: None, - }, - ..state.as_ref().clone() - }); - let binary = |lhs: Rc, rhs: Rc| { - Rc::new(SummaryNode { - schema: lhs.schema.clone(), - expr: SummaryExpr::BinaryOp { - lhs, - rhs, - operator: asap_types::post_asap::BinaryOperator { - kind: asap_types::pre_asap::BinaryOpKind::Arithmetic( - asap_types::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - timing: QUERY, - }, - guarantee: None, - }) - }; - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - for root in [ - binary(Rc::clone(&direct), readout(&state)), - binary(readout(&state), Rc::clone(&direct)), - ] { - for lifecycle in [ - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - SummaryMaintenanceLifecycle::Ephemeral, - ] { - let dag = timed_dag( - Rc::clone(&root), - &workload, - &data, - Some(Horizon(10.0)), - |deployment| { - assert!(Rc::ptr_eq(&deployment.summary, &state)); - lifecycle.clone() - }, - ); - let expected = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { - QUERY - } else { - INGEST - }; - for (kind, timing) in population_timings(&dag) { - if matches!(kind, "raw" | "population" | "state") { - assert_eq!(timing, expected, "{kind}"); - } else { - assert_eq!(timing, QUERY, "{kind}"); - } - } - } - } - } - - // A plan whose population deployment was removed after enumeration is - // refused rather than timed by a guess. - #[test] - fn timing_refuses_population_without_deployment() { - let data = continuous(1_000, 60_000); - let workload = workload(vec![], vec![repeating()], data.clone()); - let mut plan = plan_summary_maintenance_lifecycles( - population_readout(), - WorkloadDemand::new_with_data(&workload, &data, &[0]), - 1_000, - Some(Horizon(10.0)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &UnitCosts, - ) - .unwrap(); - let id = plan.deployments.remove(0).post_asap_node_id; - assert_eq!( - plan.execution_timed_dag(), - Err(SummaryMaintenanceTimingError::UnplannedMaintainedState(id)) - ); - } -} diff --git a/crates/devtools/Cargo.toml b/crates/devtools/Cargo.toml index 0abc8ccdf..484ce7991 100644 --- a/crates/devtools/Cargo.toml +++ b/crates/devtools/Cargo.toml @@ -10,7 +10,8 @@ edition = "2021" [dependencies] asap-frontend-promql = { path = "../frontend-promql" } asap-frontend-sql = { path = "../frontend-sql" } -asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-plan-selection = { path = "../plan-selection" } +asap-logical-optimizer = { path = "../logical-optimizer" } # Used by the show_ir / dag_export / variant_coverage bins (catalog schemas, # async SQL path, JSON output) and by the topk_ir / canonical_examples diff --git a/crates/devtools/examples/canonical_examples.rs b/crates/devtools/examples/canonical_examples.rs index b0df222a3..8e2de95ca 100644 --- a/crates/devtools/examples/canonical_examples.rs +++ b/crates/devtools/examples/canonical_examples.rs @@ -1,11 +1,11 @@ // cargo run -p asap-lower --example canonical_examples // -// One-off: pretty-print the QueryExpr for one canonical query per variant, +// One-off: pretty-print the `OperatorNode` DAG for one canonical query per variant, // plus custom Join/SetOp/Dedup/CTE probes, to eyeball the actual shape. use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -48,7 +48,7 @@ fn bgp_catalog() -> SqlCatalog { async fn main() { let promql_examples: &[(&str, &str)] = &[ ("Scan", "up"), - ("BinaryOp + PromqlScalarBridge", "up > 1"), + ("Filter + scalar predicate", "up > 1"), ("EvalTimestamp", "time()"), ("Aggregate", "sum(up)"), ( diff --git a/crates/devtools/examples/topk_ir.rs b/crates/devtools/examples/topk_ir.rs index 21dcefb4b..6f251e22d 100644 --- a/crates/devtools/examples/topk_ir.rs +++ b/crates/devtools/examples/topk_ir.rs @@ -4,7 +4,7 @@ // resulting pre-ASAP IR. Used for interactive exploration; not a test. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; fn col(name: &str, dtype: DataType) -> Field { diff --git a/crates/devtools/src/bin/analyze_corpora.rs b/crates/devtools/src/bin/analyze_corpora.rs index 7aa87c8e7..0b3f799a1 100644 --- a/crates/devtools/src/bin/analyze_corpora.rs +++ b/crates/devtools/src/bin/analyze_corpora.rs @@ -6,7 +6,7 @@ use asap_devtools::{lower_promql_with_data_ingestion_interval, SqlCatalog}; use asap_frontend_sql::lower_sql_dialect; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use serde::Serialize; @@ -218,7 +218,7 @@ fn run_corpus(name: &str, source: &str, interval_ms: u64) -> CorpusResult { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("OperatorNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), @@ -493,7 +493,7 @@ async fn run_sql_corpora(out_dir: PathBuf) { normalized_expression, structural_shape, lowered: true, - ir: Some(serde_json::to_value(&ir).expect("QueryExpr must serialize")), + ir: Some(serde_json::to_value(&ir).expect("OperatorNode must serialize")), ir_debug: Some(format!("{ir:#?}")), error: None, }), diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index b0615f017..f94a40781 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -12,7 +12,7 @@ // `--epsilon ` is optional and applies to every query in the run: it // lowers with `AccuracyTarget::Epsilon()` instead of the default // `AccuracyTarget::Exact`. Without it, every `AggIntent` lowers exact and -// `asap_aware_mapping::SketchAlgorithmStrategy` never has a genuine sketch +// `asap_logical_optimizer::ASAPStrategies` never has a genuine sketch // alternative to report — so no node ever picks up a `SketchApproximation` // note. Pass it to actually exercise that path, e.g.: // cargo run -p asap-lower --bin dag_export -- \ @@ -23,10 +23,10 @@ // below). // // `--post-asap` is optional and off by default. When passed, this binary -// additionally runs `asap_aware_mapping::replacement::search_workload` (this +// additionally runs `asap_logical_optimizer::pass1::replacement::search_workload` (this // binary took no strategies of its own — `default_strategies()` already // includes `AvgToSumOverCountStrategy` as of #282) over every lowered query -// and ranks each discovered `TargetSubDAGCandidates` via `CandidateLogicalASAPDAGs::cost_sorted`. The +// and ranks each discovered `TargetSubDAGCandidates` via `candidate_selection::cost_sorted`. The // best-ranked // candidate per group feeds two additive outputs: // @@ -41,7 +41,7 @@ // `asap_types::dag_export::export_post_asap`. // // Together these surface every one of the four concrete replacement kinds: -// the sketch family `SketchAlgorithmStrategy`/`HydraGroupingStrategy` bound, +// the sketch family `ASAPStrategies`/`HydraGroupingStrategy` bound, // the CSE share/recompute choice `SharedSubDAGStrategy` found, the // workload-aware roll-up `RollupStrategy` derived, and the `avg -> // sum/count` rewrite `AvgToSumOverCountStrategy` proposes. Without @@ -60,7 +60,7 @@ // - `--planner-cost-json ` supplies complete deployment-owned // physical-plan evidence. It both ranks the candidates and is exported: // every decision carries a calibrated `CostUnits` annotation. -// - `--default-cost` ranks with `asap_aware_mapping::cost_model:: +// - `--default-cost` ranks with `asap_plan_selection::cost::cost_model:: // DefaultCostModel` — structural node counts, owning no deployment // evidence. The structure of the result is real (which replacements the // search found, which one won per group, what the merged post-ASAP DAG @@ -76,44 +76,44 @@ use std::collections::HashMap; use std::rc::Rc; use std::time::Instant; -use asap_aware_mapping::analytical_cost::{ +use asap_logical_optimizer::pass1::replacement::{ + default_strategies_with_evidence, is_logical_rewrite, search_workload, search_workload_with, + Replacement, ReplacementSubDAG, +}; +use asap_logical_optimizer::{AccuracyEvidenceProvider, PropagationStats}; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::cost::analytical_cost::{ cache_hit_ratios, AnalyticalCostError, EvidenceBackedPhysicalDAG as PhysicalDAG, PhysicalNodeEvidence, ResourceCalibration, ANALYTICAL_COST_MODEL_VERSION, }; -use asap_aware_mapping::cost_model::DefaultCostModel; -use asap_aware_mapping::cost_model::{Cost, CostModel}; -use asap_aware_mapping::physical_operator_statistics::ComparisonScope; -use asap_aware_mapping::physical_plan_cost_model::{ +use asap_plan_selection::cost::cost_model::DefaultCostModel; +use asap_plan_selection::cost::cost_model::{Cost, CostModel}; +use asap_plan_selection::cost::physical_operator_statistics::ComparisonScope; +use asap_plan_selection::cost::physical_plan_cost_model::{ PhysicalEvidenceSnapshot, PhysicalPlanCostModel, PlannerPhysicalPlanProvider, }; -use asap_aware_mapping::query_physical_lowering::PhysicalNodeRequest; -use asap_aware_mapping::replacement::{ - default_strategies_with_evidence, search_workload, search_workload_with, Replacement, - ReplacementSubDAG, -}; -use asap_aware_mapping::{AccuracyEvidenceProvider, PropagationStats}; +use asap_plan_selection::cost::query_physical_lowering::PhysicalNodeRequest; use asap_types::cost::{BaselineRef, CostAnnotation, CostInput, CostSource, CostUnit}; use asap_types::dag_export::{ self, DAGDecision, DAGNote, ExportDAG, NamedDAG, PostAsapSubstitution, TargetRejection, TargetReplacement, TargetReplacementAfter, WorkloadDAG, }; -use asap_types::post_asap::SummaryExpr; -use asap_types::post_asap::SummaryNode; -use asap_types::post_asap::{CompositionOperator, FieldDataType, SketchStatistic}; -use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::resources::CacheProfile; +use asap_types::ir::cse::{structural_hash, HashCache}; +use asap_types::ir::properties::CompositionOperator; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{FieldDataType, SketchStatistic}; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; +use asap_types::workload::resources::CacheProfile; #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct PlannerCostDocument { #[serde(default, skip_serializing_if = "Option::is_none")] - storage_io: Option, + storage_io: Option, #[serde(default, skip_serializing_if = "Option::is_none")] #[serde(rename = "boundaries")] - handoffs: Option, + handoffs: Option, /// Immutable catalog/runtime evidence generation shared by this file. evidence_version: String, calibration: ResourceCalibration, @@ -137,7 +137,7 @@ fn parse_planner_cost_document(raw: &str) -> Result #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct TargetPhysicalEvidence { - target: QueryExpr, + target: Rc, scope: ComparisonScopeEvidence, candidates: Vec, } @@ -152,7 +152,7 @@ struct ComparisonScopeEvidence { time_scope: String, lookback_ms: Option, as_of_ms: Option, - sources: Vec, + sources: Vec, #[serde(default = "CacheProfile::no_cache")] cache_profile: CacheProfile, } @@ -192,8 +192,8 @@ impl ComparisonScopeEvidence { #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] struct QueryNodePhysicalEvidence { - logical_node: QueryExpr, - operator: asap_aware_mapping::analytical_cost::PhysicalOperator, + logical_node: OperatorNode, + operator: asap_plan_selection::cost::analytical_cost::PhysicalOperator, occurrence: usize, synthetic: bool, evidence: PhysicalNodeEvidence, @@ -228,11 +228,11 @@ impl CandidatePhysicalEvidence { fn matches(&self, candidate: &ReplacementSubDAG) -> bool { let actual = match (self, &candidate.replacement) { - (Self::Summary { .. }, Replacement::Summary(summary)) => { - serde_json::to_value(dag_export::export_summary(summary)) + (Self::Summary { .. }, Replacement::SubDAG(node)) if !is_logical_rewrite(node) => { + serde_json::to_value(dag_export::export(node)) } - (Self::Rewrite { .. }, Replacement::Rewrite(query)) => { - serde_json::to_value(dag_export::export(query)) + (Self::Rewrite { .. }, Replacement::SubDAG(node)) if is_logical_rewrite(node) => { + serde_json::to_value(dag_export::export(node)) } _ => return false, }; @@ -302,8 +302,8 @@ fn plan_values_match_inner( } struct ExportPhysicalProvider<'a> { - storage_io: Option<&'a asap_aware_mapping::storage_io::StorageIoProfile>, - handoffs: Option<&'a asap_aware_mapping::physical_handoff_cost::PhysicalHandoffProfile>, + storage_io: Option<&'a asap_plan_selection::cost::storage_io::StorageIoProfile>, + handoffs: Option<&'a asap_plan_selection::cost::physical_handoff_cost::PhysicalHandoffProfile>, evidence_version: &'a str, target: &'a TargetPhysicalEvidence, candidate: &'a CandidatePhysicalEvidence, @@ -319,7 +319,7 @@ impl ExportPhysicalProvider<'_> { impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { fn capture_evidence_snapshot( &self, - _target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, + _target: &asap_logical_optimizer::pass1::replacement::TargetSubDAG<'_>, ) -> Result { Ok(PhysicalEvidenceSnapshot { version: self.evidence_version.into(), @@ -369,8 +369,8 @@ impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, - _target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, + _summary: &Rc, + _target: &asap_logical_optimizer::pass1::replacement::TargetSubDAG<'_>, ) -> Result { if snapshot.scope != self.target.scope.resolve()? { return Err(AnalyticalCostError::ComparisonScopeMismatch( @@ -394,13 +394,13 @@ impl ExportPlannerCostModel<'_> { fn bound<'a>( &'a self, candidate: &ReplacementSubDAG, - target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, + target: &asap_logical_optimizer::pass1::replacement::TargetSubDAG<'_>, ) -> Option<(ExportPhysicalProvider<'a>, &'a ResourceCalibration)> { let mut targets = self .document .targets .iter() - .filter(|entry| entry.target == **target.root); + .filter(|entry| entry.target == *target.root); let target_evidence = targets.next()?; if targets.next().is_some() { return None; @@ -429,9 +429,9 @@ impl ExportPlannerCostModel<'_> { fn annotations( &self, candidate: &ReplacementSubDAG, - target: &Rc, + target: &Rc, ) -> (CostAnnotation, CostAnnotation, CostAnnotation) { - let target = asap_aware_mapping::replacement::TargetSubDAG::new(target); + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(target); let Some((provider, calibration)) = self.bound(candidate, &target) else { return winner_cost_annotations(); }; @@ -516,7 +516,7 @@ impl ExportPlannerCostModel<'_> { cache_inputs.push(input("result_cache_invalidation_ratio", ratio, "ratio")); } } - let inputs = |resources: asap_aware_mapping::analytical_cost::ResourceEstimate| { + let inputs = |resources: asap_plan_selection::cost::analytical_cost::ResourceEstimate| { let mut inputs = vec![ CostInput { name: "estimated_cpu_ops".into(), @@ -537,7 +537,7 @@ impl ExportPlannerCostModel<'_> { inputs.extend(cache_inputs.iter().cloned()); inputs }; - let storage_inputs = |storage: &asap_aware_mapping::storage_io::StorageEstimate| { + let storage_inputs = |storage: &asap_plan_selection::cost::storage_io::StorageEstimate| { let mut terms: Vec<_> = storage .total .terms() @@ -571,7 +571,7 @@ impl ExportPlannerCostModel<'_> { candidate_inputs.extend(storage_inputs(candidate)); } let handoff_inputs = - |estimate: &asap_aware_mapping::physical_handoff_cost::PhysicalHandoffEstimate| { + |estimate: &asap_plan_selection::cost::physical_handoff_cost::PhysicalHandoffEstimate| { let mut terms: Vec<_> = estimate .total .terms() @@ -650,7 +650,7 @@ impl CostModel for ExportPlannerCostModel<'_> { fn candidate_cost( &self, candidate: &ReplacementSubDAG, - target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, + target: &asap_logical_optimizer::pass1::replacement::TargetSubDAG<'_>, ) -> Option { let (provider, calibration) = self.bound(candidate, target)?; let cost = PhysicalPlanCostModel::new(&provider, calibration.clone()) @@ -661,9 +661,9 @@ impl CostModel for ExportPlannerCostModel<'_> { fn rank_candidates( &self, - _intent: &asap_types::pre_asap::AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { + _intent: &asap_types::ir::operator::AggIntent, + candidates: &[asap_types::ir::schema::SketchAlgorithm], + ) -> Vec { // Candidate generation must not reintroduce the legacy structural // cost model before complete physical alternatives are compared. candidates.to_vec() @@ -672,7 +672,7 @@ impl CostModel for ExportPlannerCostModel<'_> { fn estimate_cost( &self, candidate: &ReplacementSubDAG, - target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, + target: &asap_logical_optimizer::pass1::replacement::TargetSubDAG<'_>, ) -> f64 { self.candidate_cost(candidate, target) .map_or(f64::NAN, |cost| cost.0) @@ -960,21 +960,20 @@ fn parse_args_from(argv: impl Iterator) -> ParsedArgs { } } -/// Attach workload-wide replacement explanations to their exact DAG nodes. -/// `node_hash` is only a narrowing filter; `source_expr == Some(target)` is -/// the collision-safe identity check (`source_expr` is `None` only for a -/// post-ASAP-originated node inside a `--post-asap` `post_dag`, which this -/// function is never called on — every node it sees, from an ordinary -/// [`dag_export::export`], carries `Some`). +/// Attach workload-wide replacement explanations to their exact dag nodes. +/// `node_hash` is only a narrowing filter; `source_node == Some(target)` is +/// the collision-safe identity check (every node an ordinary +/// [`dag_export::export`] produces carries `Some`; the `None` arm is +/// defensive only). fn annotate_with_explanations( dag: &mut ExportDAG, - explanations: &[asap_aware_mapping::ReplacementExplanation], + explanations: &[asap_logical_optimizer::ReplacementExplanation], matched: &mut [bool], ) { for (i, explanation) in explanations.iter().enumerate() { for node in dag.nodes.iter_mut() { if node.hash == Some(explanation.node_hash) - && node.source_expr.as_ref() == Some(explanation.target.as_ref()) + && node.source_node.as_ref() == Some(&explanation.target) { node.notes.push(DAGNote { kind: format!("{:?}", explanation.kind), @@ -992,7 +991,7 @@ fn annotate_with_explanations( /// never disagree about which candidate won for a given target. #[allow(dead_code)] struct Winner<'a> { - target: &'a Rc, + target: &'a Rc, candidate: &'a ReplacementSubDAG, costs: (CostAnnotation, CostAnnotation, CostAnnotation), } @@ -1013,10 +1012,10 @@ fn decision_rationale(winner: &Winner<'_>) -> String { .to_string() } "SharedSubDAGStrategy" => match winner.candidate.provenance { - asap_aware_mapping::replacement::ReplacementProvenance::CseShare => { + asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseShare => { "Builds the repeated sub-DAG once and shares it across consumers.".to_string() } - asap_aware_mapping::replacement::ReplacementProvenance::CseRecompute => { + asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseRecompute => { "Recomputes the sub-DAG per consumer because that has the lower estimated cost." .to_string() } @@ -1056,7 +1055,7 @@ fn lookup_winner( by_hash: &HashMap>, winners: &[Winner<'_>], cache: &mut HashCache, - expr: &QueryExpr, + expr: &OperatorNode, ) -> Option { let hash = structural_hash(expr, cache); by_hash @@ -1102,12 +1101,10 @@ fn target_replacement( let strategy = winner.candidate.strategy.to_string(); let before = dag_export::export(winner.target); let after = match &winner.candidate.replacement { - Replacement::Summary(node) => { - TargetReplacementAfter::Summary(dag_export::export_summary(node)) - } - Replacement::Rewrite(rewritten) => { - TargetReplacementAfter::Rewrite(dag_export::export(rewritten)) + Replacement::SubDAG(node) if is_logical_rewrite(node) => { + TargetReplacementAfter::Rewrite(dag_export::export(node)) } + Replacement::SubDAG(node) => TargetReplacementAfter::Summary(dag_export::export(node)), Replacement::ExactComposition(_) => { unreachable!("composition candidates are materialized by GlobalSelection") } @@ -1131,6 +1128,20 @@ fn target_replacement( } } +/// Is `replacement` `retain_exact`'s conservative no-op fallback — the +/// target itself, unbound, carrying only an exact "kept pre-ASAP" guarantee? +/// `ASAPStrategies` emits it for an intent with no summary +/// realization at all (`STDDEV_POP`, `AVG`, ... dispatch to +/// `Realization::PassThrough`). It is "nothing to bind here", not a +/// replacement decision. A logical rewrite (no guarantee yet) and any sub-DAG +/// with an ASAP operator are real candidates. +fn is_trivial_retain_exact(replacement: &Replacement) -> bool { + matches!( + replacement, + Replacement::SubDAG(node) if node.guarantee.is_some() && !node.contains_asap() + ) +} + /// The two additive `--post-asap` outputs — see this file's top-of-file /// usage doc for what each is for. struct PostAsapResults { @@ -1159,7 +1170,7 @@ fn raw_only_post_asap_results() -> PostAsapResults { } /// Assign collision-free, explicit identities to structurally equal nodes -/// across a set of exported query DAGs. The full canonical sub-DAG string +/// across a set of exported query graphs. The full canonical sub-DAG string /// is the equality key; the compact integer is what JSON consumers receive. /// Consequently the viewer never needs to guess identity from labels, /// hashes, or a client-side node signature. @@ -1200,17 +1211,17 @@ fn assign_workload_node_ids(dags: &mut [&mut ExportDAG]) { } } -/// Run `asap_aware_mapping::replacement::search_workload` (its own +/// Run `asap_logical_optimizer::pass1::replacement::search_workload` (its own /// `default_strategies()` — which includes `AvgToSumOverCountStrategy` as of /// #282 — is exactly the strategy set this binary wants; no custom list /// needed) over every lowered query, rank each discovered `TargetSubDAGCandidates` via -/// `CandidateLogicalASAPDAGs::global_selection`, and build both `--post-asap` outputs from the +/// `candidate_selection::global_selection`, and build both `--post-asap` outputs from the /// exact same set of winning candidates (see [`Winner`]), so the flat /// `replacements` list and the merged `post_dag` can never disagree about /// which candidate won for a given target. #[allow(dead_code)] fn run_post_asap_with_progress( - lowered_queries: &[(String, String, QueryExpr)], + lowered_queries: &[(String, String, Rc)], progress: bool, cost_model: &dyn CostModel, export_model: Option<&ExportPlannerCostModel<'_>>, @@ -1220,41 +1231,36 @@ fn run_post_asap_with_progress( if progress { eprintln!("[3/4] ASAP-aware mapping is running…"); } - let roots: Vec<(String, Rc)> = lowered_queries + let roots: Vec<(String, Rc)> = lowered_queries .iter() - .map(|(name, _, qe)| (name.clone(), Rc::new(qe.clone()))) + .map(|(name, _, qe)| (name.clone(), Rc::clone(qe))) .collect(); let strategies; let space = if let Some(evidence) = evidence { - strategies = default_strategies_with_evidence(cost_model, evidence); + strategies = default_strategies_with_evidence(evidence); search_workload_with(roots, &strategies) } else { search_workload(roots) }; - let selection = space.global_selection(cost_model); - - // A group's top candidate can be `keep_pre_asap`'s own conservative - // fallback — `Replacement::Summary(SummaryNode { expr: - // KeepPreAsap(Rc::new(target.clone())), .. })` — the *whole target* - // wrapped as unbound, e.g. for a multi-measure/`HAVING`-bearing - // aggregate, or (the case that actually surfaces this: `STDDEV_POP`/ - // `AVG`/`VARIANCE` dispatch to `Realization::PassThrough` with no - // alternative at all, per `realizations_for_intent`'s own doc) an - // intent with no summary realization whatsoever. This isn't a - // replacement decision — it's `SketchAlgorithmStrategy` saying "nothing - // to bind here" — the identical "no-op candidate" concept - // `explanation.rs`'s own `sketch_finding_reason` already excludes from - // being reported as a finding ("a candidate list containing only the - // trivial no-op realization... isn't an opportunity, it's just the - // target's existing shape reflected back"). Filtered out here for a - // second, load-bearing reason beyond just matching that precedent: - // `export_post_asap`'s `find_winner` re-checks every node reached - // inside a spliced-in `KeepPreAsap` payload (by design, so a target - // nested underneath one still gets found) — if that payload structurally - // *is* the enclosing target, `find_winner` immediately matches the same - // winner again, forever. Treating this candidate as "no winner" (same - // as an empty candidate list) avoids ever handing `export_post_asap` a - // winner that can't help but recurse into itself. + let selection = global_selection(&space, cost_model); + + // A group's top candidate can be `retain_exact`'s own conservative + // fallback — the *whole target* itself, unbound, carrying only an exact + // "kept pre-ASAP" guarantee (see `is_trivial_retain_exact`) — e.g. for + // a multi-measure/`HAVING`-bearing aggregate, or (the case that actually + // surfaces this: `STDDEV_POP`/`AVG`/`VARIANCE` dispatch to + // `Realization::PassThrough` with no alternative at all, per + // `realizations_for_intent`'s own doc) an intent with no summary + // realization whatsoever. This isn't a replacement decision — it's + // `ASAPStrategies` saying "nothing to bind here" — the + // identical "no-op candidate" concept `explanation.rs`'s own + // `sketch_finding_reason` already excludes from being reported as a + // finding ("a candidate list containing only the trivial no-op + // realization... isn't an opportunity, it's just the target's existing + // shape reflected back"). Treating this candidate as "no winner" (same + // as an empty candidate list) also keeps `post_dag` honest: splicing + // the target in for itself would tag every node of an unchanged sub-DAG + // with a "replacement" decision. let winners: Vec> = selection .target_selections() .filter_map(|group| { @@ -1267,10 +1273,7 @@ fn run_post_asap_with_progress( if matches!(candidate.replacement, Replacement::ExactComposition(_)) { return None; } - if matches!( - &candidate.replacement, - Replacement::Summary(node) if matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - ) { + if is_trivial_retain_exact(&candidate.replacement) { return None; } Some(Winner { @@ -1318,7 +1321,7 @@ fn run_post_asap_with_progress( } let post_started = Instant::now(); let mut post_dag_cache = HashCache::new(); - let mut find_winner = |expr: &QueryExpr| -> Option { + let mut find_winner = |expr: &Rc| -> Option { let i = lookup_winner(&by_hash, &winners, &mut post_dag_cache, expr)?; let winner = &winners[i]; let (baseline_cost, selected_cost, benefit) = winner.costs.clone(); @@ -1337,11 +1340,11 @@ fn run_post_asap_with_progress( benefit: Some(benefit), }; Some(match &winners[i].candidate.replacement { - Replacement::Rewrite(rc) => PostAsapSubstitution::Rewrite { + Replacement::SubDAG(rc) if is_logical_rewrite(rc) => PostAsapSubstitution::Rewrite { replacement: Rc::clone(rc), decision, }, - Replacement::Summary(rc) => PostAsapSubstitution::Summary { + Replacement::SubDAG(rc) => PostAsapSubstitution::Summary { replacement: Rc::clone(rc), decision, }, @@ -1389,20 +1392,20 @@ fn run_post_asap_with_progress( for (name, _, qe) in lowered_queries { let dag = dag_export::export(qe); for node in &dag.nodes { - let Some(source_expr) = node.source_expr.as_ref() else { + let Some(source_node) = node.source_node.as_ref() else { continue; // never true for a plain `export` — defensive only. }; - if let Some(i) = lookup_winner(&by_hash, &winners, &mut lookup_cache, source_expr) { + if let Some(i) = lookup_winner(&by_hash, &winners, &mut lookup_cache, source_node) { replacements.push(( name.clone(), target_replacement(i as u32, node.id, &winners[i]), )); matched[i] = true; } - let hash = structural_hash(source_expr, &mut lookup_cache); + let hash = structural_hash(source_node, &mut lookup_cache); for &i in rejected_by_hash.get(&hash).into_iter().flatten() { let group = rejected_groups[i]; - if *source_expr != *group.target { + if *source_node != group.target { continue; } rejections.extend(group.rejected.iter().map(|rejected| { @@ -1436,7 +1439,7 @@ fn run_post_asap_with_progress( // winner's own `Replacement::Rewrite` — logged as an FYI rather than a // warning, since telling the two cases apart precisely would mean // reimplementing `search`'s own private descendant-discovery walk - // (`discover_new_descendant_targets` in `asap_aware_mapping::replacement`, + // (`discover_new_descendant_targets` in `asap_logical_optimizer::pass1::replacement`, // not exposed) a second time here just to double-check something // `post_dag`'s own construction already handled correctly. for (winner, matched) in winners.iter().zip(&matched) { @@ -1465,7 +1468,7 @@ fn run_post_asap_with_progress( } #[cfg(test)] -fn run_post_asap(lowered_queries: &[(String, String, QueryExpr)]) -> PostAsapResults { +fn run_post_asap(lowered_queries: &[(String, String, Rc)]) -> PostAsapResults { run_post_asap_with_progress(lowered_queries, false, &DefaultCostModel, None, None) } @@ -1527,7 +1530,7 @@ async fn main() { if progress { eprintln!("[2/4] Pre-ASAP DAG generation is running…"); } - let explanations = asap_aware_mapping::explain_replacements( + let explanations = asap_logical_optimizer::explain_replacements( lowered_queries .iter() .map(|(name, _, qe)| (name.clone(), qe.clone())) @@ -1696,47 +1699,55 @@ async fn main() { mod tests { use super::*; - use asap_aware_mapping::analytical_cost::{ + use asap_devtools::PromqlError; + use asap_plan_selection::cost::analytical_cost::{ ExecutionMultiplicity, PhysicalDAGNode, PhysicalOperator, }; - use asap_aware_mapping::physical_operator_statistics::{ - EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, + use asap_plan_selection::cost::physical_operator_statistics::{ + EdgeStatistics, OperatorStatistics, ScanSelection, UnaryEdgeStatistics, }; - use asap_aware_mapping::query_physical_lowering::lower_query_physical_dag; - use asap_devtools::PromqlError; - use asap_types::pre_asap::{DataType, Field, Reduction, Schema, Source}; - - fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { + use asap_plan_selection::cost::query_physical_lowering::lower_query_physical_dag; + use asap_types::ir::operator::{Reduction, Source}; + use asap_types::ir::schema::{DataType, Field, Schema}; + use asap_types::ir::NonASAPOp; + + fn lower_promql( + query: &str, + accuracy: AccuracyTarget, + ) -> Result, PromqlError> { lower_promql_with_data_ingestion_interval(query, accuracy, 1_000) } - fn non_topk_query() -> QueryExpr { - QueryExpr::Aggregate { + fn non_topk_query() -> Rc { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "events".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), + })) + .expect("scan leaf derives its schema"); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), - measures: vec![asap_types::pre_asap::AggIntent::Count { + measures: vec![asap_types::ir::operator::AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.1), }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "events".into(), - }, - predicates: vec![], - schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), - }), - } + child: scan, + })) + .expect("count aggregate derives its schema") } fn fixture_raw_dag( - query: &QueryExpr, + query: &Rc, candidate: &ReplacementSubDAG, document: &PlannerCostDocument, ) -> PhysicalDAG { let model = ExportPlannerCostModel { document }; - let root = Rc::new(query.clone()); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); + let root = Rc::clone(query); + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&root); let (provider, _) = model.bound(candidate, &target).unwrap(); let snapshot = provider.capture_evidence_snapshot(&target).unwrap(); let evidence = @@ -1747,12 +1758,12 @@ mod tests { // JSON evidence reaches calibrated ranking and structured annotation inputs. #[test] fn storage_requests_export_and_change_plan_selection() { - use asap_aware_mapping::storage_io::*; + use asap_plan_selection::cost::storage_io::*; let (query, candidate, mut document) = cost_fixture(); let raw = fixture_raw_dag(&query, &candidate, &document); let candidate_dag = cheap_candidate_dag(); - let root = Rc::new(query.clone()); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); + let root = Rc::clone(&query); + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&root); assert!(ExportPlannerCostModel { document: &document } @@ -1910,12 +1921,12 @@ mod tests { // Declared handoffs survive JSON and affect the selected physical plan. #[test] fn handoff_bytes_export_and_change_plan_selection() { - use asap_aware_mapping::physical_handoff_cost::*; + use asap_plan_selection::cost::physical_handoff_cost::*; let (query, candidate, mut document) = cost_fixture(); let raw = fixture_raw_dag(&query, &candidate, &document); let candidate_dag = cheap_candidate_dag(); - let root = Rc::new(query.clone()); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); + let root = Rc::clone(&query); + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&root); assert!(ExportPlannerCostModel { document: &document } @@ -2024,7 +2035,7 @@ mod tests { // Independent supplemental objectives add once, and either one can // price a zero-base plan while retaining both sets of export evidence. { - use asap_aware_mapping::storage_io::*; + use asap_plan_selection::cost::storage_io::*; let mut joint = handoff_only.clone(); let mut storage = StorageIoProfile { evidence_version: joint.evidence_version.clone(), @@ -2178,7 +2189,7 @@ mod tests { time_scope: "longitudinal".into(), lookback_ms: Some(10_000), as_of_ms: Some(1_000), - sources: vec![SourceCoverage { + sources: vec![ScanSelection { source: Source::Table { table_ref: "events".into(), }, @@ -2194,7 +2205,7 @@ mod tests { EdgeStatistics { rows, bytes } } - fn query_evidence(query: &QueryExpr) -> Vec { + fn query_evidence(query: &Rc) -> Vec { let entries = RefCell::new(Vec::new()); let scope = test_scope().resolve().unwrap(); let provider = |request: PhysicalNodeRequest<'_>| { @@ -2242,7 +2253,7 @@ mod tests { }); Ok(evidence) }; - lower_query_physical_dag(&Rc::new(query.clone()), &scope, &provider).unwrap(); + lower_query_physical_dag(query, &scope, &provider).unwrap(); entries.into_inner() } @@ -2261,7 +2272,7 @@ mod tests { id: "summary-read".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 2_400, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2282,49 +2293,28 @@ mod tests { fn candidate_plan(candidate: &ReplacementSubDAG) -> serde_json::Value { match &candidate.replacement { - Replacement::Summary(summary) => { - serde_json::to_value(dag_export::export_summary(summary)).unwrap() - } - Replacement::Rewrite(rewrite) => { - serde_json::to_value(dag_export::export(rewrite)).unwrap() - } + Replacement::SubDAG(node) => serde_json::to_value(dag_export::export(node)).unwrap(), Replacement::ExactComposition(_) => { unreachable!("cost fixtures select directly materialized candidates") } } } - fn cost_fixture() -> (QueryExpr, ReplacementSubDAG, PlannerCostDocument) { + fn cost_fixture() -> (Rc, ReplacementSubDAG, PlannerCostDocument) { let query = non_topk_query(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let space = search_workload(vec![(String::from("q"), Rc::clone(&root))]); let group = space .target_subdag_candidates() - .find(|group| *group.target == query) + .find(|group| group.target == query) .expect("aggregate memo group"); let candidate = group .candidates .iter() - .find(|candidate| { - !matches!( - &candidate.replacement, - Replacement::Summary(node) - if matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - ) - }) + .find(|candidate| !is_trivial_retain_exact(&candidate.replacement)) .expect("summary candidate") .clone(); - let plan = match &candidate.replacement { - Replacement::Summary(summary) => { - serde_json::to_value(dag_export::export_summary(summary)).unwrap() - } - Replacement::Rewrite(rewrite) => { - serde_json::to_value(dag_export::export(rewrite)).unwrap() - } - Replacement::ExactComposition(_) => { - unreachable!("cost fixtures select directly materialized candidates") - } - }; + let plan = candidate_plan(&candidate); let document = PlannerCostDocument { storage_io: None, handoffs: None, @@ -2339,12 +2329,14 @@ mod tests { target: query.clone(), scope: test_scope(), candidates: vec![match &candidate.replacement { - Replacement::Summary(_) => CandidatePhysicalEvidence::Summary { - plan, - query_nodes: query_evidence(&query), - physical_dag: cheap_candidate_dag(), - }, - Replacement::Rewrite(_) => CandidatePhysicalEvidence::Rewrite { + Replacement::SubDAG(node) if !is_logical_rewrite(node) => { + CandidatePhysicalEvidence::Summary { + plan, + query_nodes: query_evidence(&query), + physical_dag: cheap_candidate_dag(), + } + } + Replacement::SubDAG(_) => CandidatePhysicalEvidence::Rewrite { plan, query_nodes: query_evidence(&query), }, @@ -2365,15 +2357,15 @@ mod tests { assert_eq!(parsed.targets[0].target, query); assert!(parsed.targets[0].candidates[0].matches(&candidate)); let model = ExportPlannerCostModel { document: &parsed }; - let target_rc = Rc::new(query.clone()); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); + let target_rc = Rc::clone(&query); + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&target_rc); let (provider, calibration) = model.bound(&candidate, &target).expect("exact binding"); let estimate = PhysicalPlanCostModel::new(&provider, calibration.clone()) .unwrap() .estimate_candidate(&candidate, &target) .unwrap(); assert!(estimate.candidate_cost < estimate.raw_cost); - let (baseline, selected, benefit) = model.annotations(&candidate, &Rc::new(query)); + let (baseline, selected, benefit) = model.annotations(&candidate, &query); assert!(baseline.value.is_some()); assert!(selected.value.is_some()); assert!(benefit.value.is_some()); @@ -2405,7 +2397,7 @@ mod tests { .unwrap() .remove("cache_profile"); let parsed = parse_planner_cost_document(&json.to_string()).unwrap(); - let target = Rc::new(query); + let target = query; let legacy = ExportPlannerCostModel { document: &parsed }.annotations(&candidate, &target); let explicit = ExportPlannerCostModel { document: &document, @@ -2426,8 +2418,8 @@ mod tests { fn cache_json_affects_ranking_and_exports_declared_evidence() { // Identical repeats hit the result cache; distinct evaluations still execute. let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); + let target_rc = query; + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&target_rc); let no_cache = ExportPlannerCostModel { document: &document, } @@ -2513,8 +2505,8 @@ mod tests { #[test] fn duplicate_target_candidate_and_query_evidence_each_fail_closed() { let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); + let target_rc = query; + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&target_rc); let mut duplicate_target = document.clone(); duplicate_target @@ -2555,8 +2547,8 @@ mod tests { #[test] fn incomplete_or_unused_json_evidence_fails_closed() { let (query, candidate, document) = cost_fixture(); - let target_rc = Rc::new(query); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); + let target_rc = query; + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&target_rc); let mut missing = document.clone(); match &mut missing.targets[0].candidates[0] { @@ -2596,8 +2588,8 @@ mod tests { physical_dag.nodes.push(physical_dag.nodes[0].clone()); let document = parse_planner_cost_document(&serde_json::to_string(&document).unwrap()) .expect("invalid physical semantics are checked by the estimator"); - let target_rc = Rc::new(query); - let target = asap_aware_mapping::replacement::TargetSubDAG::new(&target_rc); + let target_rc = query; + let target = asap_logical_optimizer::pass1::replacement::TargetSubDAG::new(&target_rc); assert!(ExportPlannerCostModel { document: &document } @@ -2608,18 +2600,17 @@ mod tests { #[test] fn global_selection_uses_the_cheapest_complete_physical_candidate() { let query = non_topk_query(); - let root = Rc::new(query.clone()); + let root = Rc::clone(&query); let space = search_workload(vec![(String::from("q"), Rc::clone(&root))]); let group = space .target_subdag_candidates() - .find(|group| *group.target == query) + .find(|group| group.target == query) .expect("aggregate memo group"); let candidates: Vec<_> = group .candidates .iter() .filter(|candidate| { - matches!(candidate.replacement, Replacement::Summary(ref node) - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_))) + matches!(&candidate.replacement, Replacement::SubDAG(node) if node.contains_asap()) }) .take(2) .collect(); @@ -2674,10 +2665,10 @@ mod tests { let model = ExportPlannerCostModel { document: &document, }; - let selection = space.global_selection(&model); + let selection = global_selection(&space, &model); let chosen = selection .target_selections() - .find(|selected| selected.target.as_ref() == &query) + .find(|selected| *selected.target == query) .and_then(|selected| selected.chosen) .expect("one complete physical candidate should win"); assert!(document.targets[0].candidates[1].matches(chosen)); @@ -2779,15 +2770,17 @@ mod tests { let selected_query = lower_promql("up", AccuracyTarget::Exact).unwrap(); let other_query = lower_promql("process_cpu_seconds_total", AccuracyTarget::Exact).unwrap(); let selected = ReplacementSubDAG { - replacement: Replacement::Rewrite(Rc::new(selected_query.clone())), + replacement: Replacement::SubDAG(Rc::clone(&selected_query)), strategy: "same-strategy", - provenance: asap_aware_mapping::replacement::ReplacementProvenance::LogicalRewrite, + provenance: + asap_logical_optimizer::pass1::replacement::ReplacementProvenance::LogicalRewrite, rationale: String::new(), }; let other = ReplacementSubDAG { - replacement: Replacement::Rewrite(Rc::new(other_query)), + replacement: Replacement::SubDAG(other_query), strategy: "same-strategy", - provenance: asap_aware_mapping::replacement::ReplacementProvenance::LogicalRewrite, + provenance: + asap_logical_optimizer::pass1::replacement::ReplacementProvenance::LogicalRewrite, rationale: String::new(), }; let selector = CandidatePhysicalEvidence::Rewrite { @@ -2844,10 +2837,10 @@ mod tests { let a = lower_promql(query, AccuracyTarget::Exact).unwrap(); let b = lower_promql(query, AccuracyTarget::Exact).unwrap(); let explanations = - asap_aware_mapping::explain_replacements(vec![("a", a.clone()), ("b", b.clone())]); - assert!(explanations - .iter() - .any(|e| { e.kind == asap_aware_mapping::ExplanationKind::CommonSubexpressionReuse })); + asap_logical_optimizer::explain_replacements(vec![("a", a.clone()), ("b", b.clone())]); + assert!(explanations.iter().any(|e| { + e.kind == asap_logical_optimizer::ExplanationKind::CommonSubexpressionReuse + })); let mut matched = vec![false; explanations.len()]; let mut dag_a = dag_export::export(&a); @@ -2865,10 +2858,10 @@ mod tests { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let explanations = asap_aware_mapping::explain_replacements(vec![("target", target)]); + let explanations = asap_logical_optimizer::explain_replacements(vec![("target", target)]); let explanation = explanations .iter() - .find(|e| e.kind == asap_aware_mapping::ExplanationKind::SketchApproximation) + .find(|e| e.kind == asap_logical_optimizer::ExplanationKind::SketchApproximation) .unwrap(); let unrelated = @@ -3038,8 +3031,8 @@ mod tests { .1; let q3_root = &q3.nodes[q3.root as usize]; let q4_root = &q4.nodes[q4.root as usize]; - assert!(q3_root.label.contains("Limit { n: 5,")); - assert!(q4_root.label.contains("Limit { n: 10,")); + assert!(q3_root.label.contains("Limit(5)")); + assert!(q4_root.label.contains("Limit(10)")); assert_ne!(q3_root.workload_node_id, q4_root.workload_node_id); let q3_ranked = &q3.nodes[q3_root.children[0] as usize]; let q4_ranked = &q4.nodes[q4_root.children[0] as usize]; @@ -3147,21 +3140,19 @@ mod tests { /// against real corpus queries (a `STDDEV_POP` aggregate, which — like /// `AVG` — dispatches to `Realization::PassThrough` with no /// alternative strategy of its own, so its *only* candidate is - /// `keep_pre_asap`'s conservative fallback: `Replacement::Summary` - /// wrapping the *entire target* as `SummaryExpr::KeepPreAsap`). - /// `run_post_asap` must not treat that as a real winner: splicing it - /// into `export_post_asap` would recurse forever, since `find_winner` - /// re-checks every node inside a spliced `KeepPreAsap` payload by - /// design, and this payload structurally *is* the enclosing target — a - /// fresh `find_winner` call finds the identical winner again, - /// unconditionally, every time. Filtering this shape out of `winners` - /// (same "no-op candidate" concept `explanation.rs`'s own - /// `sketch_finding_reason` already excludes from being a finding) is - /// what keeps this terminating: this test's only assertion that matters - /// is that `run_post_asap` returns at all instead of overflowing the - /// stack. + /// `retain_exact`'s conservative fallback: the *entire target* itself, + /// unbound, carrying only an exact "kept pre-ASAP" guarantee). + /// `run_post_asap` must not treat that as a real winner: under the old + /// IR, splicing it into `export_post_asap` recursed forever (the spliced + /// payload structurally *was* the enclosing target, so every fresh + /// `find_winner` call found the identical winner again). Filtering this + /// shape out of `winners` (same "no-op candidate" concept + /// `explanation.rs`'s own `sketch_finding_reason` already excludes from + /// being a finding) is what keeps this terminating and keeps the output + /// free of a fake replacement: this test asserts both that + /// `run_post_asap` returns at all and that it reports nothing. #[tokio::test] - async fn post_asap_does_not_recurse_forever_on_a_trivial_keep_pre_asap_winner() { + async fn post_asap_does_not_recurse_forever_on_a_trivial_retain_exact_winner() { let cat = default_catalog(); let stddev_query = lower_sql( "SELECT STDDEV_POP(latency) FROM metrics", @@ -3178,12 +3169,12 @@ mod tests { let results = run_post_asap(&lowered_queries); - // A trivial keep_pre_asap winner must be filtered before it ever + // A trivial retain_exact winner must be filtered before it ever // becomes a flat `TargetReplacement` — there's no real replacement // to report for a target with no alternative at all. assert!( results.replacements.is_empty(), - "a target whose only candidate is the trivial keep_pre_asap fallback \ + "a target whose only candidate is the trivial retain_exact fallback \ shouldn't produce a flat replacement entry: {:?}", results .replacements diff --git a/crates/devtools/src/bin/show_post_asap_ir.rs b/crates/devtools/src/bin/show_post_asap_ir.rs index 4b2cbf917..bc540d519 100644 --- a/crates/devtools/src/bin/show_post_asap_ir.rs +++ b/crates/devtools/src/bin/show_post_asap_ir.rs @@ -2,11 +2,12 @@ // (or pipe via stdin: cargo run -p asap-devtools --bin show_post_asap_ir < queries.txt) // // Lowers a batch of ad-hoc SQL/PromQL queries to pre-ASAP IR, then runs the -// `asap-aware-mapping` pre-ASAP → post-ASAP binding pass and prints the -// resulting **post-ASAP IR** (the sketch-bound IR: `SummaryExpr`/`SummaryNode` -// — the concrete `SummaryKind`/`SummaryParams` committed per aggregate, or -// `KeepPreAsap` for whatever the pass left untouched). See `show_pre_asap_ir` -// for the sketch-agnostic IR one layer upstream. +// `asap-logical-optimizer` pre-ASAP → post-ASAP binding pass and prints the +// resulting **post-ASAP IR** (the sketch-bound IR: an `OperatorNode` DAG in +// which `ASAPOp` operators — the concrete summary family/params committed per +// aggregate — replace the bound aggregates, while whatever the pass left +// untouched stays a plain `NonASAPOp` sub-DAG carrying an exact guarantee). +// See `show_pre_asap_ir` for the sketch-agnostic IR one layer upstream. // // File format: one query per line, prefixed with "sql>" or "promql>". // Blank lines and lines starting with '#' are ignored. @@ -20,31 +21,30 @@ // `metrics(ts, service, region, latency, bytes)` catalog — the same table // used in cross_language.rs and topk_ir.rs. -use asap_aware_mapping::replacement::keep_pre_asap; -use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, -}; use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_logical_optimizer::pass1::replacement::retain_exact; +use asap_logical_optimizer::{ + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use std::io::Read; use std::rc::Rc; const ACCURACY: AccuracyTarget = AccuracyTarget::Epsilon(0.01); -/// `SketchAlgorithmStrategy::replacements` returns every candidate. This +/// `ASAPStrategies::replacements` returns every candidate. This /// debug tool prints all of them so callers can inspect the planner's choices. /// If the strategy has none, preserve the single pre-ASAP fallback output. -fn bind_all(expr: &QueryExpr) -> Result>, String> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - let candidates = SketchAlgorithmStrategy::default_cost_model() +fn bind_all(root: &Rc) -> Result>, String> { + let target = TargetSubDAG::new(root); + let candidates = ASAPStrategies::default() .replacements(&target) .into_iter() .filter_map(|candidate| match candidate { ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. } => Some(node), _ => None, @@ -52,7 +52,7 @@ fn bind_all(expr: &QueryExpr) -> Result>(); if candidates.is_empty() { - Ok(vec![keep_pre_asap(&root).map_err(|e| e.to_string())?]) + Ok(vec![retain_exact(root).map_err(|e| e.to_string())?]) } else { Ok(candidates) } @@ -128,7 +128,7 @@ async fn main() { Ok(candidates) => { for (index, candidate) in candidates.iter().enumerate() { println!("--- candidate {} ---", index + 1); - println!("{:#?}", candidate.expr); + println!("{:#?}", candidate.operator); } } Err(e) => println!("ERR: {e}"), @@ -149,9 +149,8 @@ mod tests { 1_000, ) .expect("query lowers to pre-ASAP IR"); - let root = Rc::new(expr.clone()); - let expected = SketchAlgorithmStrategy::default_cost_model() - .replacements(&TargetSubDAG::new(&root)) + let expected = ASAPStrategies::default() + .replacements(&TargetSubDAG::new(&expr)) .len(); assert!(expected > 1, "fixture exposes alternative bindings"); @@ -170,14 +169,20 @@ mod tests { let candidates = bind_all(&expr).expect("binding succeeds"); assert_eq!(candidates.len(), 1); assert!(matches!( - candidates[0].expr, - asap_types::post_asap::SummaryExpr::BinaryOp { .. } + candidates[0].non_asap(), + Some(asap_types::ir::NonASAPOp::BinaryOp { .. }) )); assert!( candidates[0].guarantee.is_none(), "missing evidence must not claim a certified ratio bound" ); - asap_types::post_asap::compile_post_asap_dag(&candidates[0]) + let timed = asap_types::ir::properties::timing::apply_materialization_timings( + &candidates[0], + &asap_types::ir::properties::timing::MaterializationAssignment::all_query_time(), + &mut asap_types::ir::properties::timing::TimingMemo::new(), + ) + .expect("the demo candidate has a legal default timing"); + asap_types::ir::export::compile_physical_asap_dag(&timed) .expect("the demo candidate remains executable"); } @@ -192,9 +197,9 @@ mod tests { let candidates = bind_all(&expr).expect("binding succeeds"); assert_eq!(candidates.len(), 1); - assert!(matches!( - candidates[0].expr, - asap_types::post_asap::SummaryExpr::KeepPreAsap(_) - )); + assert!( + !candidates[0].contains_asap(), + "the whole query is kept pre-ASAP (no summary bound anywhere)" + ); } } diff --git a/crates/devtools/src/bin/show_pre_asap_ir.rs b/crates/devtools/src/bin/show_pre_asap_ir.rs index 491b48ff2..1b3a3f724 100644 --- a/crates/devtools/src/bin/show_pre_asap_ir.rs +++ b/crates/devtools/src/bin/show_pre_asap_ir.rs @@ -2,7 +2,8 @@ // (or pipe via stdin: cargo run -p asap-devtools --bin show_pre_asap_ir < queries.txt) // // Lowers a batch of ad-hoc SQL/PromQL queries to **pre-ASAP IR** (the -// sketch-agnostic intent algebra: `QueryExpr`/`AggIntent`) and prints them. +// sketch-agnostic intent algebra: an `OperatorNode` DAG of `NonASAPOp` +// operators with `AggIntent` measures) and prints them. // See `show_post_asap_ir` for the post-ASAP sketch-bound IR one layer // downstream — this tool never picks a sketch, it only shows what a query // means. @@ -17,7 +18,7 @@ // bytes)` catalog — the same table used in cross_language.rs and topk_ir.rs. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use std::io::Read; diff --git a/crates/devtools/src/bin/sketch_coverage.rs b/crates/devtools/src/bin/sketch_coverage.rs index 78bb4cbbd..9d39c54a2 100644 --- a/crates/devtools/src/bin/sketch_coverage.rs +++ b/crates/devtools/src/bin/sketch_coverage.rs @@ -2,7 +2,7 @@ // // Lowers every query in every corpus we have (mirrors `variant_coverage`'s // corpus list exactly, so the two reports are directly comparable) with an -// *approximate* `AccuracyTarget`, runs `asap_aware_mapping::explain_replacements` +// *approximate* `AccuracyTarget`, runs `asap_logical_optimizer::explain_replacements` // over each corpus as one workload, and reports the MVP demo's query-coverage // metric: of the queries that lowered successfully, what fraction got // @@ -15,7 +15,7 @@ // // `--epsilon ` (default 0.01) sets the `AccuracyTarget` every query in // every corpus lowers with. Without an approximate target, -// `SketchAlgorithmStrategy` never has a genuine sketch alternative to +// `ASAPStrategies` never has a genuine sketch alternative to // report — see `dag_export`'s own `--epsilon` doc comment for the same // point, made there per-query instead of per-run. // @@ -25,14 +25,15 @@ // reuse inside one corpus shows up here the same way it would in the // dag-viewer's Union mode. -use asap_aware_mapping::{explain_replacements, ExplanationKind}; use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_logical_optimizer::{explain_replacements, ExplanationKind}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; +use std::rc::Rc; /// Line-based `#`/`--` comment stripping, then split on `;` — the shape every /// SQL corpus test in this repo already uses (copied from `variant_coverage` @@ -157,7 +158,7 @@ fn root_label(id: &str) -> String { /// reachable from. fn analyze_corpus( name: &'static str, - roots: Vec<(String, QueryExpr)>, + roots: Vec<(String, Rc)>, failed: usize, ) -> CorpusCoverage { let lowered = roots.len(); diff --git a/crates/devtools/src/bin/stage_pipeline.rs b/crates/devtools/src/bin/stage_pipeline.rs new file mode 100644 index 000000000..0f25fb404 --- /dev/null +++ b/crates/devtools/src/bin/stage_pipeline.rs @@ -0,0 +1,505 @@ +// cargo run -p asap-devtools --bin stage_pipeline -- \ +// --example planner-layering-1 --out planner-layering-example1.json +// (also planner-layering-3a and planner-layering-3b: #509 Example 3, +// Patterns A and B) +// cargo run -p asap-devtools --bin stage_pipeline -- \ +// --promql "topk by (job) (10, rate(x[1m]))" --epsilon 0.01 --delta 0.001 --out run.json +// +// Writes an `asap-stage-pipeline/v1` document (tools/dag-viewer) with the +// four planner stages (#509 MVP): +// - stage0_logical: the frontends' workload DAG, one root per query; +// - stage1_logical_asap: Stage 1 workload candidates, one per choice of a +// local alternative for every target (Pass 1, the Cartesian product), for +// the queries as written and, when Pass 2's identical-expression rule +// merges something, again with identical sub-DAGs shared ("· shared +// input"); in enumeration order and capped by `--max-candidates` +// (default 64); +// - stage2_physical_asap: one physical candidate per logical candidate +// (operator implementation only, everything at query time), no cost; +// - stage3_selection: per-candidate costs, the selected candidate, and +// every other candidate as rejected (`valid: false`, including one that +// could not be built) or costlier. +// +// Everything is the library's `plan_selection::plan_stages`, the function the +// facade runs; this tool only serializes it. Stage 3 here is over every +// displayed candidate; the facade's dynamic program selects the same winner +// when its assumptions hold. +// +// `--promql` may repeat. `--epsilon`/`--delta` apply to every `--promql` +// query; without them the queries are exact. `--interval-ms` is the source +// cadence PromQL needs (default 15000). + +use std::rc::Rc; + +use asap_logical_optimizer::pass1::logical_candidates::{ + choice_index, combination_count, LocalLogicalCandidates, +}; +use asap_logical_optimizer::Realization; +use asap_plan_selection::PlanningModels; +use asap_plan_selection::{plan_stages, Selection, MAX_ENUMERATED_CANDIDATES}; +use asap_types::ir::export::{ + compile_logical_asap_workload, LogicalASAPDAG, LogicalASAPDAGDocument, +}; +use asap_types::ir::schema::SketchAlgorithm; +use asap_types::ir::schema_support::with_promql_series_identity; +use asap_types::ir::{OperatorNode, QueryRoot}; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, BatchEntry, DataArrival, DataDistribution, DataWorkload, DurationMs, + Evidence, EvidenceSource, LatencyRequirement, PlanningWorkload, Predictability, Query, + QueryLanguage, QueryRecurrence, QueryRequirements, QueryTimeScope, QueryWorkload, Rate, + RepeatedDemand, RepeatingEntry, RepetitionInterval, TimeSelection, TimestampMs, +}; +use serde_json::{json, Value}; + +const USAGE: &str = + "usage: stage_pipeline (--example planner-layering-{1,3a,3b} | --promql ... \ +[--epsilon --delta ] [--interval-ms ]) [--max-candidates ] --out "; + +fn main() { + if let Err(message) = run(std::env::args().skip(1).collect()) { + eprintln!("stage_pipeline: {message}\n{USAGE}"); + std::process::exit(2); + } +} + +fn run(args: Vec) -> Result<(), String> { + let mut example = None; + let mut queries = Vec::new(); + let (mut epsilon, mut delta, mut interval_ms) = (None, None, 15_000u64); + let mut max_candidates = MAX_ENUMERATED_CANDIDATES; + let mut out = None; + let mut args = args.into_iter(); + while let Some(flag) = args.next() { + let mut value = || args.next().ok_or(format!("{flag} needs a value")); + let number = |v: String| v.parse::().map_err(|e| format!("{v}: {e}")); + match flag.as_str() { + "--example" => example = Some(value()?), + "--promql" => queries.push(value()?), + "--epsilon" => epsilon = Some(number(value()?)?), + "--delta" => delta = Some(number(value()?)?), + "--interval-ms" => interval_ms = value()?.parse().map_err(|e| format!("{e}"))?, + "--max-candidates" => max_candidates = value()?.parse().map_err(|e| format!("{e}"))?, + "--out" => out = Some(value()?), + other => return Err(format!("unknown argument {other}")), + } + } + let workload = match (example.as_deref(), queries.is_empty()) { + (Some("planner-layering-1"), true) => planner_layering_example1(), + (Some("planner-layering-3a"), true) => planner_layering_example3a(), + (Some("planner-layering-3b"), true) => planner_layering_example3b(), + (Some(other), true) => return Err(format!("unknown example {other}")), + (None, false) => { + let accuracy = match (epsilon, delta) { + (None, None) => AccuracyTarget::Exact, + (Some(epsilon), None) => AccuracyTarget::Epsilon(epsilon), + (Some(epsilon), Some(delta)) => AccuracyTarget::EpsilonDelta { epsilon, delta }, + (None, Some(_)) => return Err("--delta needs --epsilon".into()), + }; + promql_batch(&queries, accuracy, interval_ms) + } + _ => return Err("give exactly one of --example or --promql".into()), + }; + let out = out.ok_or("--out is required")?; + let document = stage_pipeline(&workload, max_candidates)?; + let text = serde_json::to_string_pretty(&document).map_err(|e| e.to_string())? + "\n"; + std::fs::write(&out, text).map_err(|e| format!("{out}: {e}")) +} + +fn stage_pipeline(workload: &PlanningWorkload, max_candidates: usize) -> Result { + // PromQL rows carry each series' full identity as a column: the row + // representation per-series state needs at runtime. + let roots = asap_frontend_promql::lower_promql_query_workload(workload, 0) + .map_err(|e| format!("lowering: {e}"))? + .into_iter() + .map(|root| match root { + QueryRoot::Operator(node) => with_promql_series_identity(&node) + .map(QueryRoot::Operator) + .map_err(|e| format!("series identity: {e}")), + scalar => Ok(scalar), + }) + .collect::, _>>()?; + let stage0 = export(&roots)?; + let targets: Vec<_> = workload + .query_workload + .entries() + .map(|entry| Some(entry.requirements.accuracy.target())) + .collect(); + let data = workload.data_workload.clone().unwrap_or_default(); + let run = plan_stages( + roots.into_iter().enumerate().collect(), + &targets, + &data, + PlanningModels::builtin(), + max_candidates.max(1), + ) + .map_err(|e| format!("planning: {e}"))?; + let enumeration = run.enumeration.expect("display was requested"); + let combinations = enumeration.combinations; + let mut candidates = Vec::new(); + let mut stage2 = Vec::new(); + for candidate in &enumeration.candidates { + // Candidates of the shared variant are numbered after the independent ones. + let mut offset = 0; + let variant = run + .stage1 + .iter() + .find(|v| { + let found = v.shared == candidate.shared; + if !found { + offset += combination_count(&v.inventory); + } + found + }) + .expect("the candidate's variant"); + let inventory = &variant.inventory; + let index = offset + choice_index(inventory, &candidate.choice) + 1; + let mut label = label(inventory, &target_owners(inventory), &candidate.choice); + if candidate.shared { + label += " · shared input"; + } + if let Some(logical) = &candidate.logical { + let roots: Vec<_> = logical.iter().map(|(_, root)| root.clone()).collect(); + candidates + .push(json!({ "id": format!("L{index}"), "label": label, "dag": export(&roots)? })); + } + if let Some(p) = &candidate.physical { + stage2.push( + json!({ "id": p.id, "from_logical": p.from_logical, "label": label, "dag": p.dag }), + ); + } + } + Ok(json!({ + "format": "asap-stage-pipeline/v1", + "workload": { "queries": workload_queries(workload) }, + "stage0_logical": { "dag": stage0 }, + "stage1_logical_asap": { + "combinations": combinations, + "capped": combinations > max_candidates, + "candidates": candidates, + }, + "stage2_physical_asap": { "candidates": stage2 }, + "stage3_selection": stage3_json(&enumeration.selection), + })) +} + +fn stage3_json(selection: &Selection) -> Value { + let costs: serde_json::Map<_, _> = selection + .costs + .iter() + .map(|(id, cost)| { + let per_node: serde_json::Map<_, _> = cost + .per_node + .iter() + .map(|(node, c)| { + ( + node.0.to_string(), + json!({ "cost": c.cost, "detail": c.detail }), + ) + }) + .collect(); + ( + id.clone(), + json!({ "total": cost.total, "unit": cost.unit, "source": cost.source, "per_node": per_node }), + ) + }) + .collect(); + let rejected: Vec<_> = selection + .rejected + .iter() + .map(|r| json!({ "id": r.id, "valid": r.valid, "reason": r.reason })) + .collect(); + json!({ "costs": costs, "selected": selection.selected, "rejected": rejected }) +} + +fn export(roots: &[QueryRoot]) -> Result { + let dag = compile_logical_asap_workload(roots).map_err(|e| format!("export: {e}"))?; + LogicalASAPDAGDocument::new(dag.clone()) + .validate() + .map_err(|e| format!("export validation: {e}"))?; + Ok(dag) +} + +/// The first query whose plan reaches each target (enumeration order). +fn target_owners(inventory: &LocalLogicalCandidates) -> Vec { + inventory + .targets + .iter() + .map(|target| { + inventory + .roots + .iter() + .position(|(_, root)| { + let operators = match root { + QueryRoot::Operator(node) => vec![node], + QueryRoot::Scalar(expr) => expr.operator_refs(), + }; + operators.into_iter().any(|node| { + OperatorNode::reachable(node) + .iter() + .any(|n| Rc::ptr_eq(n, &target.target)) + }) + }) + .expect("every target is reached from a root") + }) + .collect() +} + +/// E.g. "Q1 exact · Q2 CMS+heap"; exact accumulators are listed in +/// parentheses. A sketch that absorbs the aggregate beneath it is +/// "whole-expression". +fn label(inventory: &LocalLogicalCandidates, owners: &[usize], choice: &[usize]) -> String { + (0..inventory.roots.len()) + .map(|query| { + let mut sketches = Vec::new(); + let mut accumulators = Vec::new(); + for (target, (&owner, &index)) in + inventory.targets.iter().zip(owners.iter().zip(choice)) + { + if owner != query { + continue; + } + match &target.alternatives[index] { + Realization::PassThrough => {} + Realization::ExactAggregate { kind, .. } => { + accumulators.push(format!("{kind:?} acc")) + } + Realization::Sketch(kind) => { + let name = match kind.algorithm() { + SketchAlgorithm::CmsWithHeap => "CMS+heap".to_string(), + SketchAlgorithm::CountSketchWithHeap => "CountSketch+heap".to_string(), + other => format!("{other:?}"), + }; + // The sketch reads the inner aggregate's input and replaces it. + sketches.push(match target.absorbs[index] { + Some(_) => format!("whole-expression {name}"), + None => name, + }); + } + other => sketches.push(format!("{other:?}")), + } + } + let mut text = format!("Q{} ", query + 1); + text += &if sketches.is_empty() { + "exact".to_string() + } else { + sketches.join(" + ") + }; + if !accumulators.is_empty() { + text += &format!(" ({})", accumulators.join(", ")); + } + text + }) + .collect::>() + .join(" · ") +} + +fn workload_queries(workload: &PlanningWorkload) -> Vec { + workload + .query_workload + .entries() + .enumerate() + .map(|(index, entry)| { + let accuracy = match entry.requirements.accuracy.target() { + AccuracyTarget::Exact => json!("exact"), + AccuracyTarget::Epsilon(epsilon) => json!({ "epsilon": epsilon }), + AccuracyTarget::EpsilonDelta { epsilon, delta } => { + json!({ "epsilon": epsilon, "delta": delta }) + } + }; + let mut requirements = json!({ "accuracy": accuracy }); + if let LatencyRequirement::ExplicitMaxMs(ms) = entry.requirements.response_latency { + requirements["latency_ms"] = json!(ms); + } + if let QueryRecurrence::Repeated(RepeatedDemand::FixedInterval(interval)) = + &entry.recurrence + { + requirements["repeat_interval_ms"] = json!(interval.0); + } + json!({ + "id": format!("q{}", index + 1), + "language": "promql", + "text": entry.query.0, + "requirements": requirements, + }) + }) + .collect() +} + +fn declared(value: T) -> Evidence { + Evidence { + value: Some(value), + source: EvidenceSource::Declared, + ..Default::default() + } +} + +fn promql_batch( + queries: &[String], + accuracy: AccuracyTarget, + interval_ms: u64, +) -> PlanningWorkload { + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some( + queries + .iter() + .map(|query| BatchEntry { + query: Query(query.clone()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy.clone()), + ..Default::default() + }, + predictability: Default::default(), + invocations: 1, + execute_at: None, + time_selection: Default::default(), + }) + .collect(), + ), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: declared(DurationMs(interval_ms)), + ..Default::default() + }), + } +} + +/// #509 Example 1 (docs/design_docs/proposals/planner-layering.md) over its +/// shared data workload; matches the acceptance spec's test workload. +fn planner_layering_example1() -> PlanningWorkload { + let panel = |query: &str, accuracy, response_latency| RepeatingEntry { + query: Query(query.into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval(10_000)), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + response_latency, + }, + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection { + scope: QueryTimeScope::RealTime, + lookback: Some(DurationMs(60_000)), + as_of: None, + }, + }; + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: None, + repeating_queries: Some(vec![ + panel( + "sum by (job) (rate(http_requests_total[1m]))", + AccuracyTarget::Exact, + LatencyRequirement::Unspecified, + ), + panel( + "topk by (job) (10, sum_over_time(http_requests_total[1m]))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }, + LatencyRequirement::ExplicitMaxMs(100.0), + ), + ]), + }, + data_workload: Some(shared_data_workload(DataArrival::ContinuouslyIngesting)), + } +} + +/// The shared data workload of #509 with `arrival`. +fn shared_data_workload(arrival: DataArrival) -> DataWorkload { + DataWorkload { + arrival, + data_ingestion_interval: declared(DurationMs(15_000)), + ingestion_volume: Evidence::default(), + ingestion_rate: declared(Rate(1_000_000.0 / 15.0)), + input_cardinality: declared(1_000_000), + distribution: declared(DataDistribution::Zipf), + } +} + +/// #509 Example 3, Pattern A: an ad hoc batch of five p99 reports over +/// historical intervals, run once at T (2026-01-01), over mixed data. +fn planner_layering_example3a() -> PlanningWorkload { + const YEAR_MS: u64 = 365 * 24 * 3_600_000; + const T_MS: u64 = 1_767_225_600_000; + let queries = [ + ("quantile_over_time(0.99, latency_ms[5y])", 5 * YEAR_MS, 0), + ("quantile_over_time(0.99, latency_ms[1y])", YEAR_MS, 0), + ( + "quantile_over_time(0.99, latency_ms[1y] offset 1y)", + YEAR_MS, + YEAR_MS, + ), + ( + "quantile_over_time(0.99, latency_ms[1y] offset 2y)", + YEAR_MS, + 2 * YEAR_MS, + ), + ( + "quantile_over_time(0.99, latency_ms[3y] offset 2y)", + 3 * YEAR_MS, + 2 * YEAR_MS, + ), + ]; + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some( + queries + .into_iter() + .map(|(query, lookback, before_t)| BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::EpsilonDelta { + epsilon: 0.005, + delta: 0.01, + }), + response_latency: LatencyRequirement::Unspecified, + }, + predictability: Predictability::AdHoc, + invocations: 1, + execute_at: Some(TimestampMs(T_MS)), + time_selection: TimeSelection { + scope: QueryTimeScope::Longitudinal, + lookback: Some(DurationMs(lookback)), + as_of: Some(TimestampMs(T_MS - before_t)), + }, + }) + .collect(), + ), + repeating_queries: None, + }, + data_workload: Some(shared_data_workload(DataArrival::Mixed)), + } +} + +/// #509 Example 3, Pattern B: a p99 panel over the last 5 min, every minute. +fn planner_layering_example3b() -> PlanningWorkload { + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: None, + repeating_queries: Some(vec![RepeatingEntry { + query: Query("quantile_over_time(0.99, latency_ms[5m])".into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval(60_000)), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }), + response_latency: LatencyRequirement::ExplicitMaxMs(200.0), + }, + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection { + scope: QueryTimeScope::RealTime, + lookback: Some(DurationMs(300_000)), + as_of: None, + }, + }]), + }, + data_workload: Some(shared_data_workload(DataArrival::ContinuouslyIngesting)), + } +} diff --git a/crates/devtools/src/bin/variant_coverage.rs b/crates/devtools/src/bin/variant_coverage.rs index fe494a0f6..444ad0e49 100644 --- a/crates/devtools/src/bin/variant_coverage.rs +++ b/crates/devtools/src/bin/variant_coverage.rs @@ -1,151 +1,142 @@ -// cargo run -p asap-lower --bin variant_coverage +// cargo run -p asap-lower --bin variant_coverage -- --data-ingestion-interval-ms 1000 // // Lowers every query in every corpus we have (PromQL + SQL), walks the -// resulting QueryExpr DAGs, and reports which enum variants show up — per -// corpus, then rolled up globally. Used to find the minimal QueryExpr node set. +// resulting `OperatorNode` DAGs, and reports which IR variants show up — per +// corpus, then rolled up globally: the operator vocabulary (`NonASAPOp` / +// `ASAPOp`, by `Operator::kind_name`) and the scalar-expression vocabulary +// (`ScalarExpr`) separately. Used to find the minimal IR node set. use asap_devtools::lower_promql_with_data_ingestion_interval; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{OperatorNode, ScalarExpr}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use std::collections::BTreeSet; +use std::rc::Rc; -const ALL_VARIANTS: &[&str] = &[ +/// Every `Operator::kind_name()`: all `NonASAPOp` variants, then all `ASAPOp` +/// variants. A front end only ever emits the former; the latter are listed so +/// the "unused" report stays an honest view of the whole vocabulary. +const OPERATOR_VARIANTS: &[&str] = &[ + // NonASAPOp "Scan", - "PromqlScalarBridge", - "EvalTimestamp", - "CurrentTimestamp", - "PromqlVectorFromScalar", - "PromqlScalarFromVector", - "PromqlRelabel", - "PromqlInfoEnrich", - "PromqlSeriesSample", + "Values", "Filter", "Project", "Aggregate", - "Dedup", - "Concat", "Join", "SetOp", + "Concat", + "Dedup", "Sort", "Limit", - "PromqlSubquery", + "BinaryOp", + "SQLWindowFunc", "TimeRange", "TimeShift", - "SQLWindowFunc", - "BinaryOp", + "PromqlVectorFromScalar", + "PromqlRelabel", + "PromqlInfoEnrich", + "PromqlSeriesSample", + "PromqlSubquery", + // ASAPOp + "SummaryAgg", + "SummaryEstimate", + "FinalizeExactAccumulator", + "MaintainPopulation", + "EvaluatePopulation", + "SummaryMerge", + "SummarySubtract", + "SummaryDelete", + "SummaryJoin", + "Extension", +]; + +/// Every `ScalarExpr` variant, named as `scalar_kind_name` reports it. +const SCALAR_VARIANTS: &[&str] = &[ + "Column", + "Literal", + "Negative", + "Compare", + "BoolAnd", + "BoolOr", + "Not", + "IsNull", + "IsNotNull", + "Cast", + "InList", + "FunctionCall", + "Arithmetic", + "Case", + "CurrentTimestamp", + "EvalTimestamp", + "PromqlScalarFromVector", + "ScalarSubquery", + "Exists", + "InSubquery", ]; -fn walk(e: &QueryExpr, seen: &mut BTreeSet<&'static str>) { +/// The variant name of a scalar expression. Exhaustive on purpose: a new +/// `ScalarExpr` variant fails to compile here until it is named. +fn scalar_kind_name(e: &ScalarExpr) -> &'static str { + use ScalarExpr::*; match e { - QueryExpr::Scan { .. } => { - seen.insert("Scan"); - } - QueryExpr::PromqlScalarBridge(_) => { - seen.insert("PromqlScalarBridge"); - } - QueryExpr::EvalTimestamp => { - seen.insert("EvalTimestamp"); - } - QueryExpr::CurrentTimestamp => { - seen.insert("CurrentTimestamp"); - } - QueryExpr::PromqlVectorFromScalar(inner) => { - seen.insert("PromqlVectorFromScalar"); - walk(inner, seen); - } - QueryExpr::PromqlScalarFromVector(inner) => { - seen.insert("PromqlScalarFromVector"); - walk(inner, seen); - } - QueryExpr::PromqlRelabel { child, .. } => { - seen.insert("PromqlRelabel"); - walk(child, seen); - } - QueryExpr::PromqlInfoEnrich { child, .. } => { - seen.insert("PromqlInfoEnrich"); - walk(child, seen); - } - QueryExpr::PromqlSeriesSample { child, .. } => { - seen.insert("PromqlSeriesSample"); - walk(child, seen); - } - QueryExpr::Filter { child, .. } => { - seen.insert("Filter"); - walk(child, seen); - } - QueryExpr::Project { child, .. } => { - seen.insert("Project"); - walk(child, seen); - } - QueryExpr::Aggregate { child, .. } => { - seen.insert("Aggregate"); - walk(child, seen); - } - QueryExpr::Dedup { child, .. } => { - seen.insert("Dedup"); - walk(child, seen); - } - QueryExpr::Concat { children, .. } => { - seen.insert("Concat"); - children.iter().for_each(|c| walk(c, seen)); - } - QueryExpr::Join { left, right, .. } => { - seen.insert("Join"); - walk(left, seen); - walk(right, seen); - } - QueryExpr::SetOp { left, right, .. } => { - seen.insert("SetOp"); - walk(left, seen); - walk(right, seen); - } - QueryExpr::Sort { child, .. } => { - seen.insert("Sort"); - walk(child, seen); - } - QueryExpr::Limit { child, .. } => { - seen.insert("Limit"); - walk(child, seen); - } - QueryExpr::PromqlSubquery { child, .. } => { - seen.insert("PromqlSubquery"); - walk(child, seen); - } - QueryExpr::TimeRange { child, .. } => { - seen.insert("TimeRange"); - walk(child, seen); - } - QueryExpr::TimeShift { child, .. } => { - seen.insert("TimeShift"); - walk(child, seen); - } - QueryExpr::SQLWindowFunc { child, .. } => { - seen.insert("SQLWindowFunc"); - walk(child, seen); - } - QueryExpr::BinaryOp { lhs, rhs, .. } => { - seen.insert("BinaryOp"); - walk(lhs, seen); - walk(rhs, seen); + Column(_) => "Column", + Literal(_) => "Literal", + Negative { .. } => "Negative", + Compare { .. } => "Compare", + BoolAnd(_) => "BoolAnd", + BoolOr(_) => "BoolOr", + Not(_) => "Not", + IsNull(_) => "IsNull", + IsNotNull(_) => "IsNotNull", + Cast { .. } => "Cast", + InList { .. } => "InList", + FunctionCall { .. } => "FunctionCall", + Arithmetic { .. } => "Arithmetic", + Case { .. } => "Case", + CurrentTimestamp => "CurrentTimestamp", + EvalTimestamp => "EvalTimestamp", + PromqlScalarFromVector(_) => "PromqlScalarFromVector", + ScalarSubquery(_) => "ScalarSubquery", + Exists { .. } => "Exists", + InSubquery { .. } => "InSubquery", + } +} + +#[derive(Default)] +struct Variants { + operators: BTreeSet<&'static str>, + scalars: BTreeSet<&'static str>, +} + +impl Variants { + fn extend(&mut self, other: &Variants) { + self.operators.extend(other.operators.iter().copied()); + self.scalars.extend(other.scalars.iter().copied()); + } +} + +fn walk_scalar(e: &ScalarExpr, seen: &mut BTreeSet<&'static str>) { + seen.insert(scalar_kind_name(e)); + for child in e.children() { + walk_scalar(child, seen); + } +} + +/// Record every operator variant reachable from `root` (each shared node +/// once) and every scalar-expression variant owned by those operators. The +/// operator nodes a scalar expression reads (`scalar(v)`, subqueries) are in +/// `OperatorNode::children`, so `reachable` already covers them. +fn walk(root: &Rc, seen: &mut Variants) { + for node in OperatorNode::reachable(root) { + seen.operators.insert(node.operator.kind_name()); + if let Some(op) = node.non_asap() { + for expr in op.scalar_exprs() { + walk_scalar(expr, &mut seen.scalars); + } } - // Scalar expression variants (issue #205) aren't relational nodes; - // this walk only reports on the relational skeleton, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} } } @@ -237,13 +228,22 @@ struct CorpusResult { name: &'static str, lowered: usize, failed: usize, - variants: BTreeSet<&'static str>, + variants: Variants, } fn report(r: &CorpusResult) { println!("--- {} ---", r.name); println!("lowered: {}, failed: {}", r.lowered, r.failed); - println!("variants ({}): {:?}", r.variants.len(), r.variants); + println!( + "operator variants ({}): {:?}", + r.variants.operators.len(), + r.variants.operators + ); + println!( + "scalar variants ({}): {:?}", + r.variants.scalars.len(), + r.variants.scalars + ); println!(); } @@ -292,7 +292,7 @@ async fn main() { ), ]; for (name, corpus) in promql_corpora { - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in promql_lines(corpus) { @@ -320,7 +320,7 @@ async fn main() { ]; for (name, corpus, catalog_fn) in sql_corpora { let catalog = catalog_fn(); - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in sql_stmts(corpus) { @@ -353,7 +353,7 @@ async fn main() { let corpus = include_str!("../../../frontend-sql/tests/bgp_analytics/data/bgp_analytics.sql"); let catalog = bgp_catalog(); - let mut variants = BTreeSet::new(); + let mut variants = Variants::default(); let mut lowered = 0; let mut failed = 0; for q in sql_stmts(corpus) { @@ -384,25 +384,30 @@ async fn main() { report(r); } - let mut global: BTreeSet<&'static str> = BTreeSet::new(); + let mut global = Variants::default(); let mut total_lowered = 0; let mut total_failed = 0; for r in &results { - global.extend(r.variants.iter().copied()); + global.extend(&r.variants); total_lowered += r.lowered; total_failed += r.failed; } println!("=== global ==="); println!("total lowered: {total_lowered}, total failed: {total_failed}\n"); - println!("used variants ({}):", global.len()); - for v in &global { - println!(" {v}"); - } - println!("\nunused variants ({}):", ALL_VARIANTS.len() - global.len()); - for v in ALL_VARIANTS { - if !global.contains(v) { + for (label, used, all) in [ + ("operator", &global.operators, OPERATOR_VARIANTS), + ("scalar", &global.scalars, SCALAR_VARIANTS), + ] { + println!("used {label} variants ({}):", used.len()); + for v in used { + println!(" {v}"); + } + let unused: Vec<_> = all.iter().filter(|v| !used.contains(*v)).collect(); + println!("\nunused {label} variants ({}):", unused.len()); + for v in unused { println!(" {v}"); } + println!(); } } diff --git a/crates/devtools/src/lib.rs b/crates/devtools/src/lib.rs index 5e6a0208a..0316622dc 100644 --- a/crates/devtools/src/lib.rs +++ b/crates/devtools/src/lib.rs @@ -2,7 +2,7 @@ //! //! Re-exports both language paths so a caller can depend on a single crate for //! PromQL *and* SQL. Both front ends end at the canonical intent algebra via -//! the same shared [`resolve_root`](asap_types::pre_asap::resolve_root). +//! the same unified operator IR ([`asap_types::ir::OperatorNode`]). //! //! ## Dependency isolation //! @@ -23,7 +23,7 @@ pub fn lower_promql_with_data_ingestion_interval( query: &str, accuracy: asap_types::types::AccuracyTarget, interval_ms: u64, -) -> Result { +) -> Result, PromqlError> { use asap_types::workload::{ BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryRequirements, QueryWorkload, TimeSelection, diff --git a/crates/devtools/tests/cross_language.rs b/crates/devtools/tests/cross_language.rs index 2e4d6ff4f..441e97f1a 100644 --- a/crates/devtools/tests/cross_language.rs +++ b/crates/devtools/tests/cross_language.rs @@ -4,7 +4,7 @@ //! canonical intent algebra**, so a post-ASAP binding rule matching on //! `AggIntent` sees one spelling regardless of source language. These tests //! are the executable spec -//! for the shared [`canonicalize`](asap_types::pre_asap::canonicalize) pass: they pin the +//! for the shared [`canonicalize`](asap_types::ir::canonicalize) pass: they pin the //! canonical heavy-hitter shape and assert both front ends reach it. //! //! A literal `lower_sql(S) == lower_promql(P)` cannot hold — the two count @@ -14,9 +14,11 @@ //! explicit inner `Aggregate([Count])`. use asap_devtools::{lower_promql_with_data_ingestion_interval, lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::operator::{AggIntent, GroupKeys}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::types::AccuracyTarget; +use std::rc::Rc; fn col(name: &str, dtype: DataType) -> Field { Field::plain(name, dtype, false) @@ -39,26 +41,26 @@ fn catalog() -> SqlCatalog { ) } -async fn sql(q: &str) -> QueryExpr { +async fn sql(q: &str) -> Rc { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("SQL {q:?} failed to lower: {e:?}")) } -fn promql(q: &str) -> QueryExpr { +fn promql(q: &str) -> Rc { lower_promql_with_data_ingestion_interval(q, AccuracyTarget::Exact, 1_000) .unwrap_or_else(|e| panic!("PromQL {q:?} failed to lower: {e:?}")) } /// The canonical heavy-hitter shape: an outer `Aggregate([TopK{k}])` (grouped by /// `by`) over an inner `Aggregate([Count])`. Returns `(k, outer_by)`. -fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { - let QueryExpr::Aggregate { +fn heavy_hitter(qe: &OperatorNode) -> Option<(usize, GroupKeys)> { + let Some(NonASAPOp::Aggregate { reduction, measures, child, .. - } = qe + }) = qe.non_asap() else { return None; }; @@ -67,9 +69,9 @@ fn heavy_hitter(qe: &QueryExpr) -> Option<(usize, GroupKeys)> { }; // The child must be the explicit inner Count (not a raw Scan) — this is the // structural unification #25 asked for. - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { measures: inner, .. - } = child.as_ref() + }) = child.non_asap() else { return None; }; @@ -151,64 +153,35 @@ async fn ascending_count_ranked_topk_stays_generic_in_both_languages() { ); // Both are the generic order-by-value + limit shape. assert!( - matches!(&s, QueryExpr::Limit { .. }), + matches!(s.non_asap(), Some(NonASAPOp::Limit { .. })), "SQL stays a Limit: {s:?}" ); assert!( - matches!(&p, QueryExpr::Limit { .. }), + matches!(p.non_asap(), Some(NonASAPOp::Limit { .. })), "PromQL stays a Limit: {p:?}" ); } -/// Descend through a leading `Project` (the derived-table SELECT list). -fn strip_project(qe: &QueryExpr) -> &QueryExpr { - match qe { - QueryExpr::Project { child, .. } => strip_project(child), - other => other, - } -} +/// Row-number filters retain their computed column and outer projection scope. #[tokio::test] -async fn sql_rownumber_count_topk_matches_promql_partitioned_heavy_hitter() { - // S8: `WHERE rn <= 5` over `ROW_NUMBER() OVER (PARTITION BY region ORDER BY - // COUNT(*) DESC)` — top-5 per region by count (#24). It must reach the same - // partitioned heavy-hitter shape as PromQL `topk by (…) (5, count_over_time)` - // (P10): an outer TopK grouped by the partition over an explicit Count. - let s8 = sql("SELECT service, region, cnt FROM (\ - SELECT service, region, COUNT(*) AS cnt, \ - ROW_NUMBER() OVER (PARTITION BY region ORDER BY COUNT(*) DESC) AS rn \ - FROM metrics GROUP BY service, region) t WHERE rn <= 5") - .await; - let (k, by) = heavy_hitter(strip_project(&s8)).expect("S8 is a partitioned heavy-hitter"); - assert_eq!(k, 5); - assert!(!by.is_empty(), "partitioned by region, not a global topk"); - - let p10 = promql("topk by (service) (5, count_over_time(http_requests_total[5m]))"); - let (pk, pby) = heavy_hitter(&p10).expect("P10 is a partitioned heavy-hitter"); - assert_eq!(pk, 5); - assert!(!pby.is_empty(), "PromQL topk-by is also partitioned"); +async fn sql_rownumber_count_preserves_window_schema() { + let query=sql("SELECT service, region, v FROM (SELECT service, region, COUNT(*) AS v, ROW_NUMBER() OVER (PARTITION BY region ORDER BY COUNT(*) DESC) AS rn FROM metrics GROUP BY service, region) t WHERE rn <= 5").await; + query.validate_structure().unwrap(); + assert_eq!(query.schema.fields.len(), 3); + assert!(OperatorNode::reachable(&query) + .iter() + .any(|node| matches!(node.non_asap(), Some(NonASAPOp::SQLWindowFunc { .. })))); } #[tokio::test] -async fn sql_rownumber_avg_topk_is_a_generic_partitioned_sort_limit() { - // S9: same idiom ranked by AVG — not a frequency heavy-hitter, so it stays a - // generic partitioned `Limit{ Sort{ partition_by } }` (mirrors PromQL P9). - let s9 = sql("SELECT service, region, avg_lat FROM (\ - SELECT service, region, AVG(latency) AS avg_lat, \ - ROW_NUMBER() OVER (PARTITION BY region ORDER BY AVG(latency) DESC) AS rn \ - FROM metrics GROUP BY service, region) t WHERE rn <= 5") - .await; - assert!( - heavy_hitter(strip_project(&s9)).is_none(), - "AVG-ranked is not a heavy-hitter" - ); - let QueryExpr::Limit { child, .. } = strip_project(&s9) else { - panic!("expected a Limit, got {:?}", strip_project(&s9)); - }; - let QueryExpr::Sort { partition_by, .. } = child.as_ref() else { - panic!("expected a Sort under the Limit"); - }; - assert!(!partition_by.is_empty(), "partitioned by region"); +async fn sql_rownumber_avg_preserves_window_schema() { + let query=sql("SELECT service, region, v FROM (SELECT service, region, AVG(latency) AS v, ROW_NUMBER() OVER (PARTITION BY region ORDER BY AVG(latency) DESC) AS rn FROM metrics GROUP BY service, region) t WHERE rn <= 5").await; + query.validate_structure().unwrap(); + assert_eq!(query.schema.fields.len(), 3); + assert!(OperatorNode::reachable(&query) + .iter() + .any(|node| matches!(node.non_asap(), Some(NonASAPOp::SQLWindowFunc { .. })))); } #[tokio::test] diff --git a/crates/devtools/tests/stage_pipeline.rs b/crates/devtools/tests/stage_pipeline.rs new file mode 100644 index 000000000..b1ca75ee4 --- /dev/null +++ b/crates/devtools/tests/stage_pipeline.rs @@ -0,0 +1,108 @@ +//! `stage_pipeline` writes a valid four-stage viewer document for #509 Example 1. +use std::collections::HashSet; +use std::path::PathBuf; +use std::process::Command; + +use asap_types::ir::export::{ + LogicalASAPDAG, LogicalASAPDAGDocument, PhysicalASAPDAG, PhysicalASAPDAGDocument, +}; +use serde_json::Value; + +const COMMITTED: &str = "../../tools/dag-viewer/examples/planner-layering-example1.json"; + +fn generate(extra: &[&str]) -> Value { + let out = std::env::temp_dir().join(format!( + "stage_pipeline_{}_{}.json", + std::process::id(), + extra.len() + )); + let status = Command::new(env!("CARGO_BIN_EXE_stage_pipeline")) + .args(["--example", "planner-layering-1", "--out"]) + .arg(&out) + .args(extra) + .status() + .unwrap(); + assert!(status.success()); + let document = serde_json::from_str(&std::fs::read_to_string(&out).unwrap()).unwrap(); + std::fs::remove_file(out).unwrap(); + document +} + +fn validated_dag(value: &Value, queries: usize) { + let dag: LogicalASAPDAG = serde_json::from_value(value.clone()).unwrap(); + assert_eq!(dag.roots.len(), queries); + LogicalASAPDAGDocument::new(dag).validate().unwrap(); +} + +/// Every exported DAG validates and has one root per query; ids are unique; +/// Stage 2 maps Stage 1 one-to-one with no cost; Stage 3 accounts for every +/// Stage 2 candidate once and costs exactly the valid ones; the committed +/// fixture is current. +#[test] +fn example1_document_is_valid_and_committed_fixture_is_current() { + let document = generate(&[]); + let committed: Value = serde_json::from_str( + &std::fs::read_to_string(PathBuf::from(env!("CARGO_MANIFEST_DIR")).join(COMMITTED)) + .unwrap(), + ) + .unwrap(); + assert_eq!(document, committed, "regenerate the committed fixture"); + + assert_eq!(document["format"], "asap-stage-pipeline/v1"); + let queries = document["workload"]["queries"].as_array().unwrap().len(); + assert_eq!(queries, 2); + validated_dag(&document["stage0_logical"]["dag"], queries); + + let stage1 = &document["stage1_logical_asap"]; + let candidates = stage1["candidates"].as_array().unwrap(); + assert_eq!(stage1["capped"], false); + assert_eq!(stage1["combinations"], candidates.len()); + let mut ids = HashSet::new(); + for candidate in candidates { + assert!(ids.insert(candidate["id"].as_str().unwrap())); + validated_dag(&candidate["dag"], queries); + } + + let physical = document["stage2_physical_asap"]["candidates"] + .as_array() + .unwrap(); + let sources: HashSet<_> = physical + .iter() + .map(|p| p["from_logical"].as_str().unwrap()) + .collect(); + assert_eq!(physical.len(), candidates.len()); + assert_eq!(sources, ids); + for candidate in physical { + assert!(candidate.get("cost").is_none(), "Stage 2 has no cost"); + let dag: PhysicalASAPDAG = serde_json::from_value(candidate["dag"].clone()).unwrap(); + assert_eq!(dag.roots.len(), queries); + PhysicalASAPDAGDocument::new(dag).validate().unwrap(); + } + + let stage3 = &document["stage3_selection"]; + let costs = stage3["costs"].as_object().unwrap(); + let selected = stage3["selected"].as_str().unwrap(); + let mut accounted = HashSet::from([selected]); + for rejection in stage3["rejected"].as_array().unwrap() { + let id = rejection["id"].as_str().unwrap(); + assert!(accounted.insert(id), "{id} accounted for once"); + assert!(!rejection["reason"].as_str().unwrap().is_empty()); + assert_eq!( + rejection["valid"].as_bool().unwrap(), + costs.contains_key(id) + ); + } + assert!(costs.contains_key(selected)); + let all: HashSet<_> = physical.iter().map(|p| p["id"].as_str().unwrap()).collect(); + assert_eq!(accounted, all); +} + +/// The Cartesian product is cut at `--max-candidates` and says so. +#[test] +fn candidate_cap_is_recorded() { + let document = generate(&["--max-candidates", "5"]); + let stage1 = &document["stage1_logical_asap"]; + assert_eq!(stage1["capped"], true); + assert_eq!(stage1["candidates"].as_array().unwrap().len(), 5); + assert!(stage1["combinations"].as_u64().unwrap() > 5); +} diff --git a/crates/devtools/tests/viewer_contract.rs b/crates/devtools/tests/viewer_contract.rs new file mode 100644 index 000000000..b34b472fe --- /dev/null +++ b/crates/devtools/tests/viewer_contract.rs @@ -0,0 +1,128 @@ +//! `tools/dag-viewer` ↔ `asap_types::dag_export` contract: the viewer's +//! `KIND_CATEGORY_JSON` must categorize exactly the `kind` strings +//! [`asap_types::dag_export::export`] can emit — `Operator::kind_name()` of +//! every `NonASAPOp` and `ASAPOp` variant — no more (a stale kind the IR no +//! longer has) and no less (an exported kind the viewer would render +//! uncategorized). + +use std::collections::{BTreeMap, BTreeSet}; + +use asap_types::ir::{ASAPOp, NonASAPOp}; + +/// Every `NonASAPOp::kind_name()`. +const NON_ASAP_KINDS: &[&str] = &[ + "Scan", + "Values", + "Filter", + "Project", + "Aggregate", + "Join", + "SetOp", + "Concat", + "Dedup", + "Sort", + "Limit", + "BinaryOp", + "SQLWindowFunc", + "TimeRange", + "TimeShift", + "PromqlVectorFromScalar", + "PromqlRelabel", + "PromqlInfoEnrich", + "PromqlSeriesSample", + "PromqlSubquery", +]; + +/// Every `ASAPOp::kind_name()`. +const ASAP_KINDS: &[&str] = &[ + "SummaryAgg", + "SummaryEstimate", + "FinalizeExactAccumulator", + "MaintainPopulation", + "EvaluatePopulation", + "SummaryMerge", + "SummarySubtract", + "SummaryDelete", + "SummaryJoin", + "Extension", +]; + +/// Compile-time tripwire: adding an operator variant fails these exhaustive +/// matches until the matching `*_KINDS` list above is extended too. Never +/// called; the match arms are the point. +#[allow(dead_code)] +fn kind_lists_track_every_variant(non_asap: &NonASAPOp, asap: &ASAPOp) { + let listed = |name: &str, list: &[&str]| assert!(list.contains(&name)); + listed( + match non_asap { + NonASAPOp::Scan { .. } => "Scan", + NonASAPOp::Values { .. } => "Values", + NonASAPOp::Filter { .. } => "Filter", + NonASAPOp::Project { .. } => "Project", + NonASAPOp::Aggregate { .. } => "Aggregate", + NonASAPOp::Join { .. } => "Join", + NonASAPOp::SetOp { .. } => "SetOp", + NonASAPOp::Concat { .. } => "Concat", + NonASAPOp::Dedup { .. } => "Dedup", + NonASAPOp::Sort { .. } => "Sort", + NonASAPOp::Limit { .. } => "Limit", + NonASAPOp::BinaryOp { .. } => "BinaryOp", + NonASAPOp::SQLWindowFunc { .. } => "SQLWindowFunc", + NonASAPOp::TimeRange { .. } => "TimeRange", + NonASAPOp::TimeShift { .. } => "TimeShift", + NonASAPOp::PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", + NonASAPOp::PromqlRelabel { .. } => "PromqlRelabel", + NonASAPOp::PromqlInfoEnrich { .. } => "PromqlInfoEnrich", + NonASAPOp::PromqlSeriesSample { .. } => "PromqlSeriesSample", + NonASAPOp::PromqlSubquery { .. } => "PromqlSubquery", + }, + NON_ASAP_KINDS, + ); + listed( + match asap { + ASAPOp::SummaryAgg { .. } => "SummaryAgg", + ASAPOp::SummaryEstimate { .. } => "SummaryEstimate", + ASAPOp::FinalizeExactAccumulator { .. } => "FinalizeExactAccumulator", + ASAPOp::MaintainPopulation { .. } => "MaintainPopulation", + ASAPOp::EvaluatePopulation { .. } => "EvaluatePopulation", + ASAPOp::SummaryMerge { .. } => "SummaryMerge", + ASAPOp::SummarySubtract { .. } => "SummarySubtract", + ASAPOp::SummaryDelete { .. } => "SummaryDelete", + ASAPOp::SummaryJoin { .. } => "SummaryJoin", + ASAPOp::Extension { .. } => "Extension", + }, + ASAP_KINDS, + ); +} + +/// The viewer's `kind -> category` table, parsed out of the JS source the +/// same way the viewer itself does (`JSON.parse(KIND_CATEGORY_JSON)`). +fn viewer_kind_categories() -> BTreeMap { + const START: &str = "const KIND_CATEGORY_JSON = `"; + let source = include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../tools/dag-viewer/node-style.js" + )); + let json = source + .split_once(START) + .expect("node-style.js must declare KIND_CATEGORY_JSON") + .1 + .split_once("`;") + .expect("KIND_CATEGORY_JSON must be a template literal") + .0; + serde_json::from_str(json).expect("KIND_CATEGORY_JSON must be valid JSON") +} + +#[test] +fn viewer_categorizes_exactly_the_exported_node_kinds() { + let expected: BTreeSet<&str> = NON_ASAP_KINDS.iter().chain(ASAP_KINDS).copied().collect(); + assert_eq!( + expected.len(), + NON_ASAP_KINDS.len() + ASAP_KINDS.len(), + "exported kind names must be unique" + ); + let categories = viewer_kind_categories(); + let actual: BTreeSet<&str> = categories.keys().map(String::as_str).collect(); + + assert_eq!(actual, expected); +} diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/executor/Cargo.toml similarity index 51% rename from crates/asap-physical-operators/Cargo.toml rename to crates/executor/Cargo.toml index 193707c48..f8aaffb3d 100644 --- a/crates/asap-physical-operators/Cargo.toml +++ b/crates/executor/Cargo.toml @@ -1,9 +1,14 @@ [package] -name = "asap-physical-operators" +name = "asap-executor" version = "0.1.0" edition = "2021" +# #509 Stage 4 reference executor. No stage crate (asap-logical-optimizer, +# asap-physical-optimizer, asap-plan-selection) depends on it; their manifest +# guards reject it. Its tests use the stage crates as dev-dependencies. + [dependencies] +chrono = { version = "=0.4.39", default-features = false, features = ["std"] } futures = "0.3" planner-types = { package = "asap-types", path = "../types" } asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "5f03ccbd798ed5fec62bdd839bcb331123cab369" } @@ -16,5 +21,6 @@ regex = "1" [dev-dependencies] rmp-serde = "1" -asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-plan-selection = { path = "../plan-selection" } +asap-logical-optimizer = { path = "../logical-optimizer" } asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-physical-operators/README.md b/crates/executor/README.md similarity index 89% rename from crates/asap-physical-operators/README.md rename to crates/executor/README.md index 2d3e7b569..3189dfefb 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/executor/README.md @@ -1,6 +1,6 @@ -# ASAP physical operators +# ASAP executor -An independent Rust physical operator DAG runtime shared by ingestion time and +The #509 Stage 4 reference executor: an independent Rust physical operator DAG runtime shared by ingestion time and query time execution. The library requires neither backend engine, a server, a storage implementation, Arrow nor DataFusion. DataFusion informed the design; it is not the execution framework. @@ -14,20 +14,20 @@ thread pool. Poll multiple root streams concurrently when they share inputs. `operators::Operator` implements native batch sources, scalar values, projection, filtering, grouped exact aggregation, semi-join, grouped Sort and -Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. +Limit, vector-to-scalar conversion, Union, and summary construction/merge/evaluation. Sort followed by Limit implements grouped ranking; no dedicated TopK physical operator is needed. Summary construction updates state batch by batch. End of input means the supplied query range or ingestion window is complete. ```rust -use asap_physical_operators::{ +use asap_executor::{ expressions::Expression, operators::Operator, values::Value, plan::PhysicalDAG, runtime::{Limits, RunContext, Scope}, }; -use asap_physical_operators::planner::pre_asap::DataType; +use asap_executor::planner::ir::schema::DataType; use futures::{executor::block_on, StreamExt}; let source = Operator::scalar(Value::Int64(7), DataType::Int64)?; @@ -44,10 +44,10 @@ let run = RunContext::new( let mut output = plan.execute(&[1], run)?.remove(0); let batch = block_on(output.next()).unwrap()?; assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); -# Ok::<(), asap_physical_operators::dag::Error>(()) +# Ok::<(), asap_executor::dag::Error>(()) ``` -`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDAG`) and typed input contracts. +`physical_planner::compile` accepts a logical Post-ASAP DAG (`PhysicalASAPDAG`) and typed input contracts. The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and schema mismatches before starting a source. Implement `PhysicalOperator` for a deployment source, including asynchronous I/O; computation operators remain in @@ -59,7 +59,7 @@ Plain values preserve Planner scalar/collection types and nullability. Numeric arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. Boolean predicates use three-valued logic. Native summary states currently cover exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Binding checks family, -parameters and readout compatibility; source batches also validate state payloads. +parameters and evaluation compatibility; source batches also validate state payloads. Existing accumulator algorithms are reused as kernels behind these operators. This crate is owned by ASAPPlanner. Its `planner-types` dependency is the local @@ -78,9 +78,9 @@ See [the design](../../docs/design_docs/physical-planning-and-deployment.md). - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. - `sources`: raw-source interface, Scan and the memory connector. - `physical_planner`: native operator lowering, typed input contracts and checked instantiation. -- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed readout and update adapters. -- `readout`: readouts over merged exact summary states. -- `capability`: explicit kernel, native-batch and typed readout validation. +- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed evaluation and update adapters. +- `evaluation`: evaluations over merged exact summary states. +- `capability`: explicit kernel, native-batch and typed evaluation validation. The `dag`, `factory`, `traits` and `arithmetic` paths are re-exports. They contain no alternative execution implementations. @@ -104,7 +104,7 @@ There is no spill or partitioned parallel execution in this implementation. ## Physical compilation and deployment inputs -`physical_planner::compile` accepts a Planner `PostAsapDAG`, typed +`physical_planner::compile` accepts a Planner `PhysicalASAPDAG`, typed `InputContract`s and output roots. It returns a reusable `CompiledPhysicalDAG` containing selected native operators and no live readers. Compilation validates schemas, input ordering, sharing and boundedness before deployment source access. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/executor/src/capability.rs similarity index 73% rename from crates/asap-physical-operators/src/capability.rs rename to crates/executor/src/capability.rs index 5ed1c5501..48c2395be 100644 --- a/crates/asap-physical-operators/src/capability.rs +++ b/crates/executor/src/capability.rs @@ -2,22 +2,22 @@ //! //! `validate_summary_kernel` checks update kernels, including families without a //! native batch representation. `validate_native_family` and -//! `validate_sketch_readout` / `validate_exact_readout` check native state and readout support. -//! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. +//! `validate_sketch_evaluation` / `validate_exact_evaluation` check native state and evaluation support. +//! Keyed weighted-frequency evaluations are checked by `Operator::keyed_evaluation`. //! A successful kernel check alone does not mean a physical DAG will bind. //! //! Stored-state encodings belong to deployments. Full plan acceptance is //! owned by `binding`, which also validates schemas, expressions and inputs. use crate::Error; -use planner_types::post_asap::{ - ExactKind, ExactParams, FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams, - SketchStatistic, SummaryUpdate, +use planner_types::ir::schema::{ + ExactKind, ExactParams, FieldDataType as SummaryFamilyType, GroupingStrategy, SketchAlgorithm, + SketchParams, SketchStatistic, SummaryUpdate, }; /// Check the same contract used by `create_planner_accumulator` before a plan /// is accepted. Execution timing is deliberately not a kernel property. pub fn validate_summary_kernel( - family: &FieldDataType, + family: &SummaryFamilyType, input: &SummaryUpdate, grouping: &GroupingStrategy, ) -> Result<(), String> { @@ -25,7 +25,7 @@ pub fn validate_summary_kernel( return Err("shared summary grouping has no registered kernel".into()); } let keyed = match family { - FieldDataType::ExactAggregate(kind, params) => { + SummaryFamilyType::ExactAggregate(kind, params) => { use ExactKind as K; use ExactParams as P; if !matches!( @@ -41,7 +41,7 @@ pub fn validate_summary_kernel( } input.item.is_some() } - FieldDataType::Sketch(kind, layout) => { + SummaryFamilyType::Sketch(kind, layout) => { if layout != grouping { return Err("Planner family and operator grouping disagree".into()); } @@ -122,12 +122,12 @@ fn valid_matrix(width: u32, depth: u32) -> bool { .is_some() } -pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::SummaryUpdate) -> bool { - use planner_types::post_asap::{NonNegativeWeightProof, SummaryInputExpr, WeightDomain}; +pub(crate) fn is_unit_sample_frequency(update: &planner_types::ir::schema::SummaryUpdate) -> bool { + use planner_types::ir::schema::{NonNegativeWeightProof, SummaryInputExpr, WeightDomain}; matches!( update.item, Some(SummaryInputExpr::Column( - planner_types::pre_asap::ColumnRef::SampleValue + planner_types::ir::scalar::ColumnRef::SampleValue )) ) && matches!(update.weight, SummaryInputExpr::Constant(1.0)) && matches!( @@ -138,9 +138,9 @@ pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::Summar ) } -pub fn validate_native_family(family: &FieldDataType) -> Result<(), Error> { - use planner_types::post_asap::SketchAlgorithm as A; - if let FieldDataType::Sketch(kind, grouping) = family { +pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { + use planner_types::ir::schema::SketchAlgorithm as A; + if let SummaryFamilyType::Sketch(kind, grouping) = family { // Plain Count-Min is native as stored state only: it merges and reads // its bare count, but the DAG does not build it from rows. if let (A::Cms, SketchParams::Cms { width, depth }) = (kind.algorithm(), kind.params()) { @@ -165,8 +165,8 @@ pub fn validate_native_family(family: &FieldDataType) -> Result<(), Error> { } } match family { - FieldDataType::ExactAggregate(..) => {} - FieldDataType::Sketch(kind, _) + SummaryFamilyType::ExactAggregate(..) => {} + SummaryFamilyType::Sketch(kind, _) if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} _ => { return Err(Error::Invalid( @@ -176,30 +176,30 @@ pub fn validate_native_family(family: &FieldDataType) -> Result<(), Error> { } crate::capability::validate_summary_kernel( family, - &planner_types::post_asap::SummaryUpdate::column( - planner_types::pre_asap::ColumnRef::SampleValue, + &planner_types::ir::schema::SummaryUpdate::column( + planner_types::ir::scalar::ColumnRef::SampleValue, ), &Default::default(), ) .map_err(Error::Invalid) } -/// A sketch readout is native only for the families Planner can read directly. -pub fn validate_sketch_readout( - family: &FieldDataType, +/// A sketch evaluation is native only for the families Planner can read directly. +pub fn validate_sketch_evaluation( + family: &SummaryFamilyType, query: &SketchStatistic, ) -> Result<(), Error> { validate_native_family(family)?; - use planner_types::post_asap::SketchAlgorithm as A; + use planner_types::ir::schema::SketchAlgorithm as A; // A point count without an item value reads the total count. let bare_count = matches!(query, SketchStatistic::PointCount { value: None, .. }); let supported = match family { - FieldDataType::Sketch(kind, _) => match (kind.algorithm(), query) { + SummaryFamilyType::Sketch(kind, _) => match (kind.algorithm(), query) { (A::Kll, SketchStatistic::Quantile { q }) | (A::DDSketch, SketchStatistic::Quantile { q }) => { if !(0.0..=1.0).contains(q) { return Err(Error::Invalid( - "quantile readout requires quantile in [0,1]".into(), + "quantile evaluation requires quantile in [0,1]".into(), )); } true @@ -208,7 +208,7 @@ pub fn validate_sketch_readout( (A::Hll, SketchStatistic::Cardinality) => true, (A::Hll, _) => bare_count, // Only count intents read a Count-Min bare count, and their - // updates have unit weight; the readout is typed Int64 on that basis. + // updates have unit weight; the evaluation is typed Int64 on that basis. (A::Cms, _) => bare_count, _ => false, }, @@ -216,36 +216,39 @@ pub fn validate_sketch_readout( }; if !supported { return Err(Error::Invalid( - "readout is not implemented for this summary family".into(), + "evaluation is not implemented for this summary family".into(), )); } Ok(()) } -/// An exact readout must match the exact family it reads. -pub fn validate_exact_readout( - family: &FieldDataType, - readout: &crate::summary_kernels::exact::ExactReadout, +/// An exact evaluation must match the exact family it reads. +pub fn validate_exact_evaluation( + family: &SummaryFamilyType, + evaluation: &crate::summary_kernels::exact::ExactEvaluation, ) -> Result<(), Error> { validate_native_family(family)?; use crate::Statistic as S; - use planner_types::post_asap::ExactKind as E; + use planner_types::ir::schema::ExactKind as E; let supported = matches!( - (family, readout.statistic), - (FieldDataType::ExactAggregate(E::Sum, _), S::Sum) - | (FieldDataType::ExactAggregate(E::Count, _), S::Count) - | (FieldDataType::ExactAggregate(E::Min, _), S::Min) - | (FieldDataType::ExactAggregate(E::Max, _), S::Max) - | (FieldDataType::ExactAggregate(E::Rate, _), S::Rate) - | (FieldDataType::ExactAggregate(E::Increase, _), S::Increase) + (family, evaluation.statistic), + (SummaryFamilyType::ExactAggregate(E::Sum, _), S::Sum) + | (SummaryFamilyType::ExactAggregate(E::Count, _), S::Count) + | (SummaryFamilyType::ExactAggregate(E::Min, _), S::Min) + | (SummaryFamilyType::ExactAggregate(E::Max, _), S::Max) + | (SummaryFamilyType::ExactAggregate(E::Rate, _), S::Rate) + | ( + SummaryFamilyType::ExactAggregate(E::Increase, _), + S::Increase + ) ); if !supported { return Err(Error::Invalid( - "readout is not implemented for this summary family".into(), + "evaluation is not implemented for this summary family".into(), )); } - if readout.lookback_ms.is_some_and(|lookback| { - lookback <= 0 || !matches!(readout.statistic, S::Rate | S::Increase) + if evaluation.lookback_ms.is_some_and(|lookback| { + lookback <= 0 || !matches!(evaluation.statistic, S::Rate | S::Increase) }) { return Err(Error::Invalid("invalid exact counter lookback".into())); } diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/executor/src/dag/mod.rs similarity index 100% rename from crates/asap-physical-operators/src/dag/mod.rs rename to crates/executor/src/dag/mod.rs diff --git a/crates/asap-physical-operators/src/error.rs b/crates/executor/src/error.rs similarity index 100% rename from crates/asap-physical-operators/src/error.rs rename to crates/executor/src/error.rs diff --git a/crates/asap-physical-operators/src/readout.rs b/crates/executor/src/evaluation.rs similarity index 81% rename from crates/asap-physical-operators/src/readout.rs rename to crates/executor/src/evaluation.rs index e84d8c828..c0898e4f1 100644 --- a/crates/asap-physical-operators/src/readout.rs +++ b/crates/executor/src/evaluation.rs @@ -1,4 +1,4 @@ -//! Readouts over merged exact summary states. +//! Evaluations over merged exact summary states. use crate::summary_kernels::exact::ExactAccumulator; use crate::{AggregateCore, KeyByLabelValues, Statistic}; use std::sync::Arc; @@ -12,7 +12,7 @@ fn merge_exact_states( .as_any() .downcast_ref::() .cloned() - .ok_or_else(|| "readout requires Planner exact state".to_string()) + .ok_or_else(|| "evaluation requires Planner exact state".to_string()) }; let mut merged = exact(&states.next().ok_or("empty exact state input")?)?; for state in states { @@ -23,7 +23,7 @@ fn merge_exact_states( Ok(merged) } -/// PromQL counter readouts omit a series with fewer than two samples. Other +/// PromQL counter evaluations omit a series with fewer than two samples. Other /// state/type/range failures remain errors rather than empty results. pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statistic) -> bool { matches!(statistic, Statistic::Rate | Statistic::Increase) @@ -36,7 +36,7 @@ pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statis /// Merge already selected exact panes and read one population. `None` means /// the population is absent from the result: a counter with too few samples, /// or an empty MIN/MAX. -pub fn exact_readout( +pub fn exact_evaluation( states: impl IntoIterator>, statistic: Statistic, range_ms: Option<(i64, i64)>, @@ -47,17 +47,17 @@ pub fn exact_readout( return Ok(None); } merged - .readout(statistic, range_ms, key) + .evaluation(statistic, range_ms, key) .map_err(|error| error.to_string()) } #[cfg(test)] mod counter_tests { use super::*; - use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType}; + use planner_types::ir::schema::{ExactKind, ExactParams, FieldDataType as SummaryFamilyType}; fn counter(kind: ExactKind, params: ExactParams, keyed: bool) -> ExactAccumulator { - ExactAccumulator::new(FieldDataType::ExactAggregate(kind, params), keyed).unwrap() + ExactAccumulator::new(SummaryFamilyType::ExactAggregate(kind, params), keyed).unwrap() } // A counter population with a single sample is absent, keyed or not. @@ -76,7 +76,7 @@ mod counter_tests { let key = keyed.then(|| KeyByLabelValues::new_with_labels(vec!["checkout".into()])); state.update(key.as_ref(), 10., 10_000); assert_eq!( - exact_readout( + exact_evaluation( [Arc::new(state) as Arc], statistic, None, @@ -97,15 +97,15 @@ mod counter_tests { let rate = Statistic::Rate; let one = [Arc::new(state.clone()) as Arc]; assert_eq!( - exact_readout(one, rate, Some((0, 60_000)), None).unwrap(), + exact_evaluation(one, rate, Some((0, 60_000)), None).unwrap(), None ); state.update(None, 20., 20_000); let two = || [Arc::new(state.clone()) as Arc]; - assert!(exact_readout(two(), rate, Some((0, 60_000)), None) + assert!(exact_evaluation(two(), rate, Some((0, 60_000)), None) .unwrap() .is_some()); - assert!(exact_readout(two(), rate, Some((60_000, 0)), None).is_err()); - assert!(exact_readout([], rate, Some((0, 60_000)), None).is_err()); + assert!(exact_evaluation(two(), rate, Some((60_000, 0)), None).is_err()); + assert!(exact_evaluation([], rate, Some((0, 60_000)), None).is_err()); } } diff --git a/crates/asap-physical-operators/src/expressions/arithmetic.rs b/crates/executor/src/expressions/arithmetic.rs similarity index 84% rename from crates/asap-physical-operators/src/expressions/arithmetic.rs rename to crates/executor/src/expressions/arithmetic.rs index e0766763d..99a912012 100644 --- a/crates/asap-physical-operators/src/expressions/arithmetic.rs +++ b/crates/executor/src/expressions/arithmetic.rs @@ -2,11 +2,11 @@ //! Preserve IEEE non-finite results; callers own their output policies. pub fn evaluate_float64_arithmetic( - operator: &planner_types::pre_asap::ArithmeticOpKind, + operator: &planner_types::ir::scalar::ArithmeticOpKind, left: f64, right: f64, ) -> f64 { - use planner_types::pre_asap::ArithmeticOpKind::*; + use planner_types::ir::scalar::ArithmeticOpKind::*; match operator { Add => left + right, Sub => left - right, @@ -20,12 +20,13 @@ pub fn evaluate_float64_arithmetic( /// Execute the Planner binary contract after a deployment has resolved matching rows. pub fn evaluate_binary( - operator: &planner_types::post_asap::BinaryOperator, + operator: &crate::expressions::binary::BinaryOperator, left: f64, right: f64, ) -> Result { + use crate::expressions::binary::BinaryOpKind; use crate::{values::Value, Error}; - use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind}; + use planner_types::ir::scalar::ArithmeticOpKind; let invalid = || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); if operator.vector_match.is_some() { @@ -62,8 +63,8 @@ pub fn evaluate_binary( } /// IEEE comparison, as Go's: NaN is unequal to everything, itself included. -fn compare(op: &planner_types::pre_asap::CompareOpKind, left: f64, right: f64) -> Option { - use planner_types::pre_asap::CompareOpKind; +fn compare(op: &planner_types::ir::scalar::CompareOpKind, left: f64, right: f64) -> Option { + use planner_types::ir::scalar::CompareOpKind; Some(match op { CompareOpKind::Eq => left == right, CompareOpKind::Ne => left != right, diff --git a/crates/executor/src/expressions/binary.rs b/crates/executor/src/expressions/binary.rs new file mode 100644 index 000000000..5b027505e --- /dev/null +++ b/crates/executor/src/expressions/binary.rs @@ -0,0 +1,34 @@ +//! Execution configuration for a binary kernel, including comparison evaluation mode. +use planner_types::ir::operator::{PromQLVectorSetOpKind, VectorMatch}; +use planner_types::ir::scalar::{ArithmeticOpKind, CompareOpKind}; +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum BinaryOpKind { + Arithmetic(ArithmeticOpKind), + Compare(CompareOpKind), + CompareBool(CompareOpKind), + Set(PromQLVectorSetOpKind), +} +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub struct BinaryOperator { + pub kind: BinaryOpKind, + pub vector_match: Option, + pub checked_relative_division: bool, + pub checked_finite_division: bool, +} +impl BinaryOperator { + pub fn from_logical(operator: &planner_types::ir::BinaryOperator, return_bool: bool) -> Self { + use planner_types::ir::operator::BinaryOpKind as L; + Self { + kind: match &operator.kind { + L::Arithmetic(op) => BinaryOpKind::Arithmetic(op.clone()), + L::Compare(op) if return_bool => BinaryOpKind::CompareBool(op.clone()), + L::Compare(op) => BinaryOpKind::Compare(op.clone()), + L::CompareBool(op) => BinaryOpKind::CompareBool(op.clone()), + L::Set(op) => BinaryOpKind::Set(op.clone()), + }, + vector_match: operator.vector_match.clone(), + checked_relative_division: operator.checked_relative_division, + checked_finite_division: operator.checked_finite_division, + } + } +} diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/executor/src/expressions/mod.rs similarity index 98% rename from crates/asap-physical-operators/src/expressions/mod.rs rename to crates/executor/src/expressions/mod.rs index 8947743dc..b6e09c9cd 100644 --- a/crates/asap-physical-operators/src/expressions/mod.rs +++ b/crates/executor/src/expressions/mod.rs @@ -3,14 +3,16 @@ use crate::{ values::{plain, SchemaRef, Value}, Error, }; -use planner_types::pre_asap::{ArithmeticOpKind, DataType}; +use planner_types::ir::scalar::ArithmeticOpKind; +use planner_types::ir::schema::DataType; pub mod arithmetic; +pub mod binary; mod planner; pub use planner::CompiledExpression; #[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub enum Expression { Binary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, left: Box, right: Box, }, @@ -64,7 +66,8 @@ impl Expression { left, right, } => { - use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + use crate::expressions::binary::BinaryOpKind; + use planner_types::ir::scalar::CompareOpKind; let (a, n) = left.dtype(input)?; let (b, m) = right.dtype(input)?; if a != DataType::Float64 || b != a || operator.vector_match.is_some() { diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/executor/src/expressions/planner.rs similarity index 67% rename from crates/asap-physical-operators/src/expressions/planner.rs rename to crates/executor/src/expressions/planner.rs index ae04b3f7c..f877ff890 100644 --- a/crates/asap-physical-operators/src/expressions/planner.rs +++ b/crates/executor/src/expressions/planner.rs @@ -3,20 +3,23 @@ use crate::{ values::{SchemaRef, Value}, Error, }; -use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use planner_types::ir::scalar::{ArithmeticOpKind, CompareOpKind, ScalarValue}; +use planner_types::ir::schema::DataType; + +use planner_types::ir::ScalarExpr; use std::{cmp::Ordering, sync::Arc}; pub(super) fn evaluate( - expr: &QueryExpr, + expr: &ScalarExpr, row: &[Value], - schema: &planner_types::pre_asap::Schema, + schema: &planner_types::ir::schema::Schema, ) -> Result { match expr { - QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + ScalarExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( "column {index} outside row width {}", row.len() ))), - QueryExpr::Literal(value) => Ok(match value { + ScalarExpr::Literal(value) => Ok(match value { ScalarValue::Interval { months, days, @@ -32,18 +35,62 @@ pub(super) fn evaluate( ScalarValue::Boolean(value) => Value::Bool(*value), ScalarValue::Null => Value::Null, }), - QueryExpr::Compare { left, op, right } => { + ScalarExpr::Cast { expr, to, .. } => { + let value = evaluate(expr, row, schema)?; + match (value, to) { + (Value::Null, _) => Ok(Value::Null), + (Value::Int64(value), DataType::Float64) => Ok(Value::Float64(value as f64)), + (value, _) + if expr + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0 + == *to => + { + Ok(value) + } + _ => Err(Error::Invalid("unsupported cast".into())), + } + } + ScalarExpr::Negative { expr, .. } => match evaluate(expr, row, schema)? { + Value::Float64(v) => Ok(Value::Float64(-v)), + Value::Int64(v) => v + .checked_neg() + .map(Value::Int64) + .ok_or_else(|| Error::Invalid("integer negation overflow".into())), + Value::Null => Ok(Value::Null), + _ => Err(Error::Invalid("invalid negation input".into())), + }, + ScalarExpr::Compare { + left, op, right, .. + } => { let left = evaluate(left, row, schema)?; let right = evaluate(right, row, schema)?; compare(op, left, right) } - QueryExpr::Arithmetic { op, left, right } => arithmetic( + ScalarExpr::Arithmetic { + op, left, right, .. + } => arithmetic( op, evaluate(left, row, schema)?, evaluate(right, row, schema)?, ), - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { - let and = matches!(expr, QueryExpr::BoolAnd(_)); + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } => { + for (condition, value) in branches { + if matches!(evaluate(condition, row, schema)?, Value::Bool(true)) { + return evaluate(value, row, schema); + } + } + else_expr + .as_ref() + .map_or(Ok(Value::Null), |e| evaluate(e, row, schema)) + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { + let and = matches!(expr, ScalarExpr::BoolAnd(_)); let mut null = false; for part in parts { match evaluate(part, row, schema)? { @@ -55,21 +102,44 @@ pub(super) fn evaluate( } Ok(if null { Value::Null } else { Value::Bool(and) }) } - QueryExpr::Not(value) => match evaluate(value, row, schema)? { + ScalarExpr::Not(value) => match evaluate(value, row, schema)? { Value::Bool(value) => Ok(Value::Bool(!value)), Value::Null => Ok(Value::Null), _ => Err(Error::Invalid("boolean predicate required".into())), }, - QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + ScalarExpr::IsNull(value) => Ok(Value::Bool(matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + ScalarExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( evaluate(value, row, schema)?, Value::Null ))), - QueryExpr::FunctionCall { name, args } => { - use planner_types::pre_asap::scalar_type_rules::MapScalarFunction; + ScalarExpr::FunctionCall { name, args } => { + use planner_types::ir::scalar::scalar_type_rules::MapScalarFunction; + if planner_types::ir::scalar::scalar_type_rules::promql_function_arity(name).is_some() { + let values = args + .iter() + .map(|arg| match evaluate(arg, row, schema)? { + Value::Float64(v) => Ok(v), + _ => Err(Error::Invalid("PromQL function requires floats".into())), + }) + .collect::, _>>()?; + return Ok(Value::Float64(promql_function(name, &values)?)); + } + if name == "promql_drop_metric_name" { + let Value::Utf8(encoded) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("series identity must be Utf8".into())); + }; + let mut labels: std::collections::BTreeMap = + serde_json::from_str(&encoded).map_err(|e| Error::Invalid(e.to_string()))?; + labels.remove("__name__"); + return Ok(Value::Utf8( + serde_json::to_string(&labels) + .map_err(|e| Error::Invalid(e.to_string()))? + .into(), + )); + } if name.eq_ignore_ascii_case("asap_struct_field") { expr.scalar_type(schema) .map_err(|error| Error::Invalid(error.to_string()))?; @@ -81,10 +151,10 @@ pub(super) fn evaluate( unreachable!() }; let offset = match &args[1] { - QueryExpr::Literal(ScalarValue::Int64(index)) => { + ScalarExpr::Literal(ScalarValue::Int64(index)) => { usize::try_from(index - 1).ok() } - QueryExpr::Literal(ScalarValue::Utf8(name)) => { + ScalarExpr::Literal(ScalarValue::Utf8(name)) => { fields.iter().position(|field| &field.name == name) } _ => None, @@ -329,35 +399,120 @@ fn cell_cmp(left: &Value, right: &Value) -> Option { } } +fn promql_function(name: &str, args: &[f64]) -> Result { + let x = args[0]; + Ok(match &name[7..] { + "abs" => x.abs(), + "ceil" => x.ceil(), + "floor" => x.floor(), + "exp" => x.exp(), + "ln" => x.ln(), + "log2" => x.log2(), + "log10" => x.log10(), + "sqrt" => x.sqrt(), + "sgn" => { + if x.is_nan() { + f64::NAN + } else if x == 0.0 { + 0.0 + } else { + x.signum() + } + } + "sin" => x.sin(), + "cos" => x.cos(), + "tan" => x.tan(), + "asin" => x.asin(), + "acos" => x.acos(), + "atan" => x.atan(), + "sinh" => x.sinh(), + "cosh" => x.cosh(), + "tanh" => x.tanh(), + "asinh" => x.asinh(), + "acosh" => x.acosh(), + "atanh" => x.atanh(), + "deg" => x.to_degrees(), + "rad" => x.to_radians(), + "round" => { + let inverse = 1.0 / args[1]; + (x * inverse + 0.5).floor() / inverse + } + "clamp_min" => { + if x.is_nan() || args[1].is_nan() { + f64::NAN + } else { + x.max(args[1]) + } + } + "clamp_max" => { + if x.is_nan() || args[1].is_nan() { + f64::NAN + } else { + x.min(args[1]) + } + } + "clamp" => { + if args.iter().any(|x| x.is_nan()) { + f64::NAN + } else { + x.max(args[1]).min(args[2]) + } + } + part => { + use chrono::{Datelike, Timelike}; + if !x.is_finite() || x < i64::MIN as f64 || x >= i64::MAX as f64 { + return Ok(f64::NAN); + } + let Some(date) = chrono::DateTime::from_timestamp(x as i64, 0) else { + return Ok(f64::NAN); + }; + match part { + "minute" => date.minute() as f64, + "hour" => date.hour() as f64, + "day_of_week" => date.weekday().num_days_from_sunday() as f64, + "day_of_month" => date.day() as f64, + "day_of_year" => date.ordinal() as f64, + "month" => date.month() as f64, + "year" => date.year() as f64, + "days_in_month" => { + let year = date.year(); + let leap = year % 4 == 0 && (year % 100 != 0 || year % 400 == 0); + match date.month() { + 2 => { + if leap { + 29.0 + } else { + 28.0 + } + } + 4 | 6 | 9 | 11 => 30.0, + _ => 31.0, + } + } + _ => return Err(Error::Invalid("unregistered PromQL function".into())), + } + } + }) +} + #[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] pub struct CompiledExpression { - expression: QueryExpr, - schema: planner_types::pre_asap::Schema, + expression: ScalarExpr, + schema: planner_types::ir::schema::Schema, output: (DataType, bool), } impl CompiledExpression { - pub(crate) fn expression(&self) -> &QueryExpr { + pub(crate) fn expression(&self) -> &ScalarExpr { &self.expression } - pub fn compile(expression: &QueryExpr, input: &SchemaRef) -> Result { - let schema = input - .fields - .iter() - .map(|field| { - let planner_types::post_asap::FieldDataType::Plain(dtype) = &field.dtype else { - return Err(Error::Invalid( - "scalar expression cannot consume opaque summary state".into(), - )); - }; - Ok(planner_types::pre_asap::Field::plain( - field.name.clone(), - dtype.clone(), - field.nullable, - )) - }) - .collect::, Error>>()?; - let schema = planner_types::pre_asap::Schema::new(schema); + pub fn compile(expression: &ScalarExpr, input: &SchemaRef) -> Result { + if !input.is_all_plain() { + return Err(Error::Invalid( + "scalar expression cannot consume summary state".into(), + )); + } + let schema = input.as_ref().clone(); validate(expression, &schema)?; let output = expression .scalar_type(&schema) @@ -398,8 +553,7 @@ impl CompiledExpression { if row.len() != self.schema.fields.len() || row.iter().zip(&self.schema.fields).any(|(value, column)| { !column - .dtype - .plain() + .plain_dtype() .is_some_and(|dtype| value.matches(dtype, column.nullable)) }) { @@ -410,13 +564,27 @@ impl CompiledExpression { evaluate(&self.expression, row, &self.schema) } } -fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { +fn validate(expr: &ScalarExpr, schema: &planner_types::ir::schema::Schema) -> Result<(), Error> { let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); expr.scalar_type(schema) .map_err(|e| Error::Invalid(e.to_string()))?; match expr { - QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), - QueryExpr::Arithmetic { left, right, .. } => { + ScalarExpr::Column(_) | ScalarExpr::Literal(_) => Ok(()), + ScalarExpr::Cast { expr, to, .. } => { + let source = expr + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0; + if source != *to + && source != DataType::Null + && !(source == DataType::Int64 && *to == DataType::Float64) + { + return Err(invalid()); + } + validate(expr, schema) + } + ScalarExpr::Negative { expr, .. } => validate(expr, schema), + ScalarExpr::Arithmetic { left, right, .. } => { for value in [left, right] { validate(value, schema)?; if !matches!( @@ -431,7 +599,9 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Compare { left, right, op } => { + ScalarExpr::Compare { + left, right, op, .. + } => { if !matches!( op, CompareOpKind::Eq @@ -475,10 +645,13 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::FunctionCall { name, args } => { - if name != "asap_struct_field" + ScalarExpr::FunctionCall { name, args } => { + if name != "promql_drop_metric_name" + && planner_types::ir::scalar::scalar_type_rules::promql_function_arity(name) + .is_none() + && name != "asap_struct_field" && name != "asap_element_access" - && planner_types::pre_asap::scalar_type_rules::MapScalarFunction::from_name(name) + && planner_types::ir::scalar::scalar_type_rules::MapScalarFunction::from_name(name) .is_none() { return Err(invalid()); @@ -488,7 +661,29 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } => { + for (condition, value) in branches { + validate(condition, schema)?; + if condition + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0 + != DataType::Bool + { + return Err(invalid()); + } + validate(value, schema)?; + } + if let Some(value) = else_expr { + validate(value, schema)?; + } + Ok(()) + } + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { for part in parts { validate(part, schema)?; if !matches!( @@ -502,7 +697,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::Not(value) => { + ScalarExpr::Not(value) => { validate(value, schema)?; if !matches!( value @@ -515,7 +710,7 @@ fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Resul } Ok(()) } - QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), + ScalarExpr::IsNull(value) | ScalarExpr::IsNotNull(value) => validate(value, schema), _ => Err(invalid()), } } diff --git a/crates/asap-physical-operators/src/key_by_label_values.rs b/crates/executor/src/key_by_label_values.rs similarity index 100% rename from crates/asap-physical-operators/src/key_by_label_values.rs rename to crates/executor/src/key_by_label_values.rs diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/executor/src/lib.rs similarity index 97% rename from crates/asap-physical-operators/src/lib.rs rename to crates/executor/src/lib.rs index 345ee762d..ea6b73eaf 100644 --- a/crates/asap-physical-operators/src/lib.rs +++ b/crates/executor/src/lib.rs @@ -20,7 +20,7 @@ pub use planner_types as planner; pub mod dag; -pub mod readout; +pub mod evaluation; mod error; pub use error::Error; diff --git a/crates/asap-physical-operators/src/measurement.rs b/crates/executor/src/measurement.rs similarity index 100% rename from crates/asap-physical-operators/src/measurement.rs rename to crates/executor/src/measurement.rs diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/executor/src/operators/aggregate/mod.rs similarity index 95% rename from crates/asap-physical-operators/src/operators/aggregate/mod.rs rename to crates/executor/src/operators/aggregate/mod.rs index 76eccd5ee..94c2bb861 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/mod.rs +++ b/crates/executor/src/operators/aggregate/mod.rs @@ -24,7 +24,8 @@ impl Operator { } else { t.clone() }, - false, + !input.has_promql_series_identity() + && (groups.is_empty() || plain(&input, *i)?.1), ) } Reduction::Quantile { column, q } => { @@ -54,13 +55,13 @@ impl Operator { } pub fn window( input: SchemaRef, - intent: planner_types::pre_asap::AggIntent, + intent: planner_types::ir::operator::AggIntent, coordinate: usize, value: usize, groups: Vec, window: Option<(i64, i64)>, ) -> Result { - use planner_types::pre_asap::AggIntent; + use planner_types::ir::operator::AggIntent; validate_groups(&input, &groups)?; let histogram = matches!(intent, AggIntent::HistogramQuantile { .. }); if !matches!( @@ -183,7 +184,8 @@ async fn reduce( let mut work = Cooperative::new(context); let mut workspace = Workspace::new(context)?; let mut grouped = BTreeMap::>, Vec>>::new(); - if rows.is_empty() && groups.is_empty() { + // PromQL sums over an empty vector emit no sample. + if rows.is_empty() && groups.is_empty() && !input.has_promql_series_identity() { grouped.insert(vec![], vec![]); } for row in rows { @@ -318,6 +320,9 @@ async fn reduce_one( .ok_or_else(|| invalid("integer aggregate overflow"))?; count += 1; } + if count == 0 && !input.has_promql_series_identity() { + return Ok(Value::Null); + } return if matches!(measure, Reduction::Avg(_)) { Ok(Value::Float64(sum as f64 / count as f64)) } else { @@ -334,6 +339,9 @@ async fn reduce_one( }; floats.push(*v); } + if floats.is_empty() && !input.has_promql_series_identity() { + return Ok(Value::Null); + } Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { promql_avg(&floats) } else { diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/executor/src/operators/aggregate/temporal.rs similarity index 96% rename from crates/asap-physical-operators/src/operators/aggregate/temporal.rs rename to crates/executor/src/operators/aggregate/temporal.rs index e40a1fa56..d5c084a59 100644 --- a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs +++ b/crates/executor/src/operators/aggregate/temporal.rs @@ -10,7 +10,8 @@ use crate::{ values::{group_key, Value}, Error, }; -use planner_types::pre_asap::{AggIntent, ColumnRef}; +use planner_types::ir::operator::AggIntent; +use planner_types::ir::scalar::ColumnRef; use std::collections::BTreeMap; pub(in crate::operators) async fn reduce( @@ -307,34 +308,33 @@ mod tests { runtime::{batch_execution::evaluate_batch, Limits, RunContext, Scope}, values::Batch, }; - use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::DataType, - types::AccuracyTarget, + use planner_types::ir::schema::{ + DataType, Field as SummaryField, FieldDataType as SummaryFamilyType, Schema, }; + use planner_types::types::AccuracyTarget; use std::sync::Arc; // The same window operator must give the same answer in either engine phase. #[test] fn temporal_windows_execute_in_both_phases_and_count_is_integer() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![ - Field { - table: None, + SummaryField { name: "time".into(), - dtype: FieldDataType::Plain(DataType::Timestamp), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), nullable: false, - }, - Field { table: None, + }, + SummaryField { name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), + dtype: SummaryFamilyType::Plain(DataType::Float64), nullable: false, + table: None, }, ], time_index: Some(0), + unique_keys: vec![], + closed: false, }); for scope in [ Scope::Query { diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/executor/src/operators/aligned_binary.rs similarity index 95% rename from crates/asap-physical-operators/src/operators/aligned_binary.rs rename to crates/executor/src/operators/aligned_binary.rs index 188d28d6f..909dcc369 100644 --- a/crates/asap-physical-operators/src/operators/aligned_binary.rs +++ b/crates/executor/src/operators/aligned_binary.rs @@ -1,6 +1,8 @@ //! Arithmetic on complete, aligned population/window rows used by precomputation. use super::*; -use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; + use std::collections::BTreeSet; impl Operator { @@ -22,11 +24,9 @@ impl Operator { )); } for (input, value) in [(&left, values.0), (&right, values.1)] { - if input - .fields - .get(value) - .is_none_or(|f| f.nullable || f.dtype != FieldDataType::Plain(DataType::Float64)) - { + if input.fields.get(value).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Float64) + }) { return Err(invalid( "aligned arithmetic requires non-null Float64 values", )); diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/executor/src/operators/common.rs similarity index 91% rename from crates/asap-physical-operators/src/operators/common.rs rename to crates/executor/src/operators/common.rs index 5ef5023cd..f67314bb1 100644 --- a/crates/asap-physical-operators/src/operators/common.rs +++ b/crates/executor/src/operators/common.rs @@ -2,20 +2,20 @@ use super::*; pub(super) fn invalid(message: &str) -> Error { Error::Invalid(message.into()) } -pub(super) fn schema(fields: Vec) -> SchemaRef { +pub(super) fn schema(fields: Vec) -> SchemaRef { Arc::new(Schema { - closed: true, - unique_keys: vec![], fields, + unique_keys: vec![], + closed: false, time_index: None, }) } -pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> Field { - Field { - table: None, +pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { + SummaryField { name: name.into(), - dtype: FieldDataType::Plain(dtype), + dtype: SummaryFamilyType::Plain(dtype), nullable, + table: None, } } @@ -76,4 +76,3 @@ pub(super) fn key_bytes(key: &[Vec]) -> usize { .map(|part| std::mem::size_of::>() + part.len()) .sum::() } -use planner_types::pre_asap::Schema; diff --git a/crates/asap-physical-operators/src/operators/current_series.rs b/crates/executor/src/operators/current_series.rs similarity index 100% rename from crates/asap-physical-operators/src/operators/current_series.rs rename to crates/executor/src/operators/current_series.rs diff --git a/crates/asap-physical-operators/src/operators/filter.rs b/crates/executor/src/operators/filter.rs similarity index 100% rename from crates/asap-physical-operators/src/operators/filter.rs rename to crates/executor/src/operators/filter.rs diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/executor/src/operators/joins/mod.rs similarity index 94% rename from crates/asap-physical-operators/src/operators/joins/mod.rs rename to crates/executor/src/operators/joins/mod.rs index 76e2f748c..60e6d8c93 100644 --- a/crates/asap-physical-operators/src/operators/joins/mod.rs +++ b/crates/executor/src/operators/joins/mod.rs @@ -22,6 +22,15 @@ impl Operator { output: left, }) } + /// Require every candidate key to have an authoritative value at execution. + pub fn certified_semi_join( + left: SchemaRef, + right: SchemaRef, + keys: Vec<(usize, usize)>, + ) -> Result { + Ok(Self::semi_join(left, right, keys)?.require_complete_right()) + } + pub(crate) fn require_complete_right(mut self) -> Self { if let Kind::SemiJoin { require_complete_right, @@ -44,11 +53,11 @@ impl Operator { pub fn relational_join( left: SchemaRef, right: SchemaRef, - kind: planner_types::pre_asap::JoinKind, - predicate: &planner_types::pre_asap::Predicate, + kind: planner_types::ir::operator::JoinKind, + predicate: &planner_types::ir::Predicate, output: SchemaRef, ) -> Result { - use planner_types::pre_asap::JoinKind; + use planner_types::ir::operator::JoinKind; let mut joined = left.fields.clone(); joined.extend(right.fields.clone()); let predicate = @@ -92,7 +101,7 @@ pub(super) fn execute<'a>( let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; return Ok(futures::stream::once(async move { - use planner_types::pre_asap::JoinKind; + use planner_types::ir::operator::JoinKind; let ((left, _left_memory), (right, _right_memory)) = futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; let mut workspace = Workspace::new(&context)?; diff --git a/crates/asap-physical-operators/src/operators/limit.rs b/crates/executor/src/operators/limit.rs similarity index 100% rename from crates/asap-physical-operators/src/operators/limit.rs rename to crates/executor/src/operators/limit.rs diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/executor/src/operators/mod.rs similarity index 84% rename from crates/asap-physical-operators/src/operators/mod.rs rename to crates/executor/src/operators/mod.rs index 4c7e5f5c6..94d44dae0 100644 --- a/crates/asap-physical-operators/src/operators/mod.rs +++ b/crates/executor/src/operators/mod.rs @@ -7,9 +7,9 @@ use crate::{ Error, }; use futures::StreamExt; -use planner_types::{ - post_asap::{Field, FieldDataType, SummaryUpdate}, - pre_asap::{ColumnRef, DataType}, +use planner_types::ir::scalar::ColumnRef; +use planner_types::ir::schema::{ + DataType, Field as SummaryField, FieldDataType as SummaryFamilyType, Schema, SummaryUpdate, }; use std::{collections::BTreeMap, sync::Arc}; pub(crate) mod common; @@ -34,7 +34,7 @@ pub(crate) mod vector_window; pub use aggregate::Reduction; pub use series_window::SubquerySteps; pub use sort::SortKey; -pub use summary::ReadoutQuery; +pub use summary::SummaryEvaluation; #[derive(Clone, serde::Serialize, serde::Deserialize)] enum Kind { #[serde(skip)] @@ -58,20 +58,20 @@ enum Kind { column: usize, }, VectorBinary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, return_bool: bool, }, AlignedBinary { keys: Vec<(usize, usize)>, values: (usize, usize), - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, }, RangeWindow { - intent: Box>, + intent: Box>, }, HistogramQuantile, SeriesWindow { - function: Option>>, + function: Option>>, coordinate: usize, value: usize, range_ms: i64, @@ -79,17 +79,17 @@ enum Kind { at_ms: Option, steps: Option, #[serde(default, skip_serializing_if = "Option::is_none")] - range_at: Option, + range_at: Option, #[serde(default, skip_serializing_if = "Option::is_none")] - steps_range_at: Option, + steps_range_at: Option, }, SeriesLabels { - kind: planner_types::pre_asap::VectorMatchKind, + kind: planner_types::ir::operator::VectorMatchKind, labels: Vec, unique: bool, }, SeriesBinary { - operator: planner_types::post_asap::BinaryOperator, + operator: crate::expressions::binary::BinaryOperator, scalars: [bool; 2], }, SeriesRelabel { @@ -114,7 +114,7 @@ enum Kind { groups: Vec, }, Window { - intent: Box>, + intent: Box>, coordinate: usize, value: usize, groups: Vec, @@ -129,22 +129,22 @@ enum Kind { require_complete_right: bool, }, Join { - kind: planner_types::pre_asap::JoinKind, + kind: planner_types::ir::operator::JoinKind, predicate: Box, }, SummaryBuild { - family: FieldDataType, + family: SummaryFamilyType, value: usize, time: Option, groups: Vec, }, KeyedSummaryBuild { - family: FieldDataType, + family: SummaryFamilyType, value: usize, items: Vec, groups: Vec, }, - KeyedReadout { + KeyedEvaluation { state: usize, k: usize, }, @@ -152,9 +152,9 @@ enum Kind { state: usize, groups: Vec, }, - Readout { + Evaluation { state: usize, - query: ReadoutQuery, + query: SummaryEvaluation, }, } /// A bound operation has a fully checked input/output contract before execution. @@ -175,11 +175,11 @@ impl Operator { } } - pub(crate) fn is_counter_readout(&self) -> bool { + pub(crate) fn is_counter_evaluation(&self) -> bool { matches!( self.kind, - Kind::Readout { - query: ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + Kind::Evaluation { + query: SummaryEvaluation::Exact(crate::summary_kernels::exact::ExactEvaluation { statistic: crate::Statistic::Rate | crate::Statistic::Increase, .. }), @@ -192,21 +192,24 @@ impl Operator { if lookback <= 0 { return Err(invalid("counter lookback must be positive")); } - if let Kind::Readout { - query: ReadoutQuery::Exact(readout), + if let Kind::Evaluation { + query: SummaryEvaluation::Exact(evaluation), .. } = &mut self.kind { - readout.lookback_ms = Some(lookback); + evaluation.lookback_ms = Some(lookback); } Ok(self) } - /// Resolve a counter readout's logical lookback to this run's evaluation range. - pub(super) fn readout_range(&self, context: &RunContext) -> Result, Error> { - let Kind::Readout { + /// Resolve a counter evaluation's logical lookback to this run's evaluation range. + pub(super) fn evaluation_range( + &self, + context: &RunContext, + ) -> Result, Error> { + let Kind::Evaluation { query: - ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + SummaryEvaluation::Exact(crate::summary_kernels::exact::ExactEvaluation { lookback_ms: Some(lookback), .. }), @@ -252,7 +255,7 @@ impl Operator { } if output.time_index.is_some_and(|i| { i >= output.fields.len() - || output.fields[i].dtype != FieldDataType::Plain(DataType::Timestamp) + || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) }) { return Err(invalid("invalid output time column")); } @@ -335,15 +338,15 @@ impl PhysicalOperator for Operator { Kind::SemiJoin { .. } => "SemiJoin", Kind::Join { .. } => "RelationalJoin", Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", - Kind::KeyedReadout { .. } => "SummaryEstimate", + Kind::KeyedEvaluation { .. } => "SummaryEstimate", Kind::SummaryMerge { .. } => "SummaryMerge", - Kind::Readout { .. } => "SummaryReadout", + Kind::Evaluation { .. } => "SummaryEvaluation", } } fn validate_context(&self, context: &RunContext) -> Result<(), Error> { current_series::validate_context(self, context)?; series_window::validate_context(self, context)?; - self.readout_range(context).map(|_| ()) + self.evaluation_range(context).map(|_| ()) } fn input_schemas(&self) -> Vec { self.inputs.clone() @@ -387,9 +390,9 @@ impl PhysicalOperator for Operator { Kind::Join { .. } | Kind::SemiJoin { .. } => joins::execute(self, inputs, context), Kind::SummaryMerge { .. } => summary::execute_merge(self, inputs, context), Kind::SummaryBuild { .. } - | Kind::Readout { .. } + | Kind::Evaluation { .. } | Kind::KeyedSummaryBuild { .. } - | Kind::KeyedReadout { .. } => summary::execute(self, inputs, context), + | Kind::KeyedEvaluation { .. } => summary::execute(self, inputs, context), } } } diff --git a/crates/asap-physical-operators/src/operators/projection.rs b/crates/executor/src/operators/projection.rs similarity index 100% rename from crates/asap-physical-operators/src/operators/projection.rs rename to crates/executor/src/operators/projection.rs diff --git a/crates/asap-physical-operators/src/operators/scope_timestamp.rs b/crates/executor/src/operators/scope_timestamp.rs similarity index 97% rename from crates/asap-physical-operators/src/operators/scope_timestamp.rs rename to crates/executor/src/operators/scope_timestamp.rs index bb7a4e855..33fbdd1ec 100644 --- a/crates/asap-physical-operators/src/operators/scope_timestamp.rs +++ b/crates/executor/src/operators/scope_timestamp.rs @@ -26,7 +26,7 @@ impl Operator { candidate.dtype == field.dtype && candidate.nullable == field.nullable && (candidate.name == field.name - || !matches!(field.dtype, FieldDataType::Plain(_))) + || !matches!(field.dtype, SummaryFamilyType::Plain(_))) }) .map(|(index, _)| index) .collect(); diff --git a/crates/asap-physical-operators/src/operators/series_labels.rs b/crates/executor/src/operators/series_labels.rs similarity index 98% rename from crates/asap-physical-operators/src/operators/series_labels.rs rename to crates/executor/src/operators/series_labels.rs index 2a8c5d1ee..a7d5259fd 100644 --- a/crates/asap-physical-operators/src/operators/series_labels.rs +++ b/crates/executor/src/operators/series_labels.rs @@ -1,10 +1,10 @@ //! PromQL label-set rewriting and binary operators over rows that carry a //! series identity or plain label columns. use super::*; -use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{schema::PROMQL_SERIES_IDENTITY, BinaryOpKind, VectorMatchKind}, -}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; +use planner_types::ir::operator::VectorMatchKind; +use planner_types::ir::schema::PROMQL_SERIES_IDENTITY; type Labels = BTreeMap; @@ -195,7 +195,8 @@ impl Operator { operator: BinaryOperator, scalars: [bool; 2], ) -> Result { - use planner_types::pre_asap::{CompareOpKind::*, GroupSide, PromQLVectorSetOpKind}; + use planner_types::ir::operator::{GroupSide, PromQLVectorSetOpKind}; + use planner_types::ir::scalar::CompareOpKind::*; let vectors = scalars == [false, false]; let valid = match &operator.kind { BinaryOpKind::Arithmetic(_) => true, @@ -318,7 +319,7 @@ async fn series_binary( work: &mut Cooperative, workspace: &mut Workspace, ) -> Result>, Error> { - use planner_types::pre_asap::{GroupSide, PromQLVectorSetOpKind}; + use planner_types::ir::operator::{GroupSide, PromQLVectorSetOpKind}; let drops_name = matches!( binary.kind, BinaryOpKind::Arithmetic(_) | BinaryOpKind::CompareBool(_) diff --git a/crates/asap-physical-operators/src/operators/series_window.rs b/crates/executor/src/operators/series_window.rs similarity index 99% rename from crates/asap-physical-operators/src/operators/series_window.rs rename to crates/executor/src/operators/series_window.rs index b22c8e0ac..7b18f78b0 100644 --- a/crates/asap-physical-operators/src/operators/series_window.rs +++ b/crates/executor/src/operators/series_window.rs @@ -1,6 +1,6 @@ //! PromQL per-series evaluation over the samples before an evaluation instant. use super::*; -use planner_types::pre_asap::{AggIntent, AtModifier}; +use planner_types::ir::operator::{AggIntent, AtModifier}; /// A PromQL subquery grid: every multiple of `step_ms` in /// `(T - offset_ms - range_ms, T - offset_ms]`. `T` is `at_ms` (the subquery's diff --git a/crates/asap-physical-operators/src/operators/sort.rs b/crates/executor/src/operators/sort.rs similarity index 100% rename from crates/asap-physical-operators/src/operators/sort.rs rename to crates/executor/src/operators/sort.rs diff --git a/crates/asap-physical-operators/src/operators/source.rs b/crates/executor/src/operators/source.rs similarity index 100% rename from crates/asap-physical-operators/src/operators/source.rs rename to crates/executor/src/operators/source.rs diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/executor/src/operators/summary/mod.rs similarity index 83% rename from crates/asap-physical-operators/src/operators/summary/mod.rs rename to crates/executor/src/operators/summary/mod.rs index 90cc2c724..dfec3e6d8 100644 --- a/crates/asap-physical-operators/src/operators/summary/mod.rs +++ b/crates/executor/src/operators/summary/mod.rs @@ -1,22 +1,22 @@ use super::*; -/// A summary readout: a sketch query, or an exact readout with typed parameters. +/// A summary evaluation: a sketch query, or an exact evaluation with typed parameters. #[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] -pub enum ReadoutQuery { - Sketch(planner_types::post_asap::SketchStatistic), - Exact(crate::summary_kernels::exact::ExactReadout), +pub enum SummaryEvaluation { + Sketch(planner_types::ir::schema::SketchStatistic), + Exact(crate::summary_kernels::exact::ExactEvaluation), } impl Operator { pub fn keyed_summary_build( input: SchemaRef, - family: FieldDataType, + family: SummaryFamilyType, value: usize, items: Vec, groups: Vec, ) -> Result { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&family)?; - let FieldDataType::Sketch(kind, _) = &family else { + let SummaryFamilyType::Sketch(kind, _) = &family else { return Err(invalid("keyed sketch required")); }; WeightedFrequency::configuration(kind)?; @@ -43,11 +43,11 @@ impl Operator { .iter() .map(|&i| input.fields[i].clone()) .collect::>(); - fields.push(Field { - table: None, + fields.push(SummaryField { name: "state".into(), dtype: family.clone(), nullable: false, + table: None, }); Ok(Self { kind: Kind::KeyedSummaryBuild { @@ -60,7 +60,7 @@ impl Operator { output: schema(fields), }) } - pub fn keyed_readout( + pub fn keyed_evaluation( input: SchemaRef, state: usize, k: usize, @@ -68,31 +68,31 @@ impl Operator { ) -> Result { use crate::summary_kernels::weighted_frequency::WeightedFrequency; crate::values::validate_family(&field(&input, state)?.dtype)?; - let FieldDataType::Sketch(kind, _) = &field(&input, state)?.dtype else { - return Err(invalid("keyed readout requires summary state")); + let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + return Err(invalid("keyed evaluation requires summary state")); }; let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; if k > capacity || output.fields.len() <= input.fields.len() { - return Err(invalid("invalid keyed readout shape or capacity")); + return Err(invalid("invalid keyed evaluation shape or capacity")); } if state + 1 != input.fields.len() || output.fields[..state] != input.fields[..state] - || output.fields.last().unwrap().dtype != FieldDataType::Plain(DataType::Float64) + || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) { return Err(invalid( - "keyed readout must preserve partitions and return a Float64 score", + "keyed evaluation must preserve partitions and return a Float64 score", )); } crate::values::validate_schema(&output)?; Ok(Self { - kind: Kind::KeyedReadout { state, k }, + kind: Kind::KeyedEvaluation { state, k }, inputs: vec![input], output, }) } pub fn summary_build( input: SchemaRef, - family: FieldDataType, + family: SummaryFamilyType, value: usize, time: Option, groups: Vec, @@ -110,9 +110,9 @@ impl Operator { if time.is_none() && matches!( family, - FieldDataType::ExactAggregate( - planner_types::post_asap::ExactKind::Rate - | planner_types::post_asap::ExactKind::Increase, + SummaryFamilyType::ExactAggregate( + planner_types::ir::schema::ExactKind::Rate + | planner_types::ir::schema::ExactKind::Increase, _ ) ) @@ -129,11 +129,11 @@ impl Operator { .iter() .map(|&i| input.fields[i].clone()) .collect::>(); - fields.push(Field { - table: None, + fields.push(SummaryField { name: "state".into(), dtype: family.clone(), nullable: false, + table: None, }); Ok(Self { kind: Kind::SummaryBuild { @@ -153,7 +153,7 @@ impl Operator { ) -> Result { validate_groups(&input, &groups)?; crate::values::validate_family(&field(&input, state)?.dtype)?; - if matches!(field(&input, state)?.dtype, FieldDataType::Plain(_)) { + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { return Err(invalid("summary state required")); } let mut fields = groups @@ -167,21 +167,25 @@ impl Operator { output: schema(fields), }) } - pub fn readout(input: SchemaRef, state: usize, query: ReadoutQuery) -> Result { + pub fn evaluation( + input: SchemaRef, + state: usize, + query: SummaryEvaluation, + ) -> Result { let family = &field(&input, state)?.dtype; crate::values::validate_family(family)?; match &query { - ReadoutQuery::Sketch(query) => { - crate::capability::validate_sketch_readout(family, query)? + SummaryEvaluation::Sketch(query) => { + crate::capability::validate_sketch_evaluation(family, query)? } - ReadoutQuery::Exact(readout) => { - crate::capability::validate_exact_readout(family, readout)? + SummaryEvaluation::Exact(evaluation) => { + crate::capability::validate_exact_evaluation(family, evaluation)? } } let mut fields = input.fields.clone(); let result_type = if matches!( fields[state].dtype, - FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + SummaryFamilyType::ExactAggregate(planner_types::ir::schema::ExactKind::Count, _) ) || integral_count(family, &query) { DataType::Int64 @@ -193,15 +197,15 @@ impl Operator { let nullable = fields.len() == 1 && matches!( fields[state].dtype, - FieldDataType::ExactAggregate( - planner_types::post_asap::ExactKind::Min - | planner_types::post_asap::ExactKind::Max, + SummaryFamilyType::ExactAggregate( + planner_types::ir::schema::ExactKind::Min + | planner_types::ir::schema::ExactKind::Max, _ ) ); fields[state] = result_field("value", result_type, nullable); Ok(Self { - kind: Kind::Readout { state, query }, + kind: Kind::Evaluation { state, query }, inputs: vec![input], output: schema(fields), }) @@ -210,12 +214,12 @@ impl Operator { /// The Planner reads a Count-Min bare count only for count intents, whose /// output is Int64 and whose updates have unit weight; execution rejects a /// non-integral total rather than rounding it. -fn integral_count(family: &FieldDataType, query: &ReadoutQuery) -> bool { - matches!(family, FieldDataType::Sketch(kind, _) - if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::Cms) +fn integral_count(family: &SummaryFamilyType, query: &SummaryEvaluation) -> bool { + matches!(family, SummaryFamilyType::Sketch(kind, _) + if kind.algorithm() == &planner_types::ir::schema::SketchAlgorithm::Cms) && matches!( query, - ReadoutQuery::Sketch(planner_types::post_asap::SketchStatistic::PointCount { + SummaryEvaluation::Sketch(planner_types::ir::schema::SketchStatistic::PointCount { value: None, .. }) @@ -226,7 +230,7 @@ pub(super) fn execute<'a>( mut inputs: Vec>, context: RunContext, ) -> Result, Error> { - let range_ms = operator.readout_range(&context)?; + let range_ms = operator.evaluation_range(&context)?; let output = operator.output.clone(); let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; match &operator.kind { @@ -238,7 +242,7 @@ pub(super) fn execute<'a>( } => Ok(futures::stream::once(async move { Batch::try_new( output, - build_summary(input, family, *value, *time, groups, &context).await?, + build_summary(input, family, *value, *time, groups, !operator.inputs[0].has_promql_series_identity(), &context).await?, ) }) .boxed_local()), @@ -254,7 +258,7 @@ pub(super) fn execute<'a>( ) }) .boxed_local()), - Kind::KeyedReadout { state, k } => Ok(input + Kind::KeyedEvaluation { state, k } => Ok(input .map(move |batch| { let batch = batch?; let mut rows = Vec::new(); @@ -272,7 +276,7 @@ pub(super) fn execute<'a>( // The typed output schema restores epoch-millisecond // timestamp keys from the kernel's Int64 representation. for (value, field) in values.iter_mut().zip(&output.fields) { - if field.dtype == FieldDataType::Plain(DataType::Timestamp) { + if field.dtype == SummaryFamilyType::Plain(DataType::Timestamp) { if let Value::Int64(time) = value { *value = Value::Timestamp(*time); } @@ -284,25 +288,25 @@ pub(super) fn execute<'a>( Batch::try_new(output.clone(), rows) }) .boxed_local()), - Kind::Readout { state, query } => Ok(input + Kind::Evaluation { state, query } => Ok(input .map(move |batch| { let batch = batch?; let mut rows = batch.rows().to_vec(); - if let ReadoutQuery::Exact(readout) = query { + if let SummaryEvaluation::Exact(evaluation) = query { rows.retain(|row| !matches!(&row[*state], Value::Summary { state: summary, .. } - if crate::readout::insufficient_counter_samples(summary.as_ref(), readout.statistic))); + if crate::evaluation::insufficient_counter_samples(summary.as_ref(), evaluation.statistic))); } for row in &mut rows { let Value::Summary { state: summary, .. } = &row[*state] else { return Err(invalid("summary value required")); }; row[*state] = match query { - ReadoutQuery::Sketch(query) => { + SummaryEvaluation::Sketch(query) => { let value = summary .estimate(query) .map_err(|e| Error::Operator(e.to_string()))?; if output.fields[*state].dtype - == FieldDataType::Plain(DataType::Int64) + == SummaryFamilyType::Plain(DataType::Int64) { // Below 2^53 an f64 sum of unit updates is exact. if value.fract() != 0.0 || !(0.0..9.007_199_254_740_992e15).contains(&value) { @@ -315,12 +319,13 @@ pub(super) fn execute<'a>( Value::Float64(value) } } - ReadoutQuery::Exact(readout) => { + SummaryEvaluation::Exact(evaluation) => { let exact = summary .as_any() .downcast_ref::() - .ok_or_else(|| invalid("exact readout requires exact state"))?; - if output.fields[*state].dtype == FieldDataType::Plain(DataType::Int64) { + .ok_or_else(|| invalid("exact evaluation requires exact state"))?; + if output.fields[*state].nullable && exact.is_empty_sum() { Value::Null } + else if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { let count = exact.count().ok_or_else(|| { Error::Operator("exact count state lacks an integer count".into()) })?; @@ -329,7 +334,7 @@ pub(super) fn execute<'a>( })?) } else { match exact - .readout(readout.statistic, range_ms, None) + .evaluation(evaluation.statistic, range_ms, None) .map_err(|e| Error::Operator(e.to_string()))? { Some(value) => Value::Float64(value), @@ -372,10 +377,11 @@ pub(super) fn execute_merge<'a>( async fn build_summary( mut input: Input<'_, Batch>, - family: &FieldDataType, + family: &SummaryFamilyType, value: usize, time: Option, groups: &[usize], + emit_empty_global: bool, context: &RunContext, ) -> Result>, Error> { type State = ( @@ -398,14 +404,15 @@ async fn build_summary( }; let mut work = Cooperative::new(context); let mut states = BTreeMap::>, State>::new(); - if groups.is_empty() { + // PromQL aggregation of an empty vector produces no sample. + if groups.is_empty() && emit_empty_global { states.insert(vec![], create(vec![], 0)?); } let ordered_time = matches!( family, - FieldDataType::ExactAggregate( - planner_types::post_asap::ExactKind::Rate - | planner_types::post_asap::ExactKind::Increase, + SummaryFamilyType::ExactAggregate( + planner_types::ir::schema::ExactKind::Rate + | planner_types::ir::schema::ExactKind::Increase, _ ) ); @@ -472,7 +479,7 @@ async fn merge_summary( groups: &[usize], context: &RunContext, ) -> Result>, Error> { - type GroupState = (Vec, FieldDataType, Arc); + type GroupState = (Vec, SummaryFamilyType, Arc); let mut states: BTreeMap>, GroupState> = BTreeMap::new(); let mut work = Cooperative::new(context); let mut memory = context.reserve(0)?; @@ -531,14 +538,14 @@ async fn merge_summary( async fn build_keyed_summary( mut input: Input<'_, Batch>, - family: &FieldDataType, + family: &SummaryFamilyType, value: usize, items: &[usize], groups: &[usize], context: &RunContext, ) -> Result>, Error> { use crate::{summary_kernels::weighted_frequency::WeightedFrequency, AggregateCore}; - let FieldDataType::Sketch(kind, _) = family else { + let SummaryFamilyType::Sketch(kind, _) = family else { unreachable!() }; let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/executor/src/operators/unchecked.rs similarity index 94% rename from crates/asap-physical-operators/src/operators/unchecked.rs rename to crates/executor/src/operators/unchecked.rs index d3504f9a9..1b5dd7c4f 100644 --- a/crates/asap-physical-operators/src/operators/unchecked.rs +++ b/crates/executor/src/operators/unchecked.rs @@ -138,9 +138,7 @@ impl TryFrom for Operator { input(0)?, input(1)?, kind, - &planner_types::pre_asap::Predicate(std::rc::Rc::new( - predicate.expression().clone(), - )), + &planner_types::ir::Predicate(predicate.expression().clone()), output.clone(), )?, Kind::SummaryBuild { @@ -155,13 +153,13 @@ impl TryFrom for Operator { items, groups, } => Operator::keyed_summary_build(input(0)?, family, value, items, groups)?, - Kind::KeyedReadout { state, k } => { - Operator::keyed_readout(input(0)?, state, k, output.clone())? + Kind::KeyedEvaluation { state, k } => { + Operator::keyed_evaluation(input(0)?, state, k, output.clone())? } Kind::SummaryMerge { state, groups } => { Operator::summary_merge(input(0)?, state, groups)? } - Kind::Readout { state, query } => Operator::readout(input(0)?, state, query)?, + Kind::Evaluation { state, query } => Operator::evaluation(input(0)?, state, query)?, } .with_output_schema(output)?; if serde_json::to_value(&op.kind).map_err(|error| invalid(&error.to_string()))? diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/executor/src/operators/vector_binary.rs similarity index 98% rename from crates/asap-physical-operators/src/operators/vector_binary.rs rename to crates/executor/src/operators/vector_binary.rs index 7027fb9be..0f4af7cf4 100644 --- a/crates/asap-physical-operators/src/operators/vector_binary.rs +++ b/crates/executor/src/operators/vector_binary.rs @@ -1,6 +1,7 @@ //! Label matching and scalar broadcasting are physical computation, not source binding. use super::*; -use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use crate::expressions::binary::BinaryOpKind; +use crate::expressions::binary::BinaryOperator; pub(crate) fn value_schema(scalar: bool) -> SchemaRef { let mut fields = Vec::new(); diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/executor/src/operators/vector_window.rs similarity index 98% rename from crates/asap-physical-operators/src/operators/vector_window.rs rename to crates/executor/src/operators/vector_window.rs index 6b4d58347..afd99c49c 100644 --- a/crates/asap-physical-operators/src/operators/vector_window.rs +++ b/crates/executor/src/operators/vector_window.rs @@ -1,7 +1,6 @@ //! Window bounds are typed input data; aggregation and histogram semantics stay native. use super::*; -use planner_types::pre_asap::AggIntent; -use planner_types::pre_asap::Schema; +use planner_types::ir::operator::AggIntent; pub(crate) fn matrix_schema() -> SchemaRef { let mut fields = vector_binary::value_schema(false).fields.clone(); @@ -9,9 +8,9 @@ pub(crate) fn matrix_schema() -> SchemaRef { fields.push(result_field("window_start", DataType::Timestamp, false)); fields.push(result_field("window_end", DataType::Timestamp, false)); Arc::new(Schema { - closed: true, - unique_keys: vec![], fields, + unique_keys: vec![], + closed: false, time_index: Some(1), }) } diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/executor/src/physical_planner/candidates.rs similarity index 84% rename from crates/asap-physical-operators/src/physical_planner/candidates.rs rename to crates/executor/src/physical_planner/candidates.rs index 0e6ee86ba..005539a20 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/executor/src/physical_planner/candidates.rs @@ -1,12 +1,12 @@ //! Compile maintenance-selected frontiers without deployment-specific DAG rewrites. use super::*; -/// One computation realization; lifecycle/window/revision requirements accompany +/// One computation realization; materialization/window/revision requirements accompany /// it during optimization and deployment. Stored outputs have no storage identity. /// Deserialization validates the producer/reader boundary. #[derive(Clone, serde::Serialize, serde::Deserialize)] -#[serde(try_from = "UncheckedPhysicalASAPDAG")] -pub struct PhysicalASAPDAG { +#[serde(try_from = "UncheckedCompiledPhysicalPlan")] +pub struct CompiledPhysicalPlan { pub precompute: Option, pub query: CompiledPhysicalDAG, pub materialized_outputs: BTreeMap, @@ -14,18 +14,18 @@ pub struct PhysicalASAPDAG { /// Compile an explicit materialization frontier selected by Planner maintenance /// search. Operators upstream of that frontier run in precompute, including -/// readouts/reductions; query execution receives their typed output values. +/// evaluations/reductions; query execution receives their typed output values. /// Empty frontiers retain the full computation in the query DAG. /// /// Repeated windows must be instantiated with the same evaluation/population /// contract used to build each output. This API never treats a result from a /// different window or revision as interchangeable merely because types match. pub fn compile_candidate( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[NodeId], frontier: &[NodeId], -) -> Result { +) -> Result { cut_candidate(&compile(dag, inputs, roots)?, frontier) } @@ -36,9 +36,9 @@ pub fn compile_candidate( pub fn cut_candidate( compiled: &CompiledPhysicalDAG, frontier: &[NodeId], -) -> Result { +) -> Result { if frontier.is_empty() { - return Ok(PhysicalASAPDAG { + return Ok(CompiledPhysicalPlan { precompute: None, query: compiled.clone(), materialized_outputs: BTreeMap::new(), @@ -79,30 +79,32 @@ pub fn cut_candidate( "frontier contains an output shadowed by another boundary", )); } - Ok(PhysicalASAPDAG { + Ok(CompiledPhysicalPlan { precompute: Some(precompute), query, materialized_outputs, }) } -/// Materialization frontier implied by lifecycle-assigned timing: ingestion-time +/// Materialization frontier implied by materialization-assigned timing: ingestion-time /// nodes read by a query-time node, plus the root when it is ingestion-timed. /// `cut_candidate` of one [`compile`] result with this frontier realizes the /// assignment, so different assignments are different cuts of one lowering. /// That holds while timing-dependent lowering (an ingestion-time `Binary` /// aligns by value column) has the same timing at compile time as here. /// A query-time node feeding an ingestion-time node has no valid placement. -pub fn frontier_from_timing(dag: &PostAsapDAG) -> Result, Error> { - use planner_types::post_asap::ExecutionTiming::IngestionTime; +pub fn frontier_from_timing(dag: &PhysicalASAPDAG) -> Result, Error> { + use planner_types::ir::properties::ExecutionTiming::IngestionTime; let timing = dag .nodes .iter() .map(|node| (node.id, node.output_state.timing)) .collect::>(); let mut frontier = BTreeSet::new(); - if timing.get(&dag.root) == Some(&IngestionTime) { - frontier.insert(u64::from(dag.root.0)); + for root in &dag.roots { + if timing.get(root) == Some(&IngestionTime) { + frontier.insert(u64::from(root.0)); + } } for edge in &dag.edges { let (Some(&producer), Some(&consumer)) = @@ -123,11 +125,11 @@ pub fn frontier_from_timing(dag: &PostAsapDAG) -> Result, Error> { /// Enumerate bounded, reachable materialization frontiers above explicit inputs. /// Each frontier is an antichain: storing an output and its ancestor together -/// would leave the ancestor unused by query execution. Lifecycle eligibility +/// would leave the ancestor unused by query execution. Materialization eligibility /// and deployment feasibility are evaluated separately before cost selection. /// Exceeding the search budget returns an error, never a partial inventory. pub fn enumerate_frontiers( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: &BTreeMap, roots: &[NodeId], max_candidates: usize, @@ -192,11 +194,11 @@ fn enumerate_compiled_frontiers( /// individual failures visible; do not substitute another computation on error. /// The DAG is lowered once; each frontier is a [`cut_candidate`] of it. pub fn compile_candidates( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[NodeId], frontiers: &[Vec], -) -> Vec> { +) -> Vec> { match compile(dag, inputs, roots) { Ok(compiled) => frontiers .iter() @@ -216,7 +218,7 @@ pub struct CandidateCost { pub total_cost: f64, } -pub struct CandidateSelection { +pub struct CandidateSelection { pub candidate: T, pub candidate_index: usize, pub cost: CandidateCost, @@ -271,14 +273,14 @@ pub fn select_candidate( #[derive(serde::Deserialize)] #[serde(deny_unknown_fields)] -struct UncheckedPhysicalASAPDAG { +struct UncheckedCompiledPhysicalPlan { precompute: Option, query: CompiledPhysicalDAG, materialized_outputs: BTreeMap, } -impl TryFrom for PhysicalASAPDAG { +impl TryFrom for CompiledPhysicalPlan { type Error = Error; - fn try_from(candidate: UncheckedPhysicalASAPDAG) -> Result { + fn try_from(candidate: UncheckedCompiledPhysicalPlan) -> Result { let result = Self { precompute: candidate.precompute, query: candidate.query, @@ -289,7 +291,7 @@ impl TryFrom for PhysicalASAPDAG { } } -impl PhysicalASAPDAG { +impl CompiledPhysicalPlan { /// Validate the physical handoff, including the producer/reader boundary. pub fn validate(&self) -> Result<(), Error> { self.query.validate()?; @@ -331,7 +333,7 @@ mod tests { use super::*; use planner_types::workload::*; - fn grouped_rate() -> (PostAsapDAG, BTreeMap, NodeId) { + fn grouped_rate() -> (PhysicalASAPDAG, BTreeMap, NodeId) { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -361,14 +363,22 @@ mod tests { let root = asap_frontend_promql::lower_promql_workload(&workload, 0) .unwrap() .remove(0); - let root = std::rc::Rc::new(promql_rows::with_series_identity(&root).unwrap()); - let space = asap_aware_mapping::search_workload(vec![("q", root)]); - let selected = space - .global_selection(&asap_aware_mapping::cost_model::DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) - .unwrap() - .unwrap(); - let dag = planner_types::post_asap::compile_post_asap_dag(&selected).unwrap(); + let root = promql_rows::with_series_identity(&root).unwrap(); + let space = asap_logical_optimizer::search_workload(vec![("q", root)]); + let selected = asap_plan_selection::candidate_selection::global_selection( + &space, + &asap_plan_selection::cost::cost_model::DefaultCostModel, + ) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + let selected = planner_types::ir::apply_materialization_timings( + &selected, + &planner_types::ir::MaterializationAssignment::all_ingestion_time(), + &mut Default::default(), + ) + .unwrap(); + let dag = planner_types::ir::export::compile_physical_asap_dag(&selected).unwrap(); let state = dag .nodes .iter() @@ -378,7 +388,7 @@ mod tests { u64::from(state.id.0), InputContract::bounded(Arc::new(state.output_schema.clone())), )]); - (dag.clone(), inputs, u64::from(dag.root.0)) + (dag.clone(), inputs, u64::from(dag.roots[0].0)) } /// Enumerating and cutting every frontier lowers each Planner node once. @@ -399,9 +409,9 @@ mod tests { } fn with_timing( - dag: &PostAsapDAG, - timing: impl Fn(&PostAsapDAGNode) -> planner_types::post_asap::ExecutionTiming, - ) -> PostAsapDAG { + dag: &PhysicalASAPDAG, + timing: impl Fn(&PhysicalASAPDAGNode) -> planner_types::ir::properties::ExecutionTiming, + ) -> PhysicalASAPDAG { let mut timed = dag.clone(); for node in &mut timed.nodes { node.output_state.timing = timing(node); @@ -413,11 +423,18 @@ mod tests { timed } - fn raw_input(dag: &PostAsapDAG) -> BTreeMap { + fn raw_input(dag: &PhysicalASAPDAG) -> BTreeMap { let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, Payload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + Payload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); BTreeMap::from([( u64::from(raw.id.0), @@ -425,12 +442,12 @@ mod tests { )]) } - /// Cutting one compilation by a retained-state timing and by the all - /// query-time timing (what ContinuouslyMaintained and Ephemeral assign) + /// Cutting one compilation by a maintained-state timing and by the all + /// query-time timing /// lowers each Planner node once and matches `compile_candidate`. #[test] fn timing_cuts_share_one_lowering() { - use planner_types::post_asap::ExecutionTiming::QueryTime; + use planner_types::ir::properties::ExecutionTiming::QueryTime; let (retained, _, root) = grouped_rate(); let ephemeral = with_timing(&retained, |_| QueryTime); let inputs = raw_input(&retained); @@ -460,7 +477,7 @@ mod tests { /// ingestion-time root is itself the frontier. #[test] fn frontier_from_timing_includes_ingestion_root() { - use planner_types::post_asap::ExecutionTiming::IngestionTime; + use planner_types::ir::properties::ExecutionTiming::IngestionTime; let (dag, _, root) = grouped_rate(); let timed = with_timing(&dag, |_| IngestionTime); assert_eq!(frontier_from_timing(&timed).unwrap(), [root]); @@ -469,10 +486,10 @@ mod tests { /// A query-time node feeding an ingestion-time node is rejected. #[test] fn frontier_from_timing_rejects_query_time_input_to_ingestion() { - use planner_types::post_asap::ExecutionTiming::{IngestionTime, QueryTime}; + use planner_types::ir::properties::ExecutionTiming::{IngestionTime, QueryTime}; let (dag, _, _) = grouped_rate(); let timed = with_timing(&dag, |node| { - if node.id == dag.root { + if node.id == dag.roots[0] { IngestionTime } else { QueryTime diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/executor/src/physical_planner/compiled.rs similarity index 100% rename from crates/asap-physical-operators/src/physical_planner/compiled.rs rename to crates/executor/src/physical_planner/compiled.rs diff --git a/crates/executor/src/physical_planner/logical.rs b/crates/executor/src/physical_planner/logical.rs new file mode 100644 index 000000000..dbc541895 --- /dev/null +++ b/crates/executor/src/physical_planner/logical.rs @@ -0,0 +1,374 @@ +//! Reconstruct shared operator references from the transport DAG for native lowering. +use super::*; +use planner_types::ir::export::{EdgeRole, NonASAPOpKind as N, PhysicalASAPNodeId, WireScalarExpr}; +use planner_types::ir::{ + ASAPOp, NonASAPOp, Operator as LogicalOperator, OperatorNode, Predicate, ProjectItem, + ScalarExpr, SortKey as LogicalSortKey, +}; +use std::rc::Rc; +pub(super) fn scalar( + expr: &WireScalarExpr, + id_of: &mut impl FnMut(PhysicalASAPNodeId) -> Rc, +) -> ScalarExpr { + fn boxed( + e: &WireScalarExpr, + id_of: &mut impl FnMut(PhysicalASAPNodeId) -> Rc, + ) -> Box { + Box::new(scalar(e, id_of)) + } + fn list( + es: &[WireScalarExpr], + id_of: &mut impl FnMut(PhysicalASAPNodeId) -> Rc, + ) -> Vec { + es.iter().map(|e| scalar(e, id_of)).collect() + } + match expr { + WireScalarExpr::Column(id) => ScalarExpr::Column(*id), + WireScalarExpr::Literal(v) => ScalarExpr::Literal(v.clone()), + WireScalarExpr::Negative { expr, semantics } => ScalarExpr::Negative { + expr: boxed(expr, id_of), + semantics: *semantics, + }, + WireScalarExpr::Compare { + left, + op, + right, + semantics, + } => ScalarExpr::Compare { + left: boxed(left, id_of), + op: op.clone(), + right: boxed(right, id_of), + semantics: *semantics, + }, + WireScalarExpr::BoolAnd(parts) => ScalarExpr::BoolAnd(list(parts, id_of)), + WireScalarExpr::BoolOr(parts) => ScalarExpr::BoolOr(list(parts, id_of)), + WireScalarExpr::Not(e) => ScalarExpr::Not(boxed(e, id_of)), + WireScalarExpr::IsNull(e) => ScalarExpr::IsNull(boxed(e, id_of)), + WireScalarExpr::IsNotNull(e) => ScalarExpr::IsNotNull(boxed(e, id_of)), + WireScalarExpr::Cast { expr, to, try_cast } => ScalarExpr::Cast { + expr: boxed(expr, id_of), + to: to.clone(), + try_cast: *try_cast, + }, + WireScalarExpr::InList { + expr, + list: items, + negated, + } => ScalarExpr::InList { + expr: boxed(expr, id_of), + list: list(items, id_of), + negated: *negated, + }, + WireScalarExpr::FunctionCall { name, args } => ScalarExpr::FunctionCall { + name: name.clone(), + args: list(args, id_of), + }, + WireScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } => ScalarExpr::Arithmetic { + op: op.clone(), + left: boxed(left, id_of), + right: boxed(right, id_of), + semantics: *semantics, + }, + WireScalarExpr::Case { + operand, + branches, + else_expr, + } => ScalarExpr::Case { + operand: operand.as_ref().map(|e| boxed(e, id_of)), + branches: branches + .iter() + .map(|(w, t)| (scalar(w, id_of), scalar(t, id_of))) + .collect(), + else_expr: else_expr.as_ref().map(|e| boxed(e, id_of)), + }, + WireScalarExpr::CurrentTimestamp => ScalarExpr::CurrentTimestamp, + WireScalarExpr::EvalTimestamp => ScalarExpr::EvalTimestamp, + WireScalarExpr::PromqlScalarFromVector(node) => { + ScalarExpr::PromqlScalarFromVector(id_of(*node)) + } + WireScalarExpr::ScalarSubquery(node) => ScalarExpr::ScalarSubquery(id_of(*node)), + WireScalarExpr::Exists { subquery, negated } => ScalarExpr::Exists { + subquery: id_of(*subquery), + negated: *negated, + }, + WireScalarExpr::InSubquery { + expr, + subquery, + negated, + } => ScalarExpr::InSubquery { + expr: boxed(expr, id_of), + subquery: id_of(*subquery), + negated: *negated, + }, + } +} + +pub(super) fn restore(dag: &PhysicalASAPDAG) -> Result>, Error> { + dag.validate().map_err(|e| invalid(e.to_string()))?; + let mut done = BTreeMap::new(); + let mut remaining: Vec<_> = dag.nodes.iter().collect(); + while !remaining.is_empty() { + let before = remaining.len(); + let mut next = Vec::new(); + for node in remaining { + let mut edges: Vec<_> = dag.edges.iter().filter(|e| e.consumer == node.id).collect(); + if edges + .iter() + .any(|e| !done.contains_key(&u64::from(e.producer.0))) + { + next.push(node); + continue; + } + edges.sort_by_key(|e| match e.role { + EdgeRole::Left => 0, + EdgeRole::Input => 1, + EdgeRole::Right => 2, + EdgeRole::ScalarRef => 3, + }); + let inputs: Vec<_> = edges + .iter() + .filter(|e| e.role != EdgeRole::ScalarRef) + .map(|e| Rc::clone(&done[&u64::from(e.producer.0)])) + .collect(); + let input = |index: usize| { + inputs + .get(index) + .cloned() + .ok_or_else(|| invalid("operator is missing an input")) + }; + let mut missing = false; + let mut ref_node = |id: PhysicalASAPNodeId| { + if let Some(node) = done.get(&u64::from(id.0)) { + Rc::clone(node) + } else { + missing = true; + Rc::new(OperatorNode::with_schema( + LogicalOperator::NonASAP(NonASAPOp::Values { + rows: vec![], + schema: Default::default(), + }), + Default::default(), + )) + } + }; + let mut value = |expr: &WireScalarExpr| scalar(expr, &mut ref_node); + let operator = match &node.payload { + Payload::Relational { operator } => LogicalOperator::NonASAP(match operator { + N::Scan { + source, + predicates, + schema, + } => NonASAPOp::Scan { + source: source.clone(), + predicates: predicates.iter().map(|p| Predicate(value(&p.0))).collect(), + schema: schema.clone(), + }, + N::Values { rows, schema } => NonASAPOp::Values { + rows: rows + .iter() + .map(|r| r.iter().map(&mut value).collect()) + .collect(), + schema: schema.clone(), + }, + N::Filter { pred } => NonASAPOp::Filter { + pred: Predicate(value(&pred.0)), + child: input(0)?, + }, + N::Project { cols, qualifier } => NonASAPOp::Project { + cols: cols + .iter() + .map(|c| ProjectItem { + alias: c.alias.clone(), + expr: value(&c.expr), + }) + .collect(), + qualifier: qualifier.clone(), + child: input(0)?, + }, + N::Aggregate { + reduction, + measures, + output_names, + filters, + having, + } => NonASAPOp::Aggregate { + reduction: reduction.clone(), + measures: measures.clone(), + output_names: output_names.clone(), + filters: filters + .iter() + .map(|p| p.as_ref().map(|p| Predicate(value(&p.0)))) + .collect(), + having: having.as_ref().map(|p| Predicate(value(&p.0))), + child: input(0)?, + }, + N::Join { join_kind, pred } => NonASAPOp::Join { + kind: join_kind.clone(), + pred: Predicate(value(&pred.0)), + left: input(0)?, + right: input(1)?, + }, + N::SetOp { set_kind, all } => NonASAPOp::SetOp { + kind: set_kind.clone(), + all: *all, + left: input(0)?, + right: input(1)?, + }, + N::Concat { + discriminator_unique_key, + } => NonASAPOp::Concat { + children: inputs.clone(), + discriminator_unique_key: discriminator_unique_key.clone(), + }, + N::Dedup { cols } => NonASAPOp::Dedup { + cols: cols.clone(), + child: input(0)?, + }, + N::Sort { keys, partition_by } => NonASAPOp::Sort { + keys: keys + .iter() + .map(|k| LogicalSortKey { + expr: value(&k.expr), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect(), + partition_by: partition_by.clone(), + child: input(0)?, + }, + N::Limit { + n, + offset, + partition_by, + } => NonASAPOp::Limit { + n: *n, + offset: *offset, + partition_by: partition_by.clone(), + child: input(0)?, + }, + N::BinaryOp { + operator, + return_bool, + } => NonASAPOp::BinaryOp { + operator: operator.clone(), + return_bool: *return_bool, + lhs: input(0)?, + rhs: input(1)?, + }, + N::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + } => NonASAPOp::SQLWindowFunc { + func: func.clone(), + args: args.iter().map(&mut value).collect(), + partition_by: partition_by.clone(), + order_by: order_by + .iter() + .map(|k| LogicalSortKey { + expr: value(&k.expr), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect(), + frame: frame.clone(), + output_name: output_name.clone(), + child: input(0)?, + }, + N::TimeRange { range, range_kind } => NonASAPOp::TimeRange { + range: *range, + kind: *range_kind, + child: input(0)?, + }, + N::TimeShift { shift } => NonASAPOp::TimeShift { + shift: *shift, + child: input(0)?, + }, + N::PromqlVectorFromScalar { expr } => { + NonASAPOp::PromqlVectorFromScalar(value(expr)) + } + N::PromqlRelabel { dst, value: expr } => NonASAPOp::PromqlRelabel { + dst: dst.clone(), + value: value(expr), + child: input(0)?, + }, + N::PromqlInfoEnrich { selector } => NonASAPOp::PromqlInfoEnrich { + selector: selector.clone(), + child: input(0)?, + }, + N::PromqlSeriesSample { by, sample_kind } => NonASAPOp::PromqlSeriesSample { + by: by.clone(), + kind: *sample_kind, + child: input(0)?, + }, + N::PromqlSubquery { range, resolution } => NonASAPOp::PromqlSubquery { + range: *range, + resolution: *resolution, + child: input(0)?, + }, + }), + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + filter, + } => LogicalOperator::ASAP(ASAPOp::SummaryAgg { + child: input(0)?, + family: family.clone(), + input: update.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: filter.as_ref().map(|p| Predicate(value(&p.0))), + }), + Payload::SummaryEstimate { query } => { + LogicalOperator::ASAP(ASAPOp::SummaryEstimate { + summary_input: input(0)?, + query: query.clone(), + }) + } + Payload::FinalizeExactAccumulator => { + LogicalOperator::ASAP(ASAPOp::FinalizeExactAccumulator { child: input(0)? }) + } + Payload::MaintainPopulation { population } => { + LogicalOperator::ASAP(ASAPOp::MaintainPopulation { + child: input(0)?, + population: population.clone(), + }) + } + Payload::EvaluatePopulation { evaluation } => { + LogicalOperator::ASAP(ASAPOp::EvaluatePopulation { + child: input(0)?, + evaluation: evaluation.clone(), + }) + } + Payload::SummaryMerge => LogicalOperator::ASAP(ASAPOp::SummaryMerge { + children: inputs.clone(), + }), + _ => return Err(invalid("reserved ASAP operation has no native lowering")), + }; + if missing { + return Err(invalid( + "scalar reference is not a preceding DAG dependency", + )); + } + let mut rebuilt = OperatorNode::with_schema(operator, node.output_schema.clone()); + rebuilt.guarantee = node.guarantee.clone(); + rebuilt.timing = Some(node.output_state.timing); + done.insert(u64::from(node.id.0), Rc::new(rebuilt)); + } + if next.len() == before { + return Err(invalid("operator DAG is cyclic")); + } + remaining = next; + } + Ok(done) +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/executor/src/physical_planner/mod.rs similarity index 79% rename from crates/asap-physical-operators/src/physical_planner/mod.rs rename to crates/executor/src/physical_planner/mod.rs index e1490c03c..6197df233 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/executor/src/physical_planner/mod.rs @@ -1,23 +1,22 @@ //! Compile logical computation to native operators with typed external inputs. //! Compilation needs no readers; deployment resolves inputs after selection. -use crate::operators::ReadoutQuery; -use crate::summary_kernels::exact::ExactReadout; +use crate::operators::SummaryEvaluation; +use crate::summary_kernels::exact::ExactEvaluation; use crate::{ operators::{Expression, Operator, Reduction, SortKey}, plan::{Boundedness, Emission, NodeId, PhysicalDAG, PhysicalOperator, PlanProperties}, values::{Batch, SchemaRef}, Error, }; -use planner_types::{ - post_asap::{ - ExactOperation, FieldDataType, PostAsapDAG, PostAsapDAGNode, - PostAsapOperatorPayload as Payload, SketchStatistic, SummaryInputExpr, ValueOperation, - }, - pre_asap::{ - AggIntent, ColumnRef, CompareOpKind, DataType, GroupKeys, QueryExpr, - Reduction as PlannerReduction, - }, +use planner_types::ir::export::{ + NonASAPOpKind, PhysicalASAPDAG, PhysicalASAPDAGNode, PhysicalASAPOperatorPayload as Payload, + WireScalarExpr, }; +use planner_types::ir::operator::{AggIntent, GroupKeys, Reduction as PlannerReduction}; +use planner_types::ir::scalar::{ColumnRef, CompareOpKind}; +use planner_types::ir::schema::{DataType, FieldDataType, SketchStatistic, SummaryInputExpr}; +use planner_types::ir::{ASAPOp, NonASAPOp, Operator as LogicalOperator, OperatorNode, ScalarExpr}; +mod logical; use std::{ collections::{BTreeMap, BTreeSet}, sync::Arc, @@ -39,7 +38,8 @@ pub mod promql_values; mod candidates; pub use candidates::{ compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, - frontier_from_timing, select_candidate, CandidateCost, CandidateSelection, PhysicalASAPDAG, + frontier_from_timing, select_candidate, CandidateCost, CandidateSelection, + CompiledPhysicalPlan, }; mod compiled; @@ -50,7 +50,7 @@ mod row_values; /// Compile computation without opening or retaining deployment readers. /// Input contracts identify explicit boundaries selected by maintenance planning. pub fn compile( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[NodeId], ) -> Result { @@ -60,7 +60,7 @@ pub fn compile( /// Convenience for callers that already resolved inputs. Lowering still uses /// only their contracts, and instantiation checks those contracts again. pub fn bind<'a>( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, sources: BTreeMap>, roots: &[NodeId], ) -> Result, Error> { @@ -73,11 +73,12 @@ pub fn bind<'a>( /// Resolve raw scan connectors before invoking the reader-independent compiler. pub fn bind_with_data_sources<'a>( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, mut sources: BTreeMap>, roots: &[NodeId], data_sources: &crate::sources::DataSources, ) -> Result, Error> { + let restored = logical::restore(dag)?; // Only resolve scans reachable below the selected input boundaries. let mut pending = roots.to_vec(); let mut seen = BTreeSet::new(); @@ -85,16 +86,13 @@ pub fn bind_with_data_sources<'a>( if !seen.insert(id) || sources.contains_key(&id) { continue; } - let node = dag + let _node = dag .nodes .iter() .find(|n| u64::from(n.id.0) == id) .ok_or_else(|| invalid(format!("missing node {id}")))?; - if let Payload::Fallback { - expression: expression @ QueryExpr::Scan { .. }, - } = &node.payload - { - sources.insert(id, Box::new(data_sources.bind(expression)?)); + if matches!(restored[&id].non_asap(), Some(NonASAPOp::Scan { .. })) { + sources.insert(id, Box::new(data_sources.bind(&restored[&id])?)); } else { pending.extend( dag.edges @@ -123,11 +121,12 @@ fn helper_id(node: NodeId, index: u64) -> NodeId { } fn compile_internal( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, mut sources: BTreeMap, roots: &[NodeId], ) -> Result { preflight_depth(dag)?; + let restored = logical::restore(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag .nodes @@ -141,46 +140,43 @@ fn compile_internal( ( edge.consumer.0, match edge.role { - planner_types::post_asap::EdgeRole::Left => 0, - planner_types::post_asap::EdgeRole::Input => 1, - planner_types::post_asap::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::Left => 0, + planner_types::ir::export::EdgeRole::Input => 1, + planner_types::ir::export::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::ScalarRef => 3, }, ) }); - // Scalar literal operands of query-time arithmetic are folded into the consumer. - let mut literals = BTreeMap::::new(); + let literals = BTreeMap::::new(); for edge in edges { - let consumer = u64::from(edge.consumer.0); - if let ( - Payload::Fallback { expression }, - Some(PostAsapDAGNode { - payload: Payload::Binary { .. }, - .. - }), - ) = ( - &nodes[&u64::from(edge.producer.0)].payload, - nodes.get(&consumer), - ) { - if let Some(value) = row_values::scalar_literal(expression) { - let left = edge.role == planner_types::post_asap::EdgeRole::Left; - if literals.insert(consumer, (value, left)).is_some() { - return Err(invalid("binary with two scalar literals is not folded")); - } - continue; - } - } dependencies .entry(u64::from(edge.consumer.0)) .or_default() .push(u64::from(edge.producer.0)); } + let mut fallback = BTreeMap::new(); + for (&id, root) in &restored { + let raw_summary_input = matches!(root.non_asap(), Some(NonASAPOp::TimeRange { .. })) + && dag.edges.iter().any(|e| { + u64::from(e.producer.0) == id + && matches!( + nodes[&u64::from(e.consumer.0)].payload, + Payload::SummaryAgg { .. } + ) + }); + if !root.contains_asap() && !raw_summary_input { + if let Ok(lowered) = promql_fallback::lower(root) { + fallback.insert(id, lowered); + } + } + } let known = |id: &NodeId| { nodes.contains_key(id) || promql_fallback::raw_series_owner(*id).is_some_and(|owner| { matches!( nodes.get(&owner), - Some(PostAsapDAGNode { - payload: Payload::Fallback { .. }, + Some(PhysicalASAPDAGNode { + payload: Payload::Relational { .. }, .. }) ) @@ -204,7 +200,7 @@ fn compile_internal( return Err(invalid(format!("missing root {id}"))); } pending.push((id, true)); - if !sources.contains_key(&id) { + if !sources.contains_key(&id) && !fallback.contains_key(&id) { for &input in dependencies.get(&id).into_iter().flatten() { pending.push((input, false)); } @@ -241,20 +237,11 @@ fn compile_internal( inputs = vec![auxiliary]; schemas.truncate(1); } - // A consumed bare selector supplies raw range rows (e.g. to a - // per-entity summary), not an instant vector, so only its consumer computes. - let raw_rows = matches!( - &node.payload, - Payload::Fallback { - expression: QueryExpr::TimeRange { .. } - } - ) && dag.edges.iter().any(|e| u64::from(e.producer.0) == id); - if let (Payload::Fallback { expression }, false) = (&node.payload, raw_rows) { - let promql_fallback::Lowering { - selectors, - mut steps, - } = promql_fallback::lower(expression) - .map_err(|error| invalid(format!("node {id}: {error}")))?; + if let Some(promql_fallback::Lowering { + selectors, + mut steps, + }) = fallback.remove(&id) + { let mut slots = Vec::new(); for (i, (_, schema)) in selectors.iter().enumerate() { let slot = promql_fallback::raw_series_input(id, i); @@ -300,11 +287,8 @@ fn compile_internal( )?; continue; } - if let Payload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &node.payload - { - use planner_types::post_asap::maintained_population::PopulationInput; + if let Payload::MaintainPopulation { population } = &node.payload { + use planner_types::ir::operator::maintained_population::PopulationInput; let PopulationInput::CurrentSeries(spec) = &population.input else { return Err(invalid( "native maintained population requires a current-series input", @@ -336,22 +320,16 @@ fn compile_internal( )?; continue; } - if let Payload::Value { - operation: ValueOperation::ReadPopulation { readout }, - } = &node.payload - { - use planner_types::post_asap::maintained_population::{ + if let Payload::EvaluatePopulation { evaluation } = &node.payload { + use planner_types::ir::operator::maintained_population::{ PopulationInput, PopulationStatistic, }; let [producer] = inputs.as_slice() else { - return Err(invalid("population readout requires one input")); + return Err(invalid("population evaluation requires one input")); }; - let Payload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &nodes[producer].payload - else { + let Payload::MaintainPopulation { population } = &nodes[producer].payload else { return Err(invalid( - "population readout requires its declared population", + "population evaluation requires its declared population", )); }; let PopulationInput::CurrentSeries(spec) = &population.input else { @@ -363,9 +341,9 @@ fn compile_internal( )); } let input = schemas[0].clone(); - let PopulationStatistic::TopK { k } = readout else { + let PopulationStatistic::TopK { k } = evaluation else { let mut chain = - row_values::population_aggregate(&input, &spec.grouping, readout)?; + row_values::population_aggregate(&input, &spec.grouping, evaluation)?; let last = chain.pop().expect("nonempty chain"); let mut inputs = inputs; for operator in chain { @@ -415,15 +393,12 @@ fn compile_internal( let [input_id] = inputs.as_slice() else { return Err(invalid("per-entity summary requires one input")); }; - let Payload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, - } = &nodes[input_id].payload - else { + let Some(NonASAPOp::TimeRange { child, .. }) = restored[input_id].non_asap() else { return Err(invalid( "per-entity summary requires a resolved raw time range", )); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let Some(NonASAPOp::Scan { schema, .. }) = child.non_asap() else { return Err(invalid("per-entity summary requires a resolved source")); }; if !schema.closed || update.item.is_some() { @@ -462,9 +437,20 @@ fn compile_internal( )?; continue; } - if let Payload::Binary { operator } = &node.payload { + if let Payload::Relational { + operator: + NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } = &node.payload + { + let operator = crate::expressions::binary::BinaryOperator::from_logical( + operator, + *return_bool, + ); let query_time = node.output_state.timing - == planner_types::post_asap::ExecutionTiming::QueryTime; + == planner_types::ir::properties::ExecutionTiming::QueryTime; if let Some(&(value, left)) = literals.get(&id) { let [input] = schemas.as_slice() else { return Err(invalid("scalar binary requires one row input")); @@ -505,19 +491,11 @@ fn compile_internal( // carry the series identity. if let (true, [left, right]) = (query_time, schemas.as_slice()) { if !label_map(left) && !label_map(right) { - // A scalar-valued Fallback operand, such as `scalar(x)`, has no labels. - let scalar = |input: &NodeId| { - matches!( - nodes.get(input).map(|node| &node.payload), - Some(Payload::Fallback { expression }) - if promql_fallback::scalar(expression) - ) - }; let binary = Operator::series_binary( left.clone(), right.clone(), operator.clone(), - [scalar(&inputs[0]), scalar(&inputs[1])], + [false, false], ) .map_err(|error| invalid(format!("node {id}: {error}")))?; physical_dag.add(id, inputs, binary.with_output_schema(output)?)?; @@ -525,14 +503,11 @@ fn compile_internal( } } } - if let Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - } = &node.payload - { + if let Payload::FinalizeExactAccumulator = &node.payload { // Exact counts read out as Int64; PromQL declares a Float64 sample. - let readout = bind_operation(node, &schemas) + let evaluation = bind_operation(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; - let actual = readout.schema(); + let actual = evaluation.schema(); let converted = actual.fields.iter().zip(&output.fields).position(|(a, d)| { a.dtype == FieldDataType::Plain(DataType::Int64) && d.dtype == FieldDataType::Plain(DataType::Float64) @@ -555,8 +530,8 @@ fn compile_internal( .collect(); let project = Operator::project(actual, columns)?.with_output_schema(output.clone())?; - physical_dag.add(auxiliary, inputs, readout)?; - if temporal_readout_drops_name(node) { + physical_dag.add(auxiliary, inputs, evaluation)?; + if temporal_evaluation_drops_name(node) { physical_dag.add(auxiliary - 1, vec![auxiliary], project)?; physical_dag.add( id, @@ -572,7 +547,7 @@ fn compile_internal( } let mut operator = compile_node(node, &schemas) .map_err(|error| invalid(format!("node {id}: {error}")))?; - if operator.is_counter_readout() { + if operator.is_counter_evaluation() { let mut pending = vec![id]; let mut visited = BTreeSet::new(); let mut ranges = BTreeSet::new(); @@ -580,8 +555,8 @@ fn compile_internal( if !visited.insert(ancestor) { continue; } - if let Payload::Fallback { - expression: QueryExpr::TimeRange { range, .. }, + if let Payload::Relational { + operator: NonASAPOpKind::TimeRange { range, .. }, } = &nodes[&ancestor].payload { ranges.insert( @@ -593,13 +568,13 @@ fn compile_internal( pending.extend(dependencies.get(&ancestor).into_iter().flatten().copied()); } if ranges.len() > 1 { - return Err(invalid("counter readout has ambiguous logical windows")); + return Err(invalid("counter evaluation has ambiguous logical windows")); } if let Some(lookback) = ranges.into_iter().next() { operator = operator.with_counter_lookback(lookback)?; } } - if temporal_readout_drops_name(node) { + if temporal_evaluation_drops_name(node) { physical_dag.add(auxiliary, inputs, operator)?; physical_dag.add(id, vec![auxiliary], Operator::series_without_name(output)?)?; } else { @@ -611,42 +586,50 @@ fn compile_internal( Ok(physical_dag) } -// Temporal summary readouts produce PromQL vectors, whose range functions drop +// Temporal summary evaluations produce PromQL vectors, whose range functions drop // the metric name before matching/filtering. Stored state retains its full identity. -fn temporal_readout_drops_name(node: &PostAsapDAGNode) -> bool { +fn temporal_evaluation_drops_name(node: &PhysicalASAPDAGNode) -> bool { node.output_schema .fields .iter() .any(|field| field.name == promql_rows::SERIES_IDENTITY_COLUMN) && matches!( &node.payload, - Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } | Payload::SummaryEstimate { - query: SketchStatistic::Quantile { .. } - | SketchStatistic::Cardinality - | SketchStatistic::PointCount { .. } - | SketchStatistic::FrequencyL2 - | SketchStatistic::FrequencyEntropy - } + Payload::FinalizeExactAccumulator + | Payload::SummaryEstimate { + query: SketchStatistic::Quantile { .. } + | SketchStatistic::Cardinality + | SketchStatistic::PointCount { .. } + | SketchStatistic::FrequencyL2 + | SketchStatistic::FrequencyEntropy + } ) } /// Bind a Planner node against the schemas supplied by its deployment edges. /// This is the same checked path used by complete DAG binding. -pub fn compile_node(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { +pub fn compile_node(node: &PhysicalASAPDAGNode, inputs: &[SchemaRef]) -> Result { for schema in inputs { crate::values::validate_schema(schema)?; } bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) } -fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result { - if let Payload::Binary { operator } = &node.payload { +fn bind_operation(node: &PhysicalASAPDAGNode, inputs: &[SchemaRef]) -> Result { + if let Payload::Relational { + operator: NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } = &node.payload + { + let operator = + crate::expressions::binary::BinaryOperator::from_logical(operator, *return_bool); let [left, right] = inputs else { return Err(invalid("binary requires two inputs")); }; - if node.output_state.timing == planner_types::post_asap::ExecutionTiming::IngestionTime { + if node.output_state.timing == planner_types::ir::properties::ExecutionTiming::IngestionTime + { let value = |schema: &SchemaRef| -> Result { let columns = schema .fields @@ -654,7 +637,7 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result>(); @@ -688,54 +671,83 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result, Error>>() + }) + .collect::, Error>>()?; + let schema = Arc::new(schema.clone()); + return Operator::source( + schema.clone(), + vec![crate::values::Batch::try_new(schema, rows)?], + ); + } let [input] = inputs else { return Err(invalid( "native Planner binding currently requires a unary operation or an explicit source", )); }; match &node.payload { - Payload::Value { operation, .. } => match operation { - ValueOperation::Project { cols, .. } => Operator::project( + Payload::FinalizeExactAccumulator => { + let state = summary_column(input)?; + use crate::Statistic as S; + use planner_types::ir::schema::ExactKind as E; + let statistic = match &input.fields[state].dtype { + FieldDataType::ExactAggregate(kind, _) => match kind { + E::Sum => S::Sum, + E::Count => S::Count, + E::Min => S::Min, + E::Max => S::Max, + E::Rate => S::Rate, + E::Increase => S::Increase, + _ => return Err(invalid("exact family evaluation is unsupported")), + }, + _ => return Err(invalid("exact finalization requires exact state")), + }; + Operator::evaluation( + input.clone(), + state, + SummaryEvaluation::Exact(ExactEvaluation { + statistic, + lookback_ms: None, + }), + ) + } + + Payload::Relational { operator } => match operator { + NonASAPOpKind::Project { cols, .. } => Operator::project( input.clone(), cols.iter() .enumerate() @@ -748,21 +760,21 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result Expression::Column(*index), + WireScalarExpr::Column(index) => Expression::Column(*index), expr => expression(expr, input)?, }, )) }) .collect::>()?, ), - ValueOperation::Filter { pred } => { + NonASAPOpKind::Filter { pred } => { Operator::filter(input.clone(), expression(&pred.0, input)?) } - ValueOperation::Sort { keys, partition_by } => Operator::sort( + NonASAPOpKind::Sort { keys, partition_by } => Operator::sort( input.clone(), keys.iter() .map(|key| { - let QueryExpr::Column(column) = key.expr else { + let WireScalarExpr::Column(column) = key.expr else { return Err(invalid( "sort expression must be projected before sorting", )); @@ -776,23 +788,23 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result>()?, groups(input, partition_by)?, ), - ValueOperation::Limit { + NonASAPOpKind::Limit { n, offset, partition_by, } => Operator::limit( input.clone(), - *n as u64, + n.unwrap_or(usize::MAX) as u64, *offset as u64, groups(input, partition_by)?, ), - ValueOperation::Exact(ExactOperation::Aggregate { + NonASAPOpKind::Aggregate { reduction, measures, output_names, filters, having: None, - }) => { + } => { if filters.iter().any(Option::is_some) { return Err(invalid("filtered aggregate has no native implementation")); } @@ -829,31 +841,6 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result>()?; Operator::aggregate(input.clone(), groups(input, keys)?, measures) } - ValueOperation::FinalizeExactAccumulator => { - let state = summary_column(input)?; - use crate::Statistic as S; - use planner_types::post_asap::ExactKind as E; - let statistic = match &input.fields[state].dtype { - FieldDataType::ExactAggregate(kind, _) => match kind { - E::Sum => S::Sum, - E::Count => S::Count, - E::Min => S::Min, - E::Max => S::Max, - E::Rate => S::Rate, - E::Increase => S::Increase, - _ => return Err(invalid("exact family readout is unsupported")), - }, - _ => return Err(invalid("exact finalization requires exact state")), - }; - Operator::readout( - input.clone(), - state, - ReadoutQuery::Exact(ExactReadout { - statistic, - lookback_ms: None, - }), - ) - } _ => Err(invalid("value operation has no native implementation")), }, Payload::SummaryAgg { @@ -877,10 +864,10 @@ fn bind_operation(node: &PostAsapDAGNode, inputs: &[SchemaRef]) -> Result Result { if let SketchStatistic::TopK { k } = query { - return Operator::keyed_readout( + return Operator::keyed_evaluation( input.clone(), summary_column(input)?, *k, Arc::new(node.output_schema.clone()), ); } - Operator::readout( + Operator::evaluation( input.clone(), summary_column(input)?, - ReadoutQuery::Sketch(query.clone()), + SummaryEvaluation::Sketch(query.clone()), ) } _ => Err(invalid( @@ -1009,9 +996,10 @@ fn groups(input: &SchemaRef, groups: &GroupKeys) -> Result, Error> { } Ok(groups.keys().to_vec()) } -fn expression(expr: &QueryExpr, input: &SchemaRef) -> Result { +fn expression(expr: &WireScalarExpr, input: &SchemaRef) -> Result { + let expr = local_scalar(expr)?; Ok(Expression::planner( - crate::expressions::CompiledExpression::compile(expr, input)?, + crate::expressions::CompiledExpression::compile(&expr, input)?, )) } @@ -1057,7 +1045,7 @@ impl PhysicalOperator for CheckedSource<'_> { } // Bound recursion before invoking the upstream recursive provenance validator. -fn preflight_depth(dag: &PostAsapDAG) -> Result<(), Error> { +fn preflight_depth(dag: &PhysicalASAPDAG) -> Result<(), Error> { let mut remaining = dag .nodes .iter() @@ -1110,23 +1098,24 @@ fn preflight_depth(dag: &PostAsapDAG) -> Result<(), Error> { /// Join predicates address the concatenated left/right schema. fn semi_join_keys( - expr: &QueryExpr, + expr: &ScalarExpr, left: usize, right: usize, keys: &mut Vec<(usize, usize)>, ) -> Result<(), Error> { match expr { - QueryExpr::BoolAnd(parts) => { + ScalarExpr::BoolAnd(parts) => { for part in parts { semi_join_keys(part, left, right, keys)?; } } - QueryExpr::Compare { + ScalarExpr::Compare { left: a, op: CompareOpKind::Eq, right: b, + .. } => { - let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + let (ScalarExpr::Column(a), ScalarExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { return Err(invalid("semi-join requires column equality keys")); }; let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; @@ -1143,9 +1132,9 @@ fn semi_join_keys( /// Resolve equality keys against the Planner join's concatenated input schema. /// Deployments may use these positions to bind their source columns. pub fn equijoin_keys( - pred: &planner_types::pre_asap::Predicate, - left: &planner_types::post_asap::Schema, - right: &planner_types::post_asap::Schema, + pred: &planner_types::ir::Predicate, + left: &planner_types::ir::schema::Schema, + right: &planner_types::ir::schema::Schema, ) -> Result, Error> { let mut keys = Vec::new(); semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; @@ -1154,3 +1143,24 @@ pub fn equijoin_keys( } Ok(keys) } + +fn local_scalar(expr: &WireScalarExpr) -> Result { + let mut missing = false; + let result = logical::scalar(expr, &mut |_| { + missing = true; + std::rc::Rc::new(OperatorNode::with_schema( + LogicalOperator::NonASAP(NonASAPOp::Values { + rows: vec![], + schema: Default::default(), + }), + Default::default(), + )) + }); + if missing { + Err(invalid( + "scalar plan reads require explicit execution bindings", + )) + } else { + Ok(result) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/executor/src/physical_planner/precompute.rs similarity index 82% rename from crates/asap-physical-operators/src/physical_planner/precompute.rs rename to crates/executor/src/physical_planner/precompute.rs index 10dffdc0c..fc9c9c14c 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/executor/src/physical_planner/precompute.rs @@ -1,42 +1,41 @@ //! Compile immutable summary-input computation with explicit population and pane identity. use super::promql_rows::SERIES_IDENTITY_COLUMN as SERIES_IDENTITY; use super::*; -use planner_types::{ - post_asap::{ExecutionTiming, GroupingStrategy, Schema}, - pre_asap::DataType, -}; +use planner_types::ir::properties::ExecutionTiming; +use planner_types::ir::schema::FieldDataType as SummaryFamilyType; +use planner_types::ir::schema::{DataType, GroupingStrategy, Schema}; /// Physical rows carry the population and pane coordinate alongside the logical value. /// These fields preserve identities which are implicit in a stored summary instance. -pub fn population_schema(family: FieldDataType) -> SchemaRef { +pub fn population_schema(family: SummaryFamilyType) -> SchemaRef { Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![ - planner_types::post_asap::Field { - table: None, + planner_types::ir::schema::Field { name: "$population".into(), - dtype: FieldDataType::Plain(DataType::Map { + dtype: SummaryFamilyType::Plain(DataType::Map { key: Box::new(DataType::Utf8), value: Box::new(DataType::Utf8), value_nullable: false, }), nullable: false, - }, - planner_types::post_asap::Field { table: None, + }, + planner_types::ir::schema::Field { name: "$window_end".into(), - dtype: FieldDataType::Plain(DataType::Timestamp), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), nullable: false, - }, - planner_types::post_asap::Field { table: None, + }, + planner_types::ir::schema::Field { name: "value".into(), dtype: family, nullable: false, + table: None, }, ], time_index: Some(1), + unique_keys: vec![], + closed: false, }) } @@ -48,7 +47,7 @@ pub fn population_schema(family: FieldDataType) -> SchemaRef { /// must be canonical (sorted, unique, no empty values), since they are the /// population identity: build rows with [`raw_sample_row`]. pub fn raw_sample_schema() -> SchemaRef { - let mut schema = (*population_schema(FieldDataType::Plain(DataType::Float64))).clone(); + let mut schema = (*population_schema(SummaryFamilyType::Plain(DataType::Float64))).clone(); schema.fields[1].name = "$timestamp".into(); Arc::new(schema) } @@ -82,19 +81,14 @@ pub fn raw_sample_row( /// Input contract of a precompute boundary: raw sample rows for a raw time /// series scan, otherwise the stored population of its summary state. -pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { - let Payload::Fallback { expression } = &node.payload else { - return source_schema(&node.output_schema); - }; - let scan = match expression { - planner_types::pre_asap::QueryExpr::TimeRange { child, .. } => child.as_ref(), - expression => expression, - }; +pub fn boundary_schema(node: &PhysicalASAPDAGNode) -> Result { if !matches!( - scan, - planner_types::pre_asap::QueryExpr::Scan { - source: planner_types::pre_asap::Source::TimeSeries { .. }, - .. + &node.payload, + Payload::Relational { + operator: NonASAPOpKind::Scan { + source: planner_types::ir::operator::Source::TimeSeries { .. }, + .. + } | NonASAPOpKind::TimeRange { .. } } ) { return source_schema(&node.output_schema); @@ -106,11 +100,11 @@ pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { .iter() .enumerate() .all(|(i, field)| match &field.dtype { - FieldDataType::Plain(DataType::Timestamp) => { + SummaryFamilyType::Plain(DataType::Timestamp) => { Some(i) == logical.time_index && !field.nullable } - FieldDataType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, - FieldDataType::Plain(DataType::Utf8) => true, + SummaryFamilyType::Plain(DataType::Float64) => field.name == "value" && !field.nullable, + SummaryFamilyType::Plain(DataType::Utf8) => true, _ => false, }) && !logical @@ -134,14 +128,14 @@ pub fn source_schema(logical: &Schema) -> Result { let states = logical .fields .iter() - .filter(|f| !matches!(f.dtype, FieldDataType::Plain(_))) + .filter(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) .collect::>(); let [state] = states.as_slice() else { return Err(invalid( "stored population requires one typed summary state", )); }; - if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, FieldDataType::Plain(dtype) + if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, SummaryFamilyType::Plain(dtype) if field.nullable || !matches!(dtype, DataType::Utf8) && !(Some(i) == logical.time_index && *dtype == DataType::Timestamp))) { return Err(invalid("stored population metadata cannot reconstruct extra value columns")); } @@ -161,7 +155,7 @@ pub fn is_population_schema(schema: &SchemaRef) -> bool { /// Compile a complete selected precompute sub-DAG. Inputs are already-computed /// state boundaries; the deployment supplies groups, panes and states, never operations. pub fn compile( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, frontiers: &[NodeId], roots: &[NodeId], ) -> Result { @@ -184,9 +178,10 @@ pub fn compile( ( edge.consumer.0, match edge.role { - planner_types::post_asap::EdgeRole::Left => 0, - planner_types::post_asap::EdgeRole::Input => 1, - planner_types::post_asap::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::Left => 0, + planner_types::ir::export::EdgeRole::Input => 1, + planner_types::ir::export::EdgeRole::Right => 2, + planner_types::ir::export::EdgeRole::ScalarRef => 3, }, ) }); @@ -261,11 +256,11 @@ pub fn compile( CompiledPhysicalDAG::compose(sources, fragments, roots.to_vec()) } -fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { +fn validate_value_output(node: &PhysicalASAPDAGNode) -> Result<(), Error> { let schema = &node.output_schema; // Physical population rows already carry the complete identity in `$population`. // Typed logical plans may expose its opaque series-identity column as metadata. - let identity = planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY; + let identity = planner_types::ir::schema::PROMQL_SERIES_IDENTITY; let identities = schema .fields .iter() @@ -274,7 +269,7 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { if identities.len() > 1 || identities .iter() - .any(|field| field.nullable || field.dtype != FieldDataType::Plain(DataType::Utf8)) + .any(|field| field.nullable || field.dtype != SummaryFamilyType::Plain(DataType::Utf8)) { return Err(invalid( "precompute series identity requires one non-null Utf8 column", @@ -286,12 +281,11 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { .enumerate() .filter(|(i, field)| Some(*i) != schema.time_index && field.name != identity) .collect::>(); - if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == FieldDataType::Plain(DataType::Float64)) + if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == SummaryFamilyType::Plain(DataType::Float64)) || schema.time_index.is_some_and(|i| { - schema - .fields - .get(i) - .is_none_or(|f| f.nullable || f.dtype != FieldDataType::Plain(DataType::Timestamp)) + schema.fields.get(i).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) }) { return Err(invalid( @@ -302,9 +296,9 @@ fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { } fn fragment( - node: &PostAsapDAGNode, + node: &PhysicalASAPDAGNode, schemas: &[SchemaRef], - parents: &[&PostAsapDAGNode], + parents: &[&PhysicalASAPDAGNode], ) -> Result { let sources = schemas .iter() @@ -320,7 +314,15 @@ fn fragment( Ok(id) }; let root = match &node.payload { - Payload::Binary { operator } => { + Payload::Relational { + operator: + NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } => { + let operator = + crate::expressions::binary::BinaryOperator::from_logical(operator, *return_bool); validate_value_output(node)?; if node.output_schema.time_index.is_none() || parents.iter().any(|p| p.output_schema.time_index.is_none()) @@ -343,30 +345,29 @@ fn fragment( )?, )? } - Payload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - } => { + Payload::FinalizeExactAccumulator => { let [input] = schemas else { return Err(invalid("finalize requires one state input")); }; validate_value_output(node)?; let statistic = match &input.fields[2].dtype { - FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { + SummaryFamilyType::ExactAggregate(planner_types::ir::schema::ExactKind::Sum, _) => { crate::Statistic::Sum } - FieldDataType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) => { - crate::Statistic::Count - } + SummaryFamilyType::ExactAggregate( + planner_types::ir::schema::ExactKind::Count, + _, + ) => crate::Statistic::Count, _ => { return Err(invalid( "precompute finalization requires explicit Sum or Count semantics", )) } }; - let read = Operator::readout( + let read = Operator::evaluation( input.clone(), 2, - ReadoutQuery::Exact(ExactReadout { + SummaryEvaluation::Exact(ExactEvaluation { statistic, lookback_ms: None, }), @@ -384,7 +385,9 @@ fn fragment( ), ], )? - .with_output_schema(population_schema(FieldDataType::Plain(DataType::Float64)))?; + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; add(vec![read], project)? } Payload::SummaryAgg { @@ -403,17 +406,17 @@ fn fragment( return Err(invalid("summary update requires one input")); }; // Item identities resolve against the complete label set of raw - // samples; finalized readouts carry no such identity. + // samples; finalized evaluations carry no such identity. let raw = *input == raw_sample_schema(); // A unit-frequency summary (HLL) observes each raw sample value. let unit_frequency = raw && crate::capability::is_unit_sample_frequency(update) - && matches!(family, FieldDataType::Sketch(kind, _) if !matches!( + && matches!(family, SummaryFamilyType::Sketch(kind, _) if !matches!( kind.algorithm(), - planner_types::post_asap::SketchAlgorithm::Cms - | planner_types::post_asap::SketchAlgorithm::CountSketch - | planner_types::post_asap::SketchAlgorithm::CmsWithHeap - | planner_types::post_asap::SketchAlgorithm::CountSketchWithHeap + planner_types::ir::schema::SketchAlgorithm::Cms + | planner_types::ir::schema::SketchAlgorithm::CountSketch + | planner_types::ir::schema::SketchAlgorithm::CmsWithHeap + | planner_types::ir::schema::SketchAlgorithm::CountSketchWithHeap )); let keyed = update.item.is_some() && !unit_frequency; if (keyed && !raw) || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) { @@ -426,8 +429,8 @@ fn fragment( if raw && matches!( update.weight_domain, - planner_types::post_asap::WeightDomain::NonNegative { - proof: planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + planner_types::ir::schema::WeightDomain::NonNegative { + proof: planner_types::ir::schema::NonNegativeWeightProof::ResetAwareCounterDerivative } ) { @@ -436,10 +439,10 @@ fn fragment( )); } if keyed - && matches!(family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::ir::schema::SketchAlgorithm::CmsWithHeap) && !matches!( update.weight_domain, - planner_types::post_asap::WeightDomain::NonNegative { .. } + planner_types::ir::schema::WeightDomain::NonNegative { .. } ) { return Err(invalid("CMS requires a nonnegative weight contract")); @@ -461,7 +464,7 @@ fn fragment( .filter(|field| { (raw || !field.nullable) && field.name != SERIES_IDENTITY - && field.dtype == FieldDataType::Plain(DataType::Utf8) + && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) }) .map(|f| f.name.clone()) .ok_or_else(|| { @@ -481,7 +484,7 @@ fn fragment( SummaryInputExpr::Column(ColumnRef::SampleValue) => Expression::Column(2), SummaryInputExpr::Column(ColumnRef::Named(name)) if parents[0].output_schema.fields.iter().any(|f| { - f.name == *name && f.dtype == FieldDataType::Plain(DataType::Float64) + f.name == *name && f.dtype == SummaryFamilyType::Plain(DataType::Float64) }) => { Expression::Column(2) @@ -497,7 +500,7 @@ fn fragment( ("$window_end".into(), Expression::Column(1)), ("value".into(), Expression::FiniteFloat64(Box::new(weight))), ]; - let mut fields = population_schema(FieldDataType::Plain(DataType::Float64)) + let mut fields = population_schema(SummaryFamilyType::Plain(DataType::Float64)) .fields .clone(); if keyed { @@ -509,11 +512,11 @@ fn fragment( )?; for (index, (expression, dtype)) in items.into_iter().enumerate() { let name = format!("$item{index}"); - fields.push(planner_types::post_asap::Field { - table: None, + fields.push(planner_types::ir::schema::Field { name: name.clone(), - dtype: FieldDataType::Plain(dtype), + dtype: SummaryFamilyType::Plain(dtype), nullable: false, + table: None, }); columns.push((name, expression)); } @@ -521,9 +524,9 @@ fn fragment( let item_columns = (3..fields.len()).collect::>(); let project = Operator::project(input.clone(), columns)?.with_output_schema( Arc::new(Schema { - closed: true, - unique_keys: vec![], fields, + unique_keys: vec![], + closed: false, time_index: Some(1), }), )?; @@ -583,7 +586,7 @@ fn raw_items( ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } if !name.starts_with('$') && scan.fields.iter().all(|f| { - &f.name != name || f.dtype == FieldDataType::Plain(DataType::Utf8) + &f.name != name || f.dtype == SummaryFamilyType::Plain(DataType::Utf8) }) => { Some(name.clone()) @@ -621,7 +624,7 @@ fn raw_items( DataType::Utf8, )), SummaryInputExpr::EntityIdentity( - planner_types::post_asap::EntityIdentity::PromqlLabelSet { excluding }, + planner_types::ir::schema::EntityIdentity::PromqlLabelSet { excluding }, ) => items.push(identity( excluding .iter() diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/executor/src/physical_planner/promql_fallback.rs similarity index 50% rename from crates/asap-physical-operators/src/physical_planner/promql_fallback.rs rename to crates/executor/src/physical_planner/promql_fallback.rs index 237504c98..354712a93 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/executor/src/physical_planner/promql_fallback.rs @@ -3,8 +3,8 @@ //! computes selection, range functions, subqueries, matching and aggregation. use super::*; use crate::operators::SubquerySteps; -use planner_types::post_asap::execution_data_state::lift_plain; -use planner_types::pre_asap::{AtModifier, VectorMatchKind}; +use planner_types::ir::operator::{AtModifier, VectorMatchKind}; +use planner_types::physical::execution_data_state::lift_plain; /// Input slot for the raw series read by the `selector`th selector (in /// [`raw_series`] order) of Fallback node `node`. The node's own ID names its @@ -19,14 +19,14 @@ pub(super) fn raw_series_owner(slot: NodeId) -> Option { } /// A selector expression and its raw-series row schema. -pub type Selector = (QueryExpr, SchemaRef); +pub type Selector = (OperatorNode, SchemaRef); /// The selectors a Fallback expression reads, left to right, and the row /// schema of the raw series the deployment supplies for each at /// [`raw_series_input`]. The rows must cover the selector's window at every /// evaluation instant `T`, or at its `@` time: `(T - offset - range, T - offset]`; /// under a subquery `[R:S] offset O` that is `(T - O - R - offset - range, T - O - offset]`. -pub fn raw_series(expression: &QueryExpr) -> Result, Error> { +pub fn raw_series(expression: &OperatorNode) -> Result, Error> { Ok(lower(expression)?.selectors) } @@ -43,16 +43,47 @@ pub(super) struct Lowering { pub steps: Vec<(Operator, Vec)>, } -pub(super) fn lower(expression: &QueryExpr) -> Result { +pub(super) fn lower(expression: &OperatorNode) -> Result { let mut lowering = Lowering::default(); lowering.value(expression)?; Ok(lowering) } -fn declared(expression: &QueryExpr) -> Result { - let schema = expression - .output_schema() - .map_err(|error| invalid(error.to_string()))?; +/// Compile a standalone scalar expression and expose its real series dependencies. +/// Input slots use root 0; no logical wrapper node is introduced. +pub fn compile_scalar_root( + expr: &ScalarExpr, +) -> Result<(CompiledPhysicalDAG, Vec), Error> { + let mut lowering = Lowering::default(); + lowering.scalar_value(expr)?; + let mut inputs = BTreeMap::new(); + for (i, (_, schema)) in lowering.selectors.iter().enumerate() { + inputs.insert( + raw_series_input(0, i), + InputContract::bounded(schema.clone()), + ); + } + let last = lowering.steps.len() - 1; + let mut operators = BTreeMap::new(); + for (i, (operator, dependencies)) in lowering.steps.into_iter().enumerate() { + let id = if i == last { 0 } else { i as u64 + 1 }; + let dependencies = dependencies + .into_iter() + .map(|input| match input { + Input::Raw(i) => raw_series_input(0, i), + Input::Step(i) => i as u64 + 1, + }) + .collect(); + operators.insert(id, (dependencies, operator)); + } + Ok(( + CompiledPhysicalDAG::from_operators(inputs, operators, vec![0])?, + lowering.selectors, + )) +} + +fn declared(expression: &OperatorNode) -> Result { + let schema = expression.schema.clone(); Ok(Arc::new(lift_plain(&schema))) } @@ -61,7 +92,7 @@ fn millis(duration: &std::time::Duration) -> Result { } /// A fixed `@` time. `start()`/`end()` depend on the deployment's range query. -fn at(shift: &planner_types::pre_asap::TimeShift) -> Result, Error> { +fn at(shift: &planner_types::ir::operator::TimeShift) -> Result, Error> { match shift.at { None => Ok(None), Some(AtModifier::Timestamp(at)) => Ok(Some(at)), @@ -69,10 +100,10 @@ fn at(shift: &planner_types::pre_asap::TimeShift) -> Result, Error> } } -fn range_anchor(expression: &QueryExpr) -> Option { - match expression { - QueryExpr::TimeRange { child, .. } => range_anchor(child), - QueryExpr::TimeShift { shift, .. } => shift +fn range_anchor(expression: &OperatorNode) -> Option { + match expression.expect_non_asap() { + NonASAPOp::TimeRange { child, .. } => range_anchor(child), + NonASAPOp::TimeShift { shift, .. } => shift .at .filter(|at| matches!(at, AtModifier::Start | AtModifier::End)), _ => None, @@ -80,32 +111,22 @@ fn range_anchor(expression: &QueryExpr) -> Option { } /// `TimeRange { range, [TimeShift { offset, @ }], Scan }`: range, offset, `@`. -fn selector(expression: &QueryExpr) -> Result<(i64, i64, Option), Error> { - let QueryExpr::TimeRange { range, child } = expression else { +fn selector(expression: &OperatorNode) -> Result<(i64, i64, Option), Error> { + let NonASAPOp::TimeRange { range, child, .. } = expression.expect_non_asap() else { return Err(invalid("PromQL operand must be a series selector")); }; - let (offset, at, scan) = match child.as_ref() { - QueryExpr::TimeShift { shift, child } => (shift.offset_ms, at(shift)?, child.as_ref()), + let (offset, at, scan) = match child.expect_non_asap() { + NonASAPOp::TimeShift { shift, child } => { + (shift.offset_ms, at(shift)?, child.expect_non_asap()) + } scan => (0, None, scan), }; - if !matches!(scan, QueryExpr::Scan { .. }) { + if !matches!(scan, NonASAPOp::Scan { .. }) { return Err(invalid("PromQL selector must read one scan")); } Ok((millis(range)?, offset, at)) } -/// PromQL scalar-valued expressions have no labels to match. A binary -/// operator is scalar-valued when both operands are. -pub(super) fn scalar(expression: &QueryExpr) -> bool { - match expression { - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp => true, - QueryExpr::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), - _ => false, - } -} - impl Lowering { fn schema(&self, input: &Input) -> SchemaRef { match input { @@ -124,12 +145,12 @@ impl Lowering { &mut self, operator: Operator, inputs: Vec, - logical: &QueryExpr, + logical: &OperatorNode, ) -> Result { Ok(self.add(operator.with_output_schema(declared(logical)?)?, inputs)) } - fn read(&mut self, selector: &QueryExpr) -> Result { + fn read(&mut self, selector: &OperatorNode) -> Result { let schema = declared(selector)?; if !schema .fields @@ -145,12 +166,12 @@ impl Lowering { } /// An instant vector, or a scalar for scalar-valued expressions. - fn value(&mut self, expression: &QueryExpr) -> Result { - match expression { - QueryExpr::Concat { children, .. } => { - if !children.iter().all(|branch| matches!(branch, - QueryExpr::PromqlRelabel { child, .. } if matches!(child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { + fn value(&mut self, expression: &OperatorNode) -> Result { + match expression.expect_non_asap() { + NonASAPOp::Concat { children, .. } => { + if !children.iter().all(|branch| matches!(branch.expect_non_asap(), + NonASAPOp::PromqlRelabel { child, .. } if matches!(child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::HistogramQuantile { .. }])))) { return Err(invalid("PromQL concatenation requires classic histogram quantile branches")); } let inputs = children @@ -171,15 +192,17 @@ impl Lowering { expression, ) } - QueryExpr::PromqlRelabel { dst, value, child } => { + NonASAPOp::PromqlRelabel { dst, value, child } => { let step = self.value(child)?; let input = self.schema(&step); - let (replacement, source_regex) = match value.as_ref() { - QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(value)) => { + let (replacement, source_regex) = match value { + ScalarExpr::Literal(planner_types::ir::scalar::ScalarValue::Utf8(value)) => { (value.clone(), None) } - QueryExpr::FunctionCall { name, args } if name == "label_replace" => { - let [QueryExpr::Column(source), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8(pattern)), QueryExpr::Literal(planner_types::pre_asap::ScalarValue::Utf8( + ScalarExpr::FunctionCall { name, args } if name == "label_replace" => { + let [ScalarExpr::Column(source), ScalarExpr::Literal(planner_types::ir::scalar::ScalarValue::Utf8( + pattern, + )), ScalarExpr::Literal(planner_types::ir::scalar::ScalarValue::Utf8( replacement, ))] = args.as_slice() else { @@ -204,7 +227,7 @@ impl Lowering { )?; self.push(operator, vec![step], expression) } - QueryExpr::TimeRange { .. } => { + NonASAPOp::TimeRange { .. } => { let (range, offset, at) = selector(expression)?; let input = self.read(expression)?; let schema = self.schema(&input); @@ -215,8 +238,8 @@ impl Lowering { expression, ) } - QueryExpr::Aggregate { - reduction: planner_types::pre_asap::Reduction::PerEntity, + NonASAPOp::Aggregate { + reduction: planner_types::ir::operator::Reduction::PerEntity, measures, having: None, child, @@ -234,8 +257,8 @@ impl Lowering { let input = self.schema(&step); Ok(self.add(Operator::series_without_name(input)?, vec![step])) } - QueryExpr::Aggregate { - reduction: planner_types::pre_asap::Reduction::Reduce(keys), + NonASAPOp::Aggregate { + reduction: planner_types::ir::operator::Reduction::Reduce(keys), measures, having: None, child, @@ -248,7 +271,74 @@ impl Lowering { let input = self.value(child)?; self.aggregate(input, measure, keys, expression) } - QueryExpr::Sort { + NonASAPOp::Project { + cols, + child, + qualifier, + } => { + let value = planner_types::ir::scalar::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + ) + .map_err(|e| invalid(e.to_string()))?; + let sample = cols + .iter() + .find(|col| { + col.alias.as_deref() == Some(child.schema.fields[value].name.as_str()) + }) + .ok_or_else(|| invalid("missing sample projection"))?; + let keep_name = matches!(sample.expr, ScalarExpr::Negative { .. }); + let fields: Vec<_> = child + .schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| keep_name || field.name != "__name__") + .collect(); + if qualifier.is_some() || cols.len() != fields.len() { + return Err(invalid("unsupported temporal projection shape")); + } + let mut computed = None; + for (col, (index, field)) in cols.iter().zip(fields) { + if col.alias.as_deref() != Some(field.name.as_str()) { + return Err(invalid("unsupported temporal projection alias")); + } + if index == value { + computed = Some(col); + } else { + let expected = if !keep_name + && field.name == planner_types::ir::schema::PROMQL_SERIES_IDENTITY + { + ScalarExpr::FunctionCall { + name: "promql_drop_metric_name".into(), + args: vec![ScalarExpr::Column(index)], + } + } else { + ScalarExpr::Column(index) + }; + if col.expr != expected { + return Err(invalid("unsupported temporal projection expression")); + } + } + } + let computed = computed.ok_or_else(|| invalid("no computed sample"))?; + if matches!( + computed.expr, + ScalarExpr::Negative { .. } | ScalarExpr::FunctionCall { .. } + ) { + return self.pointwise_projection(cols, child, value, expression, keep_name); + } + self.sample_scalar_operation(&computed.expr, child, value, expression) + } + NonASAPOp::Filter { pred, child } => { + let value = planner_types::ir::scalar::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + ) + .map_err(|e| invalid(e.to_string()))?; + self.sample_scalar_operation(&pred.0, child, value, expression) + } + NonASAPOp::Sort { keys, partition_by, child, @@ -258,7 +348,7 @@ impl Lowering { let keys = keys .iter() .map(|key| match key.expr { - QueryExpr::Column(column) => Ok(SortKey { + ScalarExpr::Column(column) => Ok(SortKey { column, descending: !key.ascending, nulls_first: key.nulls_first, @@ -269,70 +359,224 @@ impl Lowering { let groups = groups(&input, partition_by)?; self.push(Operator::sort(input, keys, groups)?, vec![step], expression) } - QueryExpr::Limit { n, offset, child } => { + NonASAPOp::Limit { + n, offset, child, .. + } => { let step = self.value(child)?; let input = self.schema(&step); // `topk by (...)` partitions through the Sort it limits. - let groups = match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => groups(&input, partition_by)?, + let groups = match child.expect_non_asap() { + NonASAPOp::Sort { partition_by, .. } => groups(&input, partition_by)?, _ => vec![], }; self.push( - Operator::limit(input, *n as u64, *offset as u64, groups)?, + Operator::limit( + input, + n.unwrap_or(usize::MAX) as u64, + *offset as u64, + groups, + )?, vec![step], expression, ) } - QueryExpr::BinaryOp { - op, + NonASAPOp::BinaryOp { + operator, lhs, rhs, - vector_match, + return_bool, } => { let sides = vec![self.value(lhs)?, self.value(rhs)?]; - let operator = planner_types::post_asap::BinaryOperator { - kind: op.clone(), - vector_match: vector_match.clone(), - checked_relative_division: false, - checked_finite_division: false, - }; + let operator = crate::expressions::binary::BinaryOperator::from_logical( + operator, + *return_bool, + ); let binary = Operator::series_binary( self.schema(&sides[0]), self.schema(&sides[1]), operator, - [scalar(lhs), scalar(rhs)], + [false, false], )?; self.push(binary, sides, expression) } - QueryExpr::PromqlScalarFromVector(child) => { - let step = self.value(child)?; - let input = self.schema(&step); - let value = named_column(&input, &ColumnRef::SampleValue)?; - self.push( - Operator::vector_to_scalar(input, value)?, - vec![step], - expression, - ) - } - QueryExpr::PromqlVectorFromScalar(child) => { - let step = self.value(child)?; + NonASAPOp::PromqlVectorFromScalar(expr) => { + let step = self.scalar_value(expr)?; let input = self.schema(&step); Ok(self.add( Operator::scope_timestamp(input, declared(expression)?)?, vec![step], )) } - QueryExpr::EvalTimestamp => self.push(Operator::evaluation_time(), vec![], expression), - QueryExpr::PromqlScalarBridge(_) => { - let value = row_values::scalar_literal(expression) - .ok_or_else(|| invalid("PromQL scalar must be a literal"))?; - self.push( - Operator::scalar(crate::values::Value::Float64(value), DataType::Float64)?, + _ => Err(invalid("PromQL expression has no native fallback lowering")), + } + } + + fn pointwise_projection( + &mut self, + cols: &[planner_types::ir::ProjectItem], + child: &OperatorNode, + value: usize, + output: &OperatorNode, + keep_name: bool, + ) -> Result { + let mut input = self.value(child)?; + let mut projected = cols.to_vec(); + for col in &mut projected { + if col.alias.as_deref() != Some(child.schema.fields[value].name.as_str()) { + continue; + } + if let ScalarExpr::FunctionCall { name, args } = &mut col.expr { + if planner_types::ir::scalar::scalar_type_rules::promql_function_arity(name) + .is_none() + || args.first() != Some(&ScalarExpr::Column(value)) + { + return Err(invalid("unsupported pointwise function")); + } + for arg in args.iter_mut().skip(1) { + let scalar = self.scalar_value(arg)?; + let left = self.schema(&input); + let right = self.schema(&scalar); + let index = left.fields.len(); + let mut schema = (*left).clone(); + schema.fields.extend(right.fields.clone()); + let join = Operator::relational_join( + left, + right, + planner_types::ir::operator::JoinKind::Inner, + &planner_types::ir::Predicate(ScalarExpr::Literal( + planner_types::ir::scalar::ScalarValue::Boolean(true), + )), + Arc::new(schema), + )?; + input = self.add(join, vec![input, scalar]); + *arg = ScalarExpr::Column(index); + } + if name == "promql_clamp" { + let predicate = ScalarExpr::Not(Box::new(ScalarExpr::Compare { + left: Box::new(args[1].clone()), + right: Box::new(args[2].clone()), + op: planner_types::ir::scalar::CompareOpKind::Gt, + semantics: planner_types::ir::ExprSemantics::Promql, + })); + let schema = self.schema(&input); + let predicate = + crate::expressions::CompiledExpression::compile(&predicate, &schema)?; + input = self.add( + Operator::filter( + schema, + crate::expressions::Expression::planner(predicate), + )?, + vec![input], + ); + } + } + } + let schema = self.schema(&input); + let columns = projected + .iter() + .map(|col| { + Ok(( + col.alias.clone().unwrap(), + crate::expressions::Expression::planner( + crate::expressions::CompiledExpression::compile(&col.expr, &schema)?, + ), + )) + }) + .collect::, Error>>()?; + let project = Operator::project(schema, columns)?; + let result = self.push(project, vec![input], output)?; + if keep_name { + Ok(result) + } else { + self.push( + Operator::series_without_name(self.schema(&result))?, + vec![result], + output, + ) + } + } + + fn sample_scalar_operation( + &mut self, + expr: &ScalarExpr, + child: &OperatorNode, + value: usize, + output: &OperatorNode, + ) -> Result { + let (left, right, kind) = scalar_binary(expr)?; + let (scalar, scalar_left) = match (left, right) { + (ScalarExpr::Column(i), scalar) if *i == value => (scalar, false), + (scalar, ScalarExpr::Column(i)) if *i == value => (scalar, true), + _ => { + return Err(invalid( + "sample projection requires one vector sample and one scalar", + )) + } + }; + let vector = self.value(child)?; + let scalar = self.scalar_value(scalar)?; + let sides = if scalar_left { + vec![scalar, vector] + } else { + vec![vector, scalar] + }; + let operator = Operator::series_binary( + self.schema(&sides[0]), + self.schema(&sides[1]), + kernel(kind), + [scalar_left, !scalar_left], + )?; + self.push(operator, sides, output) + } + + fn scalar_value(&mut self, expr: &ScalarExpr) -> Result { + match expr { + ScalarExpr::Literal(planner_types::ir::scalar::ScalarValue::Float64(value)) => Ok(self + .add( + Operator::scalar(crate::values::Value::Float64(*value), DataType::Float64)?, vec![], - expression, - ) + )), + ScalarExpr::EvalTimestamp => Ok(self.add(Operator::evaluation_time(), vec![])), + ScalarExpr::PromqlScalarFromVector(child) => { + let step = self.value(child)?; + let input = self.schema(&step); + let values: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| f.dtype == FieldDataType::Plain(DataType::Float64)) + .map(|(i, _)| i) + .collect(); + let [value] = values.as_slice() else { + return Err(invalid("scalar() requires one float sample column")); + }; + let value = *value; + Ok(self.add(Operator::vector_to_scalar(input, value)?, vec![step])) + } + ScalarExpr::Negative { expr, .. } => { + let value = self.scalar_value(expr)?; + let minus = self.scalar_value(&ScalarExpr::literal_f64(-1.0))?; + let op = Operator::series_binary( + self.schema(&value), + self.schema(&minus), + kernel(crate::expressions::binary::BinaryOpKind::Arithmetic( + planner_types::ir::scalar::ArithmeticOpKind::Mul, + )), + [true, true], + )?; + Ok(self.add(op, vec![value, minus])) + } + _ => { + let (left, right, kind) = scalar_binary(expr)?; + let sides = vec![self.scalar_value(left)?, self.scalar_value(right)?]; + let op = Operator::series_binary( + self.schema(&sides[0]), + self.schema(&sides[1]), + kernel(kind), + [true, true], + )?; + Ok(self.add(op, sides)) } - _ => Err(invalid("PromQL expression has no native fallback lowering")), } } @@ -340,19 +584,19 @@ impl Lowering { fn range_function( &mut self, function: &AggIntent, - matrix: &QueryExpr, - logical: &QueryExpr, + matrix: &OperatorNode, + logical: &OperatorNode, ) -> Result { let function = unbound(function)?; - let (subquery, offset, at_ms) = match matrix { - QueryExpr::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), - other => (other, 0, None), + let (subquery, offset, at_ms) = match matrix.expect_non_asap() { + NonASAPOp::TimeShift { shift, child } => (child.as_ref(), shift.offset_ms, at(shift)?), + _ => (matrix, 0, None), }; - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range: outer, resolution, child, - } = subquery + } = subquery.expect_non_asap() else { let (range, offset, at) = selector(matrix)?; let input = self.read(matrix)?; @@ -374,9 +618,9 @@ impl Lowering { at_ms, }; // Each step evaluates a per-series selection or range function. - let (inner, selected) = match child.as_ref() { - QueryExpr::Aggregate { - reduction: planner_types::pre_asap::Reduction::PerEntity, + let (inner, selected) = match child.expect_non_asap() { + NonASAPOp::Aggregate { + reduction: planner_types::ir::operator::Reduction::PerEntity, measures, having: None, child: selected, @@ -385,7 +629,7 @@ impl Lowering { [inner] => (Some(unbound(inner)?), selected.as_ref()), _ => return Err(invalid("range function requires one measure")), }, - selected => (None, selected), + _ => (None, child.as_ref()), }; let (range, inner_offset, inner_at) = selector(selected)?; let raw = self.read(selected)?; @@ -425,7 +669,7 @@ impl Lowering { mut step: Input, measure: &AggIntent, keys: &GroupKeys, - logical: &QueryExpr, + logical: &OperatorNode, ) -> Result { let mut input = self.schema(&step); if let AggIntent::HistogramQuantile { q, le } = measure { @@ -447,10 +691,10 @@ impl Lowering { }; let value = *value; let reduction = match measure { - AggIntent::Sum { col: None } => Reduction::Sum(value), - AggIntent::Avg { col: None } => Reduction::Avg(value), - AggIntent::Min { col: None } => Reduction::Min(value), - AggIntent::Max { col: None } => Reduction::Max(value), + AggIntent::Sum { .. } => Reduction::Sum(value), + AggIntent::Avg { .. } => Reduction::Avg(value), + AggIntent::Min { .. } => Reduction::Min(value), + AggIntent::Max { .. } => Reduction::Max(value), AggIntent::Count { .. } => Reduction::Count, _ => return Err(invalid("vector aggregate has no native lowering")), }; @@ -527,10 +771,10 @@ fn unbound(intent: &AggIntent) -> Result, Error> { AggIntent::Count { accuracy } => AggIntent::Count { accuracy: accuracy.clone(), }, - AggIntent::Sum { col: None } => AggIntent::Sum { col: None }, - AggIntent::Avg { col: None } => AggIntent::Avg { col: None }, - AggIntent::Min { col: None } => AggIntent::Min { col: None }, - AggIntent::Max { col: None } => AggIntent::Max { col: None }, + AggIntent::Sum { .. } => AggIntent::Sum { col: None }, + AggIntent::Avg { .. } => AggIntent::Avg { col: None }, + AggIntent::Min { .. } => AggIntent::Min { col: None }, + AggIntent::Max { .. } => AggIntent::Max { col: None }, AggIntent::IRate => AggIntent::IRate, AggIntent::IDelta => AggIntent::IDelta, AggIntent::Changes => AggIntent::Changes, @@ -548,3 +792,65 @@ fn unbound(intent: &AggIntent) -> Result, Error> { _ => return Err(invalid("unsupported PromQL range function")), }) } + +fn kernel( + kind: crate::expressions::binary::BinaryOpKind, +) -> crate::expressions::binary::BinaryOperator { + crate::expressions::binary::BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + } +} + +fn scalar_binary( + expr: &ScalarExpr, +) -> Result< + ( + &ScalarExpr, + &ScalarExpr, + crate::expressions::binary::BinaryOpKind, + ), + Error, +> { + use crate::expressions::binary::BinaryOpKind as K; + match expr { + ScalarExpr::Arithmetic { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + } => Ok((left, right, K::Arithmetic(op.clone()))), + ScalarExpr::Compare { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + } => Ok((left, right, K::Compare(op.clone()))), + ScalarExpr::Case { + operand: None, + branches, + else_expr, + } if matches!(else_expr.as_deref(), Some(ScalarExpr::Literal(planner_types::ir::scalar::ScalarValue::Float64(v))) if *v == 0.0) => + { + let [( + ScalarExpr::Compare { + left, + right, + op, + semantics: planner_types::ir::ExprSemantics::Promql, + }, + ScalarExpr::Literal(planner_types::ir::scalar::ScalarValue::Float64(v)), + )] = branches.as_slice() + else { + return Err(invalid("unsupported scalar case")); + }; + if *v != 1.0 { + return Err(invalid("unsupported scalar case result")); + } + Ok((left, right, K::CompareBool(op.clone()))) + } + _ => Err(invalid("scalar expression has no native temporal lowering")), + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/executor/src/physical_planner/promql_rows.rs similarity index 60% rename from crates/asap-physical-operators/src/physical_planner/promql_rows.rs rename to crates/executor/src/physical_planner/promql_rows.rs index f3082927b..a2b24a4bb 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/executor/src/physical_planner/promql_rows.rs @@ -1,11 +1,15 @@ //! A bounded PromQL source row carries the entire label set, not just labels //! mentioned by the query. The source adapter owns this lossless encoding. use super::*; -use planner_types::pre_asap::DataType; +use planner_types::ir::export::{ + compile_physical_asap_dag, compile_physical_asap_dag_with_node_ids, +}; +use planner_types::ir::schema::DataType; +use planner_types::ir::schema::FieldDataType as SummaryFamilyType; use std::rc::Rc; /// Not a legal PromQL label name, so it cannot shadow a user label. -pub use planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY as SERIES_IDENTITY_COLUMN; +pub use planner_types::ir::schema::PROMQL_SERIES_IDENTITY as SERIES_IDENTITY_COLUMN; /// Canonical, reversible identity. JSON object encoding preserves label names, /// empty values and escaping; sorting makes ingestion order irrelevant. @@ -23,9 +27,9 @@ pub fn decode_series_identity(encoded: &str) -> Result, } /// Resolve the row representation before candidate search; see -/// [`planner_types::pre_asap::schema::with_promql_series_identity`]. -pub fn with_series_identity(root: &QueryExpr) -> Result { - planner_types::pre_asap::schema::with_promql_series_identity(root).map_err(invalid) +/// [`planner_types::ir::schema_support::with_promql_series_identity`]. +pub fn with_series_identity(root: &Rc) -> Result, Error> { + planner_types::ir::schema_support::with_promql_series_identity(root).map_err(invalid) } /// Construct source rows only from full identities. The named label columns @@ -45,7 +49,10 @@ pub fn series_row( .enumerate() .map(|(index, field)| { if field.name == SERIES_IDENTITY_COLUMN { - if field.dtype != FieldDataType::Plain(DataType::Utf8) || field.nullable || found { + if field.dtype != SummaryFamilyType::Plain(DataType::Utf8) + || field.nullable + || found + { return Err(invalid("invalid series identity column")); } found = true; @@ -53,10 +60,10 @@ pub fn series_row( } else if Some(index) == schema.time_index { Ok(Value::Timestamp(timestamp)) } else if field.name == "value" - && field.dtype == FieldDataType::Plain(DataType::Float64) + && field.dtype == SummaryFamilyType::Plain(DataType::Float64) { Ok(Value::Float64(value)) - } else if field.dtype == FieldDataType::Plain(DataType::Utf8) { + } else if field.dtype == SummaryFamilyType::Plain(DataType::Utf8) { Ok(labels.get(&field.name).map_or_else( || Value::Utf8("".into()), |value| Value::Utf8(value.clone().into()), @@ -75,18 +82,25 @@ pub fn series_row( /// Compile the selected TopK computation above an existing maintained-population /// source. The boundary supplies the complete eligible vector, not a truncated /// TopK result; ranking remains a native physical operator. -pub fn compile_current_series_readout( - selected: &Rc, +pub fn compile_current_series_evaluation( + selected: &Rc, ) -> Result { - use planner_types::post_asap::{ - compile_post_asap_dag, maintained_population::PopulationStatistic, Field, - }; - let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; + use planner_types::ir::operator::maintained_population::PopulationStatistic; + use planner_types::ir::schema::Field as SummaryField; + // This compiler emits maintained precompute: every summary at ingestion time. + let selected = planner_types::ir::apply_materialization_timings( + selected, + &planner_types::ir::MaterializationAssignment::all_ingestion_time(), + &mut planner_types::ir::TimingMemo::new(), + ) + .map_err(|e| invalid(e.to_string()))?; + let mut dag = + compile_physical_asap_dag(&selected).map_err(|error| invalid(error.to_string()))?; // Typed snapshot candidates already carry full identity throughout the DAG. - // Cut at the population output, preserving all selected heap/readout nodes. + // Cut at the population output, preserving all selected heap/evaluation nodes. let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, - Payload::Value { operation: ValueOperation::MaintainPopulation { population } } - if matches!(population.input, planner_types::post_asap::maintained_population::PopulationInput::CurrentSeries(_)) + Payload::MaintainPopulation { population } + if matches!(population.input, planner_types::ir::operator::maintained_population::PopulationInput::CurrentSeries(_)) )).collect::>(); if let [population] = populations.as_slice() { if population @@ -101,45 +115,30 @@ pub fn compile_current_series_readout( u64::from(population.id.0), InputContract::bounded(Arc::new(population.output_schema.clone())), )]), - &[u64::from(dag.root.0)], + &dag.roots.iter().map(|r| u64::from(r.0)).collect::>(), ); } } - if dag.nodes.len() != 3 - || !dag.nodes.iter().any(|node| { - node.id == dag.root - && matches!( - node.payload, - Payload::Value { - operation: ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { .. } - } - } - ) - }) - { - return Err(invalid( - "expected one selected current-series TopK computation", - )); - } let mut frontier = None; for node in &mut dag.nodes { match &mut node.payload { - Payload::Fallback { expression } => { - *expression = with_series_identity(expression)?; + Payload::Relational { operator } => { + if let NonASAPOpKind::Scan { schema, .. } = operator { + schema.fields.push(SummaryField::new( + SERIES_IDENTITY_COLUMN, + SummaryFamilyType::Plain(DataType::Utf8), + false, + )); + schema.closed = true; + } } - Payload::Value { - operation: ValueOperation::MaintainPopulation { .. }, - } => { + Payload::MaintainPopulation { .. } => { frontier = Some(u64::from(node.id.0)); } - Payload::Value { - operation: - ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { .. }, - }, + Payload::EvaluatePopulation { + evaluation: PopulationStatistic::TopK { .. }, } => {} - _ => return Err(invalid("unsupported current-series readout dependency")), + _ => return Err(invalid("unsupported current-series evaluation dependency")), } if node .output_schema @@ -151,11 +150,11 @@ pub fn compile_current_series_readout( "current-series input already has a physical identity column", )); } - node.output_schema.fields.push(Field { - table: None, + node.output_schema.fields.push(SummaryField { name: SERIES_IDENTITY_COLUMN.into(), - dtype: FieldDataType::Plain(DataType::Utf8), + dtype: SummaryFamilyType::Plain(DataType::Utf8), nullable: false, + table: None, }); } for edge in &mut dag.edges { @@ -179,48 +178,37 @@ pub fn compile_current_series_readout( compile( &dag, BTreeMap::from([(frontier, InputContract::bounded(schema))]), - &[u64::from(dag.root.0)], + &dag.roots.iter().map(|r| u64::from(r.0)).collect::>(), ) } /// Compile selected ranking or aggregation above an exact per-series Rate -/// readout. Deployments bind complete window readouts at this boundary; +/// evaluation. Deployments bind complete window evaluations at this boundary; /// the heap is rebuilt independently for each evaluation. This does not move /// that frontier to ingestion time or authorize combining finalized rates. pub fn compile_rate_ranking( - selected: &Rc, -) -> Result< - ( - Rc, - CompiledPhysicalDAG, - ), - Error, -> { - use planner_types::post_asap::{ - compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, - }; - fn frontier(node: &Rc) -> Option> { - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: planner_types::post_asap::ExecutionTiming::QueryTime, - } if matches!(&child.expr, SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Rate, _), - reduction: planner_types::pre_asap::Reduction::PerEntity, - child: raw, .. - } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => - { - Some(Rc::clone(node)) - } - SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { - frontier(child) - } - SummaryExpr::SummaryEstimate { summary_input, .. } => frontier(summary_input), - _ => None, + selected: &Rc, +) -> Result<(Rc, CompiledPhysicalDAG), Error> { + use planner_types::ir::schema::ExactKind; + fn frontier(node: &Rc) -> Option> { + if matches!(&node.operator, LogicalOperator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!(&child.operator, LogicalOperator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::ir::operator::Reduction::PerEntity, child: raw, .. + }) if matches!(raw.non_asap(), Some(NonASAPOp::TimeRange { .. })))) + { + return Some(Rc::clone(node)); } + node.children().into_iter().find_map(frontier) } - let source = frontier(selected) + // This compiler emits maintained precompute: every summary at ingestion time. + let selected = planner_types::ir::apply_materialization_timings( + selected, + &planner_types::ir::MaterializationAssignment::all_ingestion_time(), + &mut planner_types::ir::TimingMemo::new(), + ) + .map_err(|e| invalid(e.to_string()))?; + let source = frontier(&selected) .ok_or_else(|| invalid("ranking requires one exact per-series Rate frontier"))?; if !source .schema @@ -230,7 +218,7 @@ pub fn compile_rate_ranking( { return Err(invalid("Rate ranking requires complete series identity")); } - let compiled = compile_post_asap_dag_with_node_ids(selected) + let compiled = compile_physical_asap_dag_with_node_ids(&selected) .map_err(|error| invalid(error.to_string()))?; let id = u64::from( compiled @@ -242,18 +230,24 @@ pub fn compile_rate_ranking( let program = compile( &compiled.dag, BTreeMap::from([(id, InputContract::bounded(Arc::new(source.schema.clone())))]), - &[u64::from(compiled.dag.root.0)], + &compiled + .dag + .roots + .iter() + .map(|r| u64::from(r.0)) + .collect::>(), )?; Ok((source, program)) } -/// Compile a lifecycle-timed DAG whose heap or grouped Sum over per-series -/// Rate readouts runs at ingestion time: fresh aggregate state per closed +/// Compile a materialization-timed DAG whose heap or grouped Sum over per-series +/// Rate evaluations runs at ingestion time: fresh aggregate state per closed /// window. The input is the complete collection of per-series counter states. pub fn compile_fixed_window_rate_aggregation( - dag: &planner_types::post_asap::PostAsapDAG, -) -> Result { - use planner_types::post_asap::{ExactKind, ExecutionTiming, SketchAlgorithm}; + dag: &planner_types::ir::export::PhysicalASAPDAG, +) -> Result { + use planner_types::ir::properties::ExecutionTiming; + use planner_types::ir::schema::{ExactKind, SketchAlgorithm}; let sources = dag .nodes .iter() @@ -261,8 +255,8 @@ pub fn compile_fixed_window_rate_aggregation( matches!( &n.payload, Payload::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Rate, _), - reduction: planner_types::pre_asap::Reduction::PerEntity, + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::ir::operator::Reduction::PerEntity, .. } ) @@ -275,14 +269,14 @@ pub fn compile_fixed_window_rate_aggregation( n.output_state.timing == ExecutionTiming::IngestionTime && match &n.payload { Payload::SummaryAgg { - family: FieldDataType::Sketch(kind, _), + family: SummaryFamilyType::Sketch(kind, _), .. } => matches!( kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap ), Payload::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Sum, _), + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), .. } => true, _ => false, @@ -310,7 +304,7 @@ pub fn compile_fixed_window_rate_aggregation( u64::from(source.id.0), InputContract::bounded(Arc::new(source.output_schema.clone())), )]), - &[u64::from(dag.root.0)], + &dag.roots.iter().map(|r| u64::from(r.0)).collect::>(), &[u64::from(heap.id.0)], ) } diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/executor/src/physical_planner/promql_values.rs similarity index 88% rename from crates/asap-physical-operators/src/physical_planner/promql_values.rs rename to crates/executor/src/physical_planner/promql_values.rs index 880916ed2..43e5faec1 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/executor/src/physical_planner/promql_values.rs @@ -1,5 +1,6 @@ //! Physical scalar/vector contracts preserve complete label sets across native computation. use super::*; +use planner_types::ir::schema::FieldDataType as SummaryFamilyType; pub fn scalar_schema() -> SchemaRef { crate::operators::vector_binary::value_schema(true) @@ -15,7 +16,7 @@ pub fn matrix_schema() -> SchemaRef { pub fn compile_scalar(value: f64) -> Result { let operator = Operator::scalar( crate::values::Value::Float64(value), - planner_types::pre_asap::DataType::Float64, + planner_types::ir::schema::DataType::Float64, )? .with_output_schema(scalar_schema())?; CompiledPhysicalDAG::from_operators( @@ -63,7 +64,7 @@ pub fn compile_histogram_quantile() -> Result { /// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. pub fn compile_binary( - operator: &planner_types::post_asap::BinaryOperator, + operator: &crate::expressions::binary::BinaryOperator, return_bool: bool, left_scalar: bool, right_scalar: bool, @@ -209,8 +210,8 @@ pub fn compile_vector_to_scalar() -> Result { /// A stored exact-state input retains the complete population identity. The /// deployment supplies eligible panes; merging and finalization are computation. -pub fn exact_state_schema(family: FieldDataType) -> Result { - if !matches!(family, FieldDataType::ExactAggregate(..)) { +pub fn exact_state_schema(family: SummaryFamilyType) -> Result { + if !matches!(family, SummaryFamilyType::ExactAggregate(..)) { return Err(invalid("exact-state input requires an exact family")); } crate::values::validate_family(&family)?; @@ -219,31 +220,33 @@ pub fn exact_state_schema(family: FieldDataType) -> Result { Ok(Arc::new(schema)) } -/// Retain exact readout semantics before any deployment state is opened. -pub fn compile_exact_readout( - family: FieldDataType, +/// Retain exact evaluation semantics before any deployment state is opened. +pub fn compile_exact_evaluation( + family: SummaryFamilyType, lookback_ms: u64, preserve_metric_name: bool, ) -> Result { - use planner_types::post_asap::ExactKind; + use planner_types::ir::schema::ExactKind; let statistic = match &family { - FieldDataType::ExactAggregate(kind, _) => match kind { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { ExactKind::Sum => crate::Statistic::Sum, ExactKind::Count => crate::Statistic::Count, ExactKind::Min => crate::Statistic::Min, ExactKind::Max => crate::Statistic::Max, ExactKind::Rate => crate::Statistic::Rate, ExactKind::Increase => crate::Statistic::Increase, - ExactKind::IRate => return Err(invalid("instant-rate state readout is not supported")), + ExactKind::IRate => { + return Err(invalid("instant-rate state evaluation is not supported")) + } }, - _ => return Err(invalid("exact readout requires an exact family")), + _ => return Err(invalid("exact evaluation requires an exact family")), }; let input = exact_state_schema(family)?; let merge = Operator::summary_merge(input.clone(), 1, vec![0])?; - let mut readout = Operator::readout( + let mut evaluation = Operator::evaluation( merge.schema(), 1, - ReadoutQuery::Exact(ExactReadout { + SummaryEvaluation::Exact(ExactEvaluation { statistic, lookback_ms: None, }), @@ -252,12 +255,12 @@ pub fn compile_exact_readout( statistic, crate::Statistic::Rate | crate::Statistic::Increase ) { - readout = readout.with_counter_lookback( + evaluation = evaluation.with_counter_lookback( i64::try_from(lookback_ms).map_err(|_| invalid("counter lookback exceeds Int64"))?, )?; } let project = Operator::project( - readout.schema(), + evaluation.schema(), vec![ ( "labels".into(), @@ -274,5 +277,5 @@ pub fn compile_exact_readout( ("value".into(), Expression::ExactFloat64(1)), ], )?; - unary(vec![merge, readout, project], input) + unary(vec![merge, evaluation, project], input) } diff --git a/crates/asap-physical-operators/src/physical_planner/row_values.rs b/crates/executor/src/physical_planner/row_values.rs similarity index 71% rename from crates/asap-physical-operators/src/physical_planner/row_values.rs rename to crates/executor/src/physical_planner/row_values.rs index 3396df685..4c67064a2 100644 --- a/crates/asap-physical-operators/src/physical_planner/row_values.rs +++ b/crates/executor/src/physical_planner/row_values.rs @@ -1,29 +1,20 @@ //! Query-time PromQL value computation over logical row schemas. use super::*; -use planner_types::post_asap::maintained_population::PopulationStatistic; -use planner_types::pre_asap::{DataType, ScalarValue}; +use planner_types::ir::operator::maintained_population::PopulationStatistic; +use planner_types::ir::schema::DataType; -/// A PromQL number literal has no row schema; its consumer folds it in. -pub(super) fn scalar_literal(expression: &QueryExpr) -> Option { - match expression { - QueryExpr::PromqlScalarBridge(child) => scalar_literal(child), - QueryExpr::Literal(ScalarValue::Float64(value)) => Some(*value), - _ => None, - } -} - -/// Aggregate readouts of a maintained current-series population, as a chain. +/// Aggregate evaluations of a maintained current-series population, as a chain. pub(super) fn population_aggregate( input: &SchemaRef, grouping: &[String], - readout: &PopulationStatistic, + evaluation: &PopulationStatistic, ) -> Result, Error> { let groups = grouping .iter() .map(|name| named_column(input, &ColumnRef::Named(name.clone()))) .collect::, _>>()?; let value = named_column(input, &ColumnRef::SampleValue)?; - let reduction = match readout { + let reduction = match evaluation { PopulationStatistic::Sum => Reduction::Sum(value), PopulationStatistic::Count => Reduction::Count, PopulationStatistic::Average => Reduction::Avg(value), @@ -33,7 +24,7 @@ pub(super) fn population_aggregate( }, PopulationStatistic::TopK { .. } => { return Err(invalid( - "TopK population readout ranks; it does not aggregate", + "TopK population evaluation ranks; it does not aggregate", )) } }; diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/executor/src/plan/mod.rs similarity index 100% rename from crates/asap-physical-operators/src/plan/mod.rs rename to crates/executor/src/plan/mod.rs diff --git a/crates/asap-physical-operators/src/plan/properties.rs b/crates/executor/src/plan/properties.rs similarity index 100% rename from crates/asap-physical-operators/src/plan/properties.rs rename to crates/executor/src/plan/properties.rs diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/executor/src/runtime/batch_execution.rs similarity index 95% rename from crates/asap-physical-operators/src/runtime/batch_execution.rs rename to crates/executor/src/runtime/batch_execution.rs index ca59877cb..50132f462 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/executor/src/runtime/batch_execution.rs @@ -87,9 +87,8 @@ mod tests { runtime::{Limits, Scope}, values::Value, }; - use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::DataType, + use planner_types::ir::schema::{ + DataType, Field as SummaryField, FieldDataType as SummaryFamilyType, Schema, }; use std::sync::Arc; @@ -97,15 +96,15 @@ mod tests { #[test] fn same_native_chain_inside_query_and_ingestion_execution() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], - fields: vec![Field { - table: None, + fields: vec![SummaryField { name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), + dtype: SummaryFamilyType::Plain(DataType::Float64), nullable: false, + table: None, }], time_index: None, + unique_keys: vec![], + closed: false, }); for scope in [ Scope::Query { @@ -143,10 +142,10 @@ mod tests { #[test] fn in_memory_source_drives_cooperative_yields() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); let source = Operator::source(schema, vec![batch; 65]).unwrap(); @@ -165,10 +164,10 @@ mod tests { #[test] fn returned_batches_keep_their_resource_reservation() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); let bytes = batch.bytes(); @@ -196,10 +195,10 @@ mod tests { #[test] fn cancellation_is_not_bypassed_by_in_memory_execution() { let schema = Arc::new(Schema { - closed: true, - unique_keys: vec![], fields: vec![], time_index: None, + unique_keys: vec![], + closed: false, }); let batch = Batch::try_new(schema, vec![vec![]]).unwrap(); let context = RunContext::new( diff --git a/crates/asap-physical-operators/src/runtime/context.rs b/crates/executor/src/runtime/context.rs similarity index 100% rename from crates/asap-physical-operators/src/runtime/context.rs rename to crates/executor/src/runtime/context.rs diff --git a/crates/asap-physical-operators/src/runtime/cooperative.rs b/crates/executor/src/runtime/cooperative.rs similarity index 100% rename from crates/asap-physical-operators/src/runtime/cooperative.rs rename to crates/executor/src/runtime/cooperative.rs diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/executor/src/runtime/mod.rs similarity index 100% rename from crates/asap-physical-operators/src/runtime/mod.rs rename to crates/executor/src/runtime/mod.rs diff --git a/crates/asap-physical-operators/src/runtime/tests.rs b/crates/executor/src/runtime/tests.rs similarity index 100% rename from crates/asap-physical-operators/src/runtime/tests.rs rename to crates/executor/src/runtime/tests.rs diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/executor/src/sources/memory.rs similarity index 95% rename from crates/asap-physical-operators/src/sources/memory.rs rename to crates/executor/src/sources/memory.rs index 1055888de..856c73cb0 100644 --- a/crates/asap-physical-operators/src/sources/memory.rs +++ b/crates/executor/src/sources/memory.rs @@ -11,7 +11,7 @@ impl MemorySource { if schema .fields .iter() - .any(|f| !matches!(f.dtype, FieldDataType::Plain(_))) + .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) { return Err(Error::Invalid( "raw source cannot contain summary states".into(), diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/executor/src/sources/mod.rs similarity index 94% rename from crates/asap-physical-operators/src/sources/mod.rs rename to crates/executor/src/sources/mod.rs index 4b401d666..621fe692b 100644 --- a/crates/asap-physical-operators/src/sources/mod.rs +++ b/crates/executor/src/sources/mod.rs @@ -7,10 +7,9 @@ use crate::{ Error, }; use futures::{stream, StreamExt}; -use planner_types::{ - post_asap::{FieldDataType, Schema}, - pre_asap::{DataType, QueryExpr, Source}, -}; +use planner_types::ir::operator::Source; +use planner_types::ir::schema::{DataType, FieldDataType as SummaryFamilyType}; +use planner_types::ir::{NonASAPOp, OperatorNode}; use std::sync::Arc; /// A bound data source. Metadata must be stable for the lifetime of the binding. @@ -40,18 +39,18 @@ impl DataSources { self.sources.push((identity, source)); Ok(()) } - pub fn bind(&self, expression: &QueryExpr) -> Result { - let QueryExpr::Scan { + pub fn bind(&self, expression: &OperatorNode) -> Result { + let Some(NonASAPOp::Scan { source, predicates, schema, - } = expression + }) = expression.non_asap() else { return Err(Error::Invalid( "raw Scan requires a Planner Scan leaf".into(), )); }; - let output = Arc::new(Schema::lifted(schema.fields.clone(), schema.time_index)); + let output = Arc::new(schema.clone()); crate::values::validate_schema(&output)?; let reader = self .sources diff --git a/crates/asap-physical-operators/src/statistic.rs b/crates/executor/src/statistic.rs similarity index 100% rename from crates/asap-physical-operators/src/statistic.rs rename to crates/executor/src/statistic.rs diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs b/crates/executor/src/summary_kernels/count_min_sketch.rs similarity index 96% rename from crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs rename to crates/executor/src/summary_kernels/count_min_sketch.rs index af0815c25..20f49f092 100644 --- a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs +++ b/crates/executor/src/summary_kernels/count_min_sketch.rs @@ -1,7 +1,7 @@ //! Count-Min Sketch frequency summary over `asap_sketchlib::CountMinSketch`. use crate::{AggregateCore, KernelError, KeyByLabelValues}; use asap_sketchlib::CountMinSketch; -use planner_types::post_asap::SketchStatistic; +use planner_types::ir::schema::SketchStatistic; #[derive(Debug, Clone)] pub struct CountMinSketchAccumulator { @@ -87,7 +87,7 @@ mod tests { #[test] fn bare_count_reads_total_weight() { let bare_count = SketchStatistic::PointCount { - key: planner_types::pre_asap::ColumnRef::SampleValue, + key: planner_types::ir::scalar::ColumnRef::SampleValue, value: None, }; let mut state = CountMinSketchAccumulator::new(2, 1); diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs b/crates/executor/src/summary_kernels/count_min_sketch_with_heap.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs rename to crates/executor/src/summary_kernels/count_min_sketch_with_heap.rs diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs b/crates/executor/src/summary_kernels/count_sketch.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_kernels/count_sketch.rs rename to crates/executor/src/summary_kernels/count_sketch.rs diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs b/crates/executor/src/summary_kernels/count_sketch_with_heap.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs rename to crates/executor/src/summary_kernels/count_sketch_with_heap.rs diff --git a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs b/crates/executor/src/summary_kernels/datasketches_kll.rs similarity index 98% rename from crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs rename to crates/executor/src/summary_kernels/datasketches_kll.rs index e4ecb56a0..d64607062 100644 --- a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs +++ b/crates/executor/src/summary_kernels/datasketches_kll.rs @@ -1,7 +1,7 @@ //! KLL quantile summary over `asap_sketchlib::KllSketch`. use crate::{AggregateCore, KernelError}; use asap_sketchlib::KllSketch; -use planner_types::post_asap::SketchStatistic; +use planner_types::ir::schema::SketchStatistic; #[derive(Clone)] pub struct DatasketchesKLLAccumulator { diff --git a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs b/crates/executor/src/summary_kernels/dd_sketch.rs similarity index 96% rename from crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs rename to crates/executor/src/summary_kernels/dd_sketch.rs index e9cdf9d9c..72f786ef5 100644 --- a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs +++ b/crates/executor/src/summary_kernels/dd_sketch.rs @@ -1,7 +1,7 @@ //! DDSketch quantile summary over `asap_sketchlib::DdSketch`. use crate::{AggregateCore, KernelError}; use asap_sketchlib::DdSketch; -use planner_types::post_asap::SketchStatistic; +use planner_types::ir::schema::SketchStatistic; #[derive(Debug, Clone)] pub struct DDSketchAccumulator { @@ -55,7 +55,7 @@ impl AggregateCore for DDSketchAccumulator { #[cfg(test)] mod tests { use super::*; - use planner_types::pre_asap::ColumnRef; + use planner_types::ir::scalar::ColumnRef; fn bare_count() -> SketchStatistic { SketchStatistic::PointCount { diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/executor/src/summary_kernels/exact.rs similarity index 80% rename from crates/asap-physical-operators/src/summary_kernels/exact.rs rename to crates/executor/src/summary_kernels/exact.rs index d4754686e..38713be15 100644 --- a/crates/asap-physical-operators/src/summary_kernels/exact.rs +++ b/crates/executor/src/summary_kernels/exact.rs @@ -2,7 +2,7 @@ use super::increase::IncreaseAccumulator; use crate::Statistic; use crate::{AggregateCore, KeyByLabelValues, Measurement}; -use planner_types::post_asap::{ExactKind, ExactParams, FieldDataType}; +use planner_types::ir::schema::{ExactKind, ExactParams, FieldDataType as SummaryFamilyType}; use serde::{Deserialize, Serialize}; use std::collections::HashMap; @@ -10,7 +10,11 @@ type Error = Box; #[derive(Debug, Clone, Serialize, Deserialize)] enum ScalarState { - Sum { sum: f64, compensation: f64 }, + Sum { + sum: f64, + compensation: f64, + seen: bool, + }, Count(u64), Min(Option), Max(Option), @@ -18,7 +22,7 @@ enum ScalarState { } /// Both the family and population layout survive persistence. Sharing counter -/// arithmetic never authorizes a Rate state to answer an Increase readout. +/// arithmetic never authorizes a Rate state to answer an Increase evaluation. /// /// Deserialization validates the payload against its declared family, so /// deployments can persist this state with any serde format without mirroring @@ -26,14 +30,14 @@ enum ScalarState { #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(try_from = "ExactPayload")] pub struct ExactAccumulator { - family: FieldDataType, + family: SummaryFamilyType, scalar: ScalarState, keyed: Option>, } #[derive(Deserialize)] struct ExactPayload { - family: FieldDataType, + family: SummaryFamilyType, scalar: ScalarState, keyed: Option>, } @@ -63,10 +67,10 @@ impl TryFrom for ExactAccumulator { } } -/// Planned readout of an exact summary. `lookback_ms` is the logical PromQL +/// Planned evaluation of an exact summary. `lookback_ms` is the logical PromQL /// counter window; the evaluation range is resolved from it at run time. #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -pub struct ExactReadout { +pub struct ExactEvaluation { pub statistic: Statistic, #[serde(default, skip_serializing_if = "Option::is_none")] pub lookback_ms: Option, @@ -75,22 +79,24 @@ pub struct ExactReadout { impl ExactAccumulator { /// Read one population. An empty MIN/MAX population reads as `None`. /// `range_ms` extrapolates a counter Rate/Increase to that evaluation range. - pub fn readout( + pub fn evaluation( &self, statistic: Statistic, range_ms: Option<(i64, i64)>, key: Option<&KeyByLabelValues>, ) -> Result, Error> { if statistic != self.statistic() { - return Err("readout differs from Planner exact family".into()); + return Err("evaluation differs from Planner exact family".into()); } let state = match (&self.keyed, key) { (Some(states), Some(key)) => states.get(key).ok_or("unknown exact population")?, (None, None) => &self.scalar, - _ => return Err("readout population differs from installed layout".into()), + _ => return Err("evaluation population differs from installed layout".into()), }; match state { - ScalarState::Sum { sum, compensation } => Ok(Some(sum + compensation)), + ScalarState::Sum { + sum, compensation, .. + } => Ok(Some(sum + compensation)), ScalarState::Count(count) => Ok(Some(*count as f64)), ScalarState::Min(value) | ScalarState::Max(value) => Ok(*value), ScalarState::Counter(Some(counter)) => counter @@ -100,6 +106,11 @@ impl ExactAccumulator { } } + /// SQL SUM distinguishes an empty/all-NULL input from an observed zero. + pub(crate) fn is_empty_sum(&self) -> bool { + self.keyed.is_none() && matches!(self.scalar, ScalarState::Sum { seen: false, .. }) + } + /// Exact integer count of an unkeyed Count state. pub fn count(&self) -> Option { match (&self.keyed, &self.scalar) { @@ -128,19 +139,22 @@ impl ExactAccumulator { Ok(()) } - pub fn new(family: FieldDataType, keyed: bool) -> Result { + pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { use ExactKind as K; use ExactParams as P; let scalar = match &family { - FieldDataType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum { + SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, - FieldDataType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), - FieldDataType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), - FieldDataType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), - FieldDataType::ExactAggregate(K::Rate, P::Rate) - | FieldDataType::ExactAggregate(K::Increase, P::Increase) => ScalarState::Counter(None), + SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), + SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), + SummaryFamilyType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), + SummaryFamilyType::ExactAggregate(K::Rate, P::Rate) + | SummaryFamilyType::ExactAggregate(K::Increase, P::Increase) => { + ScalarState::Counter(None) + } _ => return Err(format!("unsupported exact Planner family: {family:?}")), }; Ok(Self { @@ -150,7 +164,7 @@ impl ExactAccumulator { }) } - pub fn family(&self) -> &FieldDataType { + pub fn family(&self) -> &SummaryFamilyType { &self.family } pub(crate) fn insufficient_counter_samples( @@ -188,7 +202,14 @@ impl ExactAccumulator { _ => panic!("exact update population layout differs from installed DAG"), }; match state { - ScalarState::Sum { sum, compensation } => compensated_add(sum, compensation, value), + ScalarState::Sum { + sum, + compensation, + seen, + } => { + compensated_add(sum, compensation, value); + *seen = true; + } ScalarState::Count(count) => { *count = count.checked_add(1).expect("exact count overflow") } @@ -214,12 +235,12 @@ impl ExactAccumulator { fn statistic(&self) -> Statistic { match self.family { - FieldDataType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, - FieldDataType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, - FieldDataType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, - FieldDataType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, - FieldDataType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, - FieldDataType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, + SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, + SummaryFamilyType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, + SummaryFamilyType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, + SummaryFamilyType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, + SummaryFamilyType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, + SummaryFamilyType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, _ => unreachable!("validated exact family"), } } @@ -244,16 +265,22 @@ fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result { let (mut sum, mut compensation) = (*a, *ac); compensated_add(&mut sum, &mut compensation, *b); compensated_add(&mut sum, &mut compensation, *bc); - ScalarState::Sum { sum, compensation } + ScalarState::Sum { + sum, + compensation, + seen: *a_seen || *b_seen, + } } (ScalarState::Count(a), ScalarState::Count(b)) => { ScalarState::Count(a.checked_add(*b).ok_or("exact count overflow")?) @@ -311,7 +338,7 @@ mod tests { #[derive(Serialize)] struct Payload { - family: FieldDataType, + family: SummaryFamilyType, scalar: ScalarState, keyed: Option>, } @@ -320,8 +347,8 @@ mod tests { rmp_serde::from_slice(&rmp_serde::to_vec_named(payload).unwrap()) } - fn sum() -> FieldDataType { - FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + fn sum() -> SummaryFamilyType { + SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) } // Stored Sum preserves low-order increments across updates, persistence and pane merge. @@ -337,7 +364,7 @@ mod tests { negative.update(None, -1e16, 1); restored.merge_from(&negative).unwrap(); assert_eq!( - restored.readout(Statistic::Sum, None, None).unwrap(), + restored.evaluation(Statistic::Sum, None, None).unwrap(), Some(1.0) ); } @@ -349,18 +376,18 @@ mod tests { state.update(None, f64::INFINITY, 0); state.update(None, 1.0, 0); assert_eq!( - state.readout(Statistic::Sum, None, None).unwrap(), + state.evaluation(Statistic::Sum, None, None).unwrap(), Some(f64::INFINITY) ); state.update(None, f64::NEG_INFINITY, 0); assert!(state - .readout(Statistic::Sum, None, None) + .evaluation(Statistic::Sum, None, None) .unwrap() .unwrap() .is_nan()); } - // A persisted exact state decodes back to the same family, layout and readout. + // A persisted exact state decodes back to the same family, layout and evaluation. #[test] fn serialized_state_round_trips() { let mut state = ExactAccumulator::new(sum(), true).unwrap(); @@ -370,7 +397,9 @@ mod tests { let restored: ExactAccumulator = rmp_serde::from_slice(&bytes).unwrap(); assert_eq!(restored.family(), &sum()); assert_eq!( - restored.readout(Statistic::Sum, None, Some(&key)).unwrap(), + restored + .evaluation(Statistic::Sum, None, Some(&key)) + .unwrap(), Some(2.5) ); } @@ -390,6 +419,7 @@ mod tests { scalar: ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, keyed: Some(HashMap::from([(key, ScalarState::Max(Some(1.0)))])), }; @@ -400,10 +430,11 @@ mod tests { #[test] fn decode_rejects_unsupported_family() { let payload = Payload { - family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Count), + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Count), scalar: ScalarState::Sum { sum: 0.0, compensation: 0.0, + seen: false, }, keyed: None, }; diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/executor/src/summary_kernels/factory.rs similarity index 97% rename from crates/asap-physical-operators/src/summary_kernels/factory.rs rename to crates/executor/src/summary_kernels/factory.rs index 4be126d6e..f6aca692a 100644 --- a/crates/asap-physical-operators/src/summary_kernels/factory.rs +++ b/crates/executor/src/summary_kernels/factory.rs @@ -6,7 +6,9 @@ use crate::summary_kernels::{ HydraKllSketchAccumulator, }; use crate::{AggregateCore, KeyByLabelValues}; -use planner_types::post_asap::{FieldDataType, SketchAlgorithm, SketchParams}; +use planner_types::ir::schema::{ + FieldDataType as SummaryFamilyType, SketchAlgorithm, SketchParams, +}; /// Generate the clone-based `AccumulatorUpdater` methods for updaters whose /// inner `acc` field implements `Clone + AggregateCore`. @@ -527,16 +529,16 @@ fn cms_heap_dims(params: &SketchParams) -> (usize, usize, usize) { /// Construct the kernel declared by a Planner SummaryAgg. No deployment config /// tags participate in this dispatch and unsupported payloads are errors. pub fn create_planner_accumulator( - family: &FieldDataType, - input: &planner_types::post_asap::SummaryUpdate, - grouping: &planner_types::post_asap::GroupingStrategy, + family: &SummaryFamilyType, + input: &planner_types::ir::schema::SummaryUpdate, + grouping: &planner_types::ir::schema::GroupingStrategy, ) -> Result, String> { if input.item.is_some() && matches!( input.weight_domain, - planner_types::post_asap::WeightDomain::NonNegative { + planner_types::ir::schema::WeightDomain::NonNegative { proof: - planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + planner_types::ir::schema::NonNegativeWeightProof::ResetAwareCounterDerivative } ) { @@ -544,11 +546,11 @@ pub fn create_planner_accumulator( } crate::capability::validate_summary_kernel(family, input, grouping)?; - use planner_types::post_asap::GroupingStrategy; + use planner_types::ir::schema::GroupingStrategy; if grouping != &GroupingStrategy::PerSubpopulationInstance { return Err("shared summary grouping requires a supported Planner Hydra kernel".into()); } - if matches!(family, FieldDataType::ExactAggregate(..)) { + if matches!(family, SummaryFamilyType::ExactAggregate(..)) { return Ok(Box::new(PlannerExactUpdater { acc: crate::summary_kernels::exact::ExactAccumulator::new( family.clone(), @@ -556,7 +558,7 @@ pub fn create_planner_accumulator( )?, })); } - let FieldDataType::Sketch(kind, family_grouping) = family else { + let SummaryFamilyType::Sketch(kind, family_grouping) = family else { return Err(format!("unsupported Planner summary family {family:?}")); }; if family_grouping != grouping { @@ -711,7 +713,7 @@ impl AccumulatorUpdater for UnivMonUpdater { #[cfg(test)] mod planner_parameter_regression { use super::*; - use planner_types::post_asap::{SketchKind, SummaryInputExpr, SummaryUpdate}; + use planner_types::ir::schema::{SketchKind, SummaryInputExpr, SummaryUpdate}; // Planner width is the bucket count; depth is the independent hash-row count. #[test] @@ -748,13 +750,13 @@ mod planner_parameter_regression { }, ), ] { - let family = FieldDataType::Sketch( + let family = SummaryFamilyType::Sketch( SketchKind::new(algorithm.clone(), params), Default::default(), ); let update = SummaryUpdate { item: Some(SummaryInputExpr::Column( - planner_types::pre_asap::ColumnRef::Named("host".into()), + planner_types::ir::scalar::ColumnRef::Named("host".into()), )), weight: SummaryInputExpr::Constant(1.0), weight_domain: Default::default(), diff --git a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs b/crates/executor/src/summary_kernels/hll_sketch.rs similarity index 98% rename from crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs rename to crates/executor/src/summary_kernels/hll_sketch.rs index 993f9d9b6..166d8c2db 100644 --- a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs +++ b/crates/executor/src/summary_kernels/hll_sketch.rs @@ -1,7 +1,7 @@ //! HyperLogLog distinct-count summary over `asap_sketchlib::HllSketch`. use crate::{AggregateCore, KernelError}; use asap_sketchlib::{HllSketch, HllVariant}; -use planner_types::post_asap::SketchStatistic; +use planner_types::ir::schema::SketchStatistic; #[derive(Debug, Clone)] pub struct HllSketchAccumulator { diff --git a/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs b/crates/executor/src/summary_kernels/hydra_kll.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs rename to crates/executor/src/summary_kernels/hydra_kll.rs diff --git a/crates/asap-physical-operators/src/summary_kernels/increase.rs b/crates/executor/src/summary_kernels/increase.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_kernels/increase.rs rename to crates/executor/src/summary_kernels/increase.rs diff --git a/crates/asap-physical-operators/src/summary_kernels/mod.rs b/crates/executor/src/summary_kernels/mod.rs similarity index 100% rename from crates/asap-physical-operators/src/summary_kernels/mod.rs rename to crates/executor/src/summary_kernels/mod.rs diff --git a/crates/asap-physical-operators/src/summary_kernels/traits.rs b/crates/executor/src/summary_kernels/traits.rs similarity index 84% rename from crates/asap-physical-operators/src/summary_kernels/traits.rs rename to crates/executor/src/summary_kernels/traits.rs index 9c5d028fb..8fd0b52d1 100644 --- a/crates/asap-physical-operators/src/summary_kernels/traits.rs +++ b/crates/executor/src/summary_kernels/traits.rs @@ -1,11 +1,11 @@ -use planner_types::post_asap::SketchStatistic; +use planner_types::ir::schema::SketchStatistic; pub type KernelError = Box; /// In-memory state of one population's summary. /// /// Kernels adapt `asap_sketchlib` structures (or exact Planner state) to the -/// operations physical operators need: merge, typed readout and memory +/// operations physical operators need: merge, typed evaluation and memory /// accounting. Grouping belongs to operators; byte encodings belong to /// `asap_sketchlib` and deployments. pub trait AggregateCore: Send + Sync { @@ -20,8 +20,8 @@ pub trait AggregateCore: Send + Sync { /// Merge with a state of the same family and shape, leaving both inputs unchanged. fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError>; - /// Answer a sketch readout. Exact states are read through - /// [`ExactAccumulator::readout`](super::exact::ExactAccumulator::readout). + /// Answer a sketch evaluation. Exact states are read through + /// [`ExactAccumulator::evaluation`](super::exact::ExactAccumulator::evaluation). fn estimate(&self, query: &SketchStatistic) -> Result { Err(format!("{query:?} is not supported by this summary").into()) } @@ -53,7 +53,7 @@ mod tests { .unwrap(); dd.inner.update(3.0); let count = SketchStatistic::PointCount { - key: planner_types::pre_asap::ColumnRef::SampleValue, + key: planner_types::ir::scalar::ColumnRef::SampleValue, value: None, }; assert_eq!(state.estimate(&count).unwrap(), 1.0); diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/executor/src/summary_kernels/univmon.rs similarity index 95% rename from crates/asap-physical-operators/src/summary_kernels/univmon.rs rename to crates/executor/src/summary_kernels/univmon.rs index 9340b1c0e..54904282e 100644 --- a/crates/asap-physical-operators/src/summary_kernels/univmon.rs +++ b/crates/executor/src/summary_kernels/univmon.rs @@ -1,8 +1,9 @@ -//! One frequency state shared by count, distinct, L2 and entropy readouts. +//! One frequency state shared by count, distinct, L2 and entropy evaluations. use crate::AggregateCore; use asap_sketchlib::{DataInput, UnivMon}; -use planner_types::{post_asap::SketchStatistic, pre_asap::ColumnRef}; +use planner_types::ir::scalar::ColumnRef; +use planner_types::ir::schema::SketchStatistic; type Error = Box; @@ -169,10 +170,10 @@ mod tests { } } - // Count, distinct, L2 and entropy readouts count each non-NaN sample once; + // Count, distinct, L2 and entropy evaluations count each non-NaN sample once; // signed zero is one identity. #[test] - fn frequency_readouts() { + fn frequency_evaluations() { let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); for value in [0.0, -0.0, 2.0, 2.0, f64::NAN] { state.insert_sample(value).unwrap(); @@ -187,9 +188,9 @@ mod tests { .is_err()); } - // A sketch taken out and adopted back answers the same readouts. + // A sketch taken out and adopted back answers the same evaluations. #[test] - fn adopted_sketch_keeps_readouts() { + fn adopted_sketch_keeps_evaluations() { let mut state = UnivMonAccumulator::new(32, 5, 1024, 4).unwrap(); for value in [1.0, 2.0, 2.0] { state.insert_sample(value).unwrap(); diff --git a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs b/crates/executor/src/summary_kernels/weighted_frequency.rs similarity index 97% rename from crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs rename to crates/executor/src/summary_kernels/weighted_frequency.rs index 7c1122305..8af54a37b 100644 --- a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs +++ b/crates/executor/src/summary_kernels/weighted_frequency.rs @@ -41,9 +41,9 @@ pub struct WeightedFrequency { } impl WeightedFrequency { pub(crate) fn configuration( - kind: &planner_types::post_asap::SketchKind, + kind: &planner_types::ir::schema::SketchKind, ) -> Result<(FrequencyAlgorithm, usize, usize, usize), Error> { - use planner_types::post_asap::{SketchAlgorithm as A, SketchParams as P}; + use planner_types::ir::schema::{SketchAlgorithm as A, SketchParams as P}; let (algorithm, width, depth, capacity) = match (kind.algorithm(), kind.params()) { ( A::CmsWithHeap, diff --git a/crates/asap-physical-operators/src/values.rs b/crates/executor/src/values.rs similarity index 92% rename from crates/asap-physical-operators/src/values.rs rename to crates/executor/src/values.rs index b1186c171..9feb0fac3 100644 --- a/crates/asap-physical-operators/src/values.rs +++ b/crates/executor/src/values.rs @@ -1,9 +1,8 @@ //! Runtime values preserve Planner schemas; summary states are typed values too. use crate::AggregateCore; use crate::Error; -use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::DataType, +use planner_types::ir::schema::{ + DataType, Field as SummaryField, FieldDataType as SummaryFamilyType, Schema, }; use std::{cmp::Ordering, sync::Arc}; /// Shared ownership of schema metadata; the field model is identical at planning @@ -28,7 +27,7 @@ pub enum Value { Map(Arc<[(Value, Value)]>), #[serde(skip)] Summary { - family: FieldDataType, + family: SummaryFamilyType, state: Arc, }, } @@ -214,7 +213,9 @@ impl Batch { } for (value, field) in row.iter().zip(&schema.fields) { let matches = match (&field.dtype, value) { - (FieldDataType::Plain(dtype), value) => value.matches(dtype, field.nullable), + (SummaryFamilyType::Plain(dtype), value) => { + value.matches(dtype, field.nullable) + } (expected, Value::Summary { family, state }) => { expected == family && validate_state(family, state.as_ref()).is_ok() } @@ -260,15 +261,15 @@ pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result> pub(crate) use crate::capability::validate_native_family as validate_family; -fn validate_state(family: &FieldDataType, state: &dyn AggregateCore) -> Result<(), Error> { +fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { use crate::summary_kernels::{ count_min_sketch::CountMinSketchAccumulator, datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, }; - use planner_types::post_asap::SketchParams; + use planner_types::ir::schema::SketchParams; validate_family(family)?; let valid = match family { - FieldDataType::Sketch(kind, _) + SummaryFamilyType::Sketch(kind, _) if matches!( kind.params(), SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } @@ -284,11 +285,11 @@ fn validate_state(family: &FieldDataType, state: &dyn AggregateCore) -> Result<( }) } - FieldDataType::ExactAggregate(..) => state + SummaryFamilyType::ExactAggregate(..) => state .as_any() .downcast_ref::() .is_some_and(|s| s.family() == family && !s.is_keyed()), - FieldDataType::Sketch(kind, _) => match kind.params() { + SummaryFamilyType::Sketch(kind, _) => match kind.params() { SketchParams::Kll { k } => state .as_any() .downcast_ref::() @@ -325,14 +326,14 @@ pub(crate) fn validate_schema(schema: &SchemaRef) -> Result<(), Error> { schema .fields .get(index) - .is_none_or(|field| field.dtype != FieldDataType::Plain(DataType::Timestamp)) + .is_none_or(|field| field.dtype != SummaryFamilyType::Plain(DataType::Timestamp)) }) { return Err(Error::Invalid( "time index must name a Timestamp column".into(), )); } for field in &schema.fields { - if !matches!(field.dtype, FieldDataType::Plain(_)) { + if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { validate_family(&field.dtype)?; if field.nullable { return Err(Error::Invalid( @@ -344,7 +345,7 @@ pub(crate) fn validate_schema(schema: &SchemaRef) -> Result<(), Error> { Ok(()) } -pub(crate) fn field(schema: &SchemaRef, column: usize) -> Result<&Field, Error> { +pub(crate) fn field(schema: &SchemaRef, column: usize) -> Result<&SummaryField, Error> { schema .fields .get(column) @@ -352,7 +353,7 @@ pub(crate) fn field(schema: &SchemaRef, column: usize) -> Result<&Field, Error> } pub(crate) fn plain(schema: &SchemaRef, column: usize) -> Result<(&DataType, bool), Error> { let f = field(schema, column)?; - let FieldDataType::Plain(dtype) = &f.dtype else { + let SummaryFamilyType::Plain(dtype) = &f.dtype else { return Err(Error::Invalid("plain value required".into())); }; Ok((dtype, f.nullable)) @@ -362,12 +363,12 @@ pub(crate) fn plain(schema: &SchemaRef, column: usize) -> Result<(&DataType, boo mod weighted_state_tests { use super::*; use crate::summary_kernels::weighted_frequency::{FrequencyAlgorithm, WeightedFrequency}; - use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + use planner_types::ir::schema::{SketchAlgorithm, SketchKind, SketchParams}; // A state cannot acquire a different family or shape merely by relabeling its batch. #[test] fn weighted_state_family_and_shape_must_match() { - let cms = FieldDataType::Sketch( + let cms = SummaryFamilyType::Sketch( SketchKind::new( SketchAlgorithm::CmsWithHeap, SketchParams::CmsWithHeap { @@ -378,7 +379,7 @@ mod weighted_state_tests { ), Default::default(), ); - let cs = FieldDataType::Sketch( + let cs = SummaryFamilyType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { @@ -395,7 +396,7 @@ mod weighted_state_tests { let wrong_shape = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 5, 8).unwrap(); assert!(validate_state(&cs, &wrong_shape).is_err()); - let even_depth = FieldDataType::Sketch( + let even_depth = SummaryFamilyType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, SketchParams::CountSketchWithHeap { diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/executor/tests/blocking_resources.rs similarity index 91% rename from crates/asap-physical-operators/tests/blocking_resources.rs rename to crates/executor/tests/blocking_resources.rs index 37f780812..a407f7e0c 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/executor/tests/blocking_resources.rs @@ -1,5 +1,5 @@ //! Blocking operators enforce resources before returning their first batch. -use asap_physical_operators::{ +use asap_executor::{ operators::Operator, plan::{PhysicalDAG, PhysicalOperator}, runtime::{Limits, RunContext, Scope}, @@ -7,16 +7,17 @@ use asap_physical_operators::{ Error, }; use futures::{executor::block_on, FutureExt, StreamExt}; -use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, -}; +use planner_types::ir::operator::JoinKind; +use planner_types::ir::scalar::ScalarValue; +use planner_types::ir::schema::{DataType, Field, FieldDataType}; +use planner_types::ir::Predicate; +use planner_types::ir::ScalarExpr; use std::sync::Arc; fn schema(width: usize) -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: (0..width) .map(|i| Field { table: None, @@ -60,9 +61,7 @@ fn cross_join() -> Operator { schema(1), schema(1), JoinKind::Cross, - &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))), + &Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), schema(2), ) .unwrap() @@ -108,7 +107,7 @@ fn join_yields_during_computation_and_observes_cancellation() { // Sorting and grouping yield even for one large batch. #[test] fn blocking_reductions_yield_and_release_memory_on_cancellation() { - use asap_physical_operators::{ + use asap_executor::{ operators::{Reduction, SortKey}, plan::PhysicalOperator, }; @@ -144,7 +143,7 @@ fn blocking_reductions_yield_and_release_memory_on_cancellation() { // Merge-sort rounds preserve input order for tied keys across chunk boundaries. #[test] fn cooperative_sort_preserves_ties_across_chunks() { - use asap_physical_operators::operators::SortKey; + use asap_executor::operators::SortKey; let batch = Batch::try_new( schema(2), (0..1025) @@ -188,10 +187,11 @@ fn cooperative_sort_preserves_ties_across_chunks() { // The integrated weighted-summary path obeys the same cooperative cancellation contract. #[test] fn weighted_summary_build_yields_within_a_batch() { - use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - let input = Arc::new(Schema { - closed: true, + use planner_types::ir::schema::{SketchAlgorithm, SketchKind, SketchParams}; + + let input = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, diff --git a/crates/executor/tests/common/mod.rs b/crates/executor/tests/common/mod.rs new file mode 100644 index 000000000..b3602adbe --- /dev/null +++ b/crates/executor/tests/common/mod.rs @@ -0,0 +1,17 @@ +#![allow(dead_code)] +use planner_types::ir::export::PhysicalASAPDAG; +use planner_types::ir::{ + apply_materialization_timings, MaterializationAssignment, OperatorNode, TimingMemo, +}; +use std::rc::Rc; + +pub fn compile_physical_asap_dag( + root: &Rc, +) -> Result> { + let root = apply_materialization_timings( + root, + &MaterializationAssignment::default(), + &mut TimingMemo::default(), + )?; + Ok(planner_types::ir::export::compile_physical_asap_dag(&root)?) +} diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/executor/tests/current_series_heap.rs similarity index 90% rename from crates/asap-physical-operators/tests/current_series_heap.rs rename to crates/executor/tests/current_series_heap.rs index 079c2bd69..e4ece4407 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/executor/tests/current_series_heap.rs @@ -1,5 +1,6 @@ //! Spatial heap weights come from a fresh instant vector, never sample history. -use asap_physical_operators::{ +mod common; +use asap_executor::{ operators::Operator, physical_planner::{ promql_rows::{decode_series_identity, series_row, SERIES_IDENTITY_COLUMN}, @@ -8,15 +9,16 @@ use asap_physical_operators::{ runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_physical_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; -use planner_types::{post_asap::*, pre_asap::DataType}; +use planner_types::ir::export::PhysicalASAPOperatorPayload; +use planner_types::ir::schema::{DataType, *}; use std::{collections::BTreeMap, sync::Arc}; fn schema() -> Arc { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: [ ("ts", DataType::Timestamp), ("value", DataType::Float64), @@ -169,9 +171,9 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { }; let family = FieldDataType::Sketch(SketchKind::new(algorithm, params), Default::default()); let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); - let output = Arc::new(Schema { - closed: true, + let output = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![ schema().fields[2].clone(), schema().fields[3].clone(), @@ -179,7 +181,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { ], time_index: None, }); - let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); + let read = Operator::keyed_evaluation(build.schema(), 1, 1, output).unwrap(); let plan = CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema()))]), BTreeMap::from([ @@ -231,7 +233,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { // Blocking membership selection shares the run's cancellation and byte budget. #[test] fn current_series_observes_resource_limits() { - use asap_physical_operators::Error; + use asap_executor::Error; let plan = snapshot_plan(); for cancelled in [false, true] { let data = input(&[("one", 50_000, 1.)]); @@ -270,7 +272,7 @@ fn current_series_observes_resource_limits() { #[test] fn identity_encoding_is_lossless_and_rejects_noncanonical_inputs() { - use asap_physical_operators::physical_planner::promql_rows::encode_series_identity; + use asap_executor::physical_planner::promql_rows::encode_series_identity; let labels = BTreeMap::from([ ("a".into(), "quote\"slash\\".into()), ("other".into(), "".into()), @@ -293,7 +295,7 @@ fn identity_encoding_is_lossless_and_rejects_noncanonical_inputs() { // test does not manually assemble the computation or its dependency edges. #[test] fn planner_current_series_candidate_compiles_with_dynamic_identity() { - use asap_physical_operators::physical_planner::{compile, promql_rows::with_series_identity}; + use asap_executor::physical_planner::{compile, promql_rows::with_series_identity}; use planner_types::{types::AccuracyTarget, workload::*}; use std::rc::Rc; let workload = PlanningWorkload { @@ -325,13 +327,13 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { .remove(0); let open_root = Rc::new(original.clone()); let open_selected = - asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + asap_logical_optimizer::pass1::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&open_root), ) .candidate(&open_root) .unwrap(); let snapshot_program = - asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + asap_executor::physical_planner::promql_rows::compile_current_series_evaluation( &open_selected, ) .unwrap(); @@ -343,16 +345,24 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { assert!(encoded.contains("Sort") && encoded.contains("Limit")); assert_eq!(snapshot_program.input_contracts().count(), 1); let root = Rc::new(with_series_identity(&original).unwrap()); - let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( - std::slice::from_ref(&root), - ) - .candidate(&root) - .unwrap(); - let logical = compile_post_asap_dag(&selected).unwrap(); + let selected = + asap_logical_optimizer::pass1::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let logical = compile_physical_asap_dag(&selected).unwrap(); let raw = logical .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); let raw_schema = Arc::new(raw.output_schema.clone()); let physical = compile( @@ -361,7 +371,7 @@ fn planner_current_series_candidate_compiles_with_dynamic_identity() { u64::from(raw.id.0), InputContract::bounded(raw_schema.clone()), )]), - &[u64::from(logical.root.0)], + &[u64::from(logical.roots[0].0)], ) .unwrap(); let bytes = String::from_utf8(serde_json::to_vec(&physical).unwrap()).unwrap(); diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/executor/tests/deployment.rs similarity index 84% rename from crates/asap-physical-operators/tests/deployment.rs rename to crates/executor/tests/deployment.rs index 2857a81cb..270c8d6fe 100644 --- a/crates/asap-physical-operators/tests/deployment.rs +++ b/crates/executor/tests/deployment.rs @@ -1,11 +1,11 @@ //! Exercise the public library without a backend server, store, or scheduler. -use asap_physical_operators::planner::{ - post_asap::{ +use asap_executor::planner::ir::{ + scalar::ColumnRef, + schema::{ FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate, }, - pre_asap::ColumnRef, }; -use asap_physical_operators::{factory::create_planner_accumulator, AggregateCore}; +use asap_executor::{factory::create_planner_accumulator, AggregateCore}; fn family(k: u32) -> FieldDataType { FieldDataType::Sketch( @@ -28,14 +28,12 @@ fn build(values: &[f64]) -> Box { } fn read(state: &dyn AggregateCore) -> f64 { state - .estimate( - &asap_physical_operators::planner::post_asap::SketchStatistic::Quantile { q: 0.5 }, - ) + .estimate(&asap_executor::planner::ir::schema::SketchStatistic::Quantile { q: 0.5 }) .unwrap() } // The same kernels work when every build is query-time, when only a prefix -// was precomputed, and when all state was precomputed before the readout. +// was precomputed, and when all state was precomputed before the evaluation. #[test] fn raw_partial_and_fully_precomputed_use_the_same_kernels() { let raw: Vec = (0..128).map(f64::from).collect(); @@ -64,8 +62,8 @@ fn invalid_kll_parameters_are_rejected_at_binding() { // a packed-wire column-bit budget must not be imposed on this constructor. #[test] fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { - use asap_physical_operators::planner::post_asap::SummaryInputExpr; - use asap_physical_operators::KeyByLabelValues; + use asap_executor::planner::ir::schema::SummaryInputExpr; + use asap_executor::KeyByLabelValues; let family = FieldDataType::Sketch( SketchKind::new( SketchAlgorithm::CountSketchWithHeap, @@ -85,7 +83,7 @@ fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { let state = operator.into_accumulator(); let state = state .as_any() - .downcast_ref::() + .downcast_ref::() .unwrap(); assert_eq!(state.query_key(&key), 7.0); } diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/executor/tests/deployment_computation.rs similarity index 76% rename from crates/asap-physical-operators/tests/deployment_computation.rs rename to crates/executor/tests/deployment_computation.rs index e20fc27ee..96164c74c 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/executor/tests/deployment_computation.rs @@ -1,21 +1,23 @@ //! Planner-selected PromQL computation compiles from the timed DAG alone; //! the deployment supplies only raw rows at the ingestion frontier. -use asap_physical_operators::{ +mod common; +use asap_executor::{ operators::Operator, physical_planner::{compile, promql_rows, CompiledPhysicalDAG, InputContract, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_physical_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; -use planner_types::{post_asap::*, pre_asap::QueryExpr, types::AccuracyTarget, workload::*}; +use planner_types::ir::export::{PhysicalASAPDAG, PhysicalASAPOperatorPayload}; +use planner_types::{ir::schema::*, types::AccuracyTarget, workload::*}; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { lower_with(query, AccuracyTarget::Exact) } -fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn lower_with(query: &str, accuracy: AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -46,54 +48,59 @@ fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { } /// The first exact summary candidate, as Planner selection would hand it over. -fn exact_dag(query: &str) -> PostAsapDAG { - use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; +fn exact_dag(query: &str) -> PhysicalASAPDAG { let expression = lower(query); - let root = Rc::new(promql_rows::with_series_identity(&expression).unwrap_or(expression)); - asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) - .replacements(&TargetSubDAG::new(&root)) - .into_iter() - .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; - dag.nodes - .iter() - .all(|n| !matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { family, .. } if !matches!(family, FieldDataType::ExactAggregate(..)))) - .then_some(dag) - } - _ => None, - }) - .unwrap() -} - -fn population_dag(query: &str) -> PostAsapDAG { - let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); - let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( - std::slice::from_ref(&root), + let root = promql_rows::with_series_identity(&expression).unwrap_or(expression); + let space = asap_logical_optimizer::search_workload(vec![("q", root)]); + let selected = asap_plan_selection::candidate_selection::global_selection( + &space, + &asap_plan_selection::DefaultCostModel, ) - .candidate(&root) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() .unwrap(); - compile_post_asap_dag(&selected).unwrap() + compile_physical_asap_dag(&selected).unwrap() +} + +fn population_dag(query: &str) -> PhysicalASAPDAG { + let root = promql_rows::with_series_identity(&lower(query)).unwrap(); + let selected = + asap_logical_optimizer::pass1::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + compile_physical_asap_dag(&selected).unwrap() } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &PhysicalASAPDAG) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { child, .. }, - } => match child.as_ref() { - QueryExpr::Scan { - source: planner_types::pre_asap::Source::TimeSeries { metric }, - .. - } => Some(( - u64::from(node.id.0), - Arc::new(node.output_schema.clone()), - metric.clone(), - )), - _ => None, - }, + PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. }, + } => { + let mut id = node.id; + loop { + let n = dag.nodes.iter().find(|n| n.id == id)?; + if let PhysicalASAPOperatorPayload::Relational { + operator: + planner_types::ir::export::NonASAPOpKind::Scan { + source: planner_types::ir::operator::Source::TimeSeries { metric }, + .. + }, + } = &n.payload + { + return Some(( + u64::from(node.id.0), + Arc::new(node.output_schema.clone()), + metric.clone(), + )); + } + id = dag.edges.iter().find(|e| e.consumer == id)?.producer; + } + } _ => None, }) .collect() @@ -104,21 +111,21 @@ type Sample = (&'static str, &'static str, &'static str, i64, f64); /// Compile, round-trip, bind raw `(metric, job, instance, ts, value)` samples, /// and return the root's batches. fn execute( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, samples: &[Sample], end: i64, -) -> Result>, String> { +) -> Result>, String> { execute_relabeled(dag, samples, end, &BTreeMap::new()) } /// [`execute`], supplying samples of each instance in `relabel` under its /// `(__name__, instance)` instead. fn execute_relabeled( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, samples: &[Sample], end: i64, relabel: &BTreeMap<&str, (&str, &str)>, -) -> Result>, String> { +) -> Result>, String> { let inputs = raw_inputs(dag); let program = compile( dag, @@ -126,7 +133,7 @@ fn execute_relabeled( .iter() .map(|(id, schema, _)| (*id, InputContract::bounded(schema.clone()))) .collect(), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], ) .map_err(|e| e.to_string())?; let program: CompiledPhysicalDAG = @@ -195,7 +202,11 @@ fn execute_relabeled( } /// [`execute`], returning `(job, value)` rows of the root. -fn run(dag: &PostAsapDAG, samples: &[Sample], end: i64) -> Result, String> { +fn run( + dag: &PhysicalASAPDAG, + samples: &[Sample], + end: i64, +) -> Result, String> { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let job = batch.schema().fields.iter().position(|f| f.name == "job"); @@ -250,7 +261,7 @@ fn population_aggregates_match_current_series_reference() { } } -// A global readout of an empty population is an empty vector, as in PromQL. +// A global evaluation of an empty population is an empty vector, as in PromQL. #[test] fn global_population_aggregate_of_no_members_is_empty() { // Latest values are [1, 2, 5, 7] at 60s; every member has expired by 1000s. @@ -313,40 +324,7 @@ fn grouped_vector_arithmetic_matches_labels() { // then roll up per job: api has 2 + 1 + 1 samples in 5m, db has 1. #[test] fn exact_count_finalizes_to_declared_float_value() { - let mut dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); - let finalize = dag - .nodes - .iter() - .find(|node| { - matches!( - node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } - ) - }) - .unwrap() - .clone(); - let root = dag.nodes.iter().find(|n| n.id == dag.root).unwrap().clone(); - let mut edge = dag - .edges - .iter() - .find(|e| e.producer == finalize.id) - .unwrap() - .clone(); - // Read the rolled-up exact state the same way the query path does. - let mut read = finalize.clone(); - read.id = PostAsapNodeId(root.id.0 + 1); - read.output_schema = root.output_schema.clone(); - read.output_schema.fields.last_mut().unwrap().dtype = - FieldDataType::Plain(planner_types::pre_asap::DataType::Float64); - edge.producer = root.id; - edge.consumer = read.id; - edge.intermediate_schema = root.output_schema.clone(); - edge.data_state = root.output_state; - dag.root = read.id; - dag.nodes.push(read); - dag.edges.push(edge); + let dag = exact_dag("sum by (job) (count_over_time(m[5m]))"); assert_eq!( run(&dag, SAMPLES, 60_000).unwrap(), reference(&[("api", 4.), ("db", 1.)]) @@ -354,10 +332,22 @@ fn exact_count_finalizes_to_declared_float_value() { } /// `dag` with its Binary operator replaced by `kind`. -fn with_kind(mut dag: PostAsapDAG, kind: planner_types::pre_asap::BinaryOpKind) -> PostAsapDAG { +fn with_kind( + mut dag: PhysicalASAPDAG, + kind: planner_types::ir::operator::BinaryOpKind, + bool_result: bool, +) -> PhysicalASAPDAG { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { + if let PhysicalASAPOperatorPayload::Relational { + operator: + planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + } = &mut node.payload + { operator.kind = kind.clone(); + *return_bool = bool_result; } } dag @@ -367,30 +357,31 @@ fn with_kind(mut dag: PostAsapDAG, kind: planner_types::pre_asap::BinaryOpKind) // holds, with their value, on either side of the literal; `bool` yields 1 or 0. #[test] fn grouped_comparisons_filter_or_return_bool() { - use planner_types::pre_asap::{BinaryOpKind::*, CompareOpKind::Gt}; - // sum_over_time over 5m per job: api = 14, db = 5. - let right = exact_dag("sum by (job) (sum_over_time(m[5m])) * 10"); - let left = exact_dag("10 - sum by (job) (sum_over_time(m[5m]))"); - for (dag, expected) in [ + for (query, expected) in [ ( - with_kind(right.clone(), Compare(Gt)), + "sum by(job)(sum_over_time(m[5m])) > 10", reference(&[("api", 14.)]), ), ( - with_kind(right, CompareBool(Gt)), + "sum by(job)(sum_over_time(m[5m])) > bool 10", reference(&[("api", 1.), ("db", 0.)]), ), - (with_kind(left, Compare(Gt)), reference(&[("db", 5.)])), + ( + "10 > sum by(job)(sum_over_time(m[5m]))", + reference(&[("db", 5.)]), + ), ] { - assert_eq!(run(&dag, SAMPLES, 60_000).unwrap(), expected); + assert_eq!(run(&exact_dag(query), SAMPLES, 60_000).unwrap(), expected); } } -// A `bool` comparison Binary over per-series readouts matches one-to-one and +// A `bool` comparison Binary over per-series evaluations matches one-to-one and // drops the metric name; a filter keeps the surviving left value. #[test] fn per_series_comparisons_filter_or_return_bool() { - use planner_types::pre_asap::{BinaryOpKind::*, CompareOpKind::*}; + use planner_types::ir::operator::BinaryOpKind::*; + use planner_types::ir::scalar::CompareOpKind::*; + let samples = counter("a", "api", 10., 10.) .chain(counter("a", "db", 10., 10.)) .chain(counter("b", "api", 5., 5.)) @@ -399,18 +390,23 @@ fn per_series_comparisons_filter_or_return_bool() { // rate: a{api} = a{db} = 50/300, b{api} = 25/300, b{db} = 100/300. let dag = exact_dag("rate(a[5m]) / rate(b[5m])"); assert_eq!( - run_series(&with_kind(dag.clone(), Compare(Gt)), &samples, 300_000).unwrap(), + run_series( + &with_kind(dag.clone(), Compare(Gt), false), + &samples, + 300_000 + ) + .unwrap(), series(&[("api", "x", 50. / 300.)]) ); assert_eq!( - run_series(&with_kind(dag, CompareBool(Lt)), &samples, 300_000).unwrap(), + run_series(&with_kind(dag, Compare(Lt), true), &samples, 300_000).unwrap(), series(&[("api", "x", 0.), ("db", "x", 1.)]) ); } /// [`execute`], returning per-series `(identity, value)` rows of the root, /// with NaN-aware formatting for comparison. -fn run_series(dag: &PostAsapDAG, samples: &[Sample], end: i64) -> Result { +fn run_series(dag: &PhysicalASAPDAG, samples: &[Sample], end: i64) -> Result { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let schema = batch.schema(); @@ -509,13 +505,13 @@ fn per_series_rate_ratio_matches_prometheus() { // A literal operand applies to every stored per-series value, on either side, // and drops the metric name. #[test] -fn per_series_scalar_arithmetic_applies_to_stored_readouts() { +fn per_series_scalar_arithmetic_applies_to_stored_evaluations() { let samples = counter("m", "api", 10., 10.).collect::>(); // rate = 40 * 1.25 / 300 = 1/6. for (query, expected) in [ ("rate(m[5m]) * 2", 50. / 300. * 2.), ("1 - rate(m[5m])", 1. - 50. / 300.), - // The stored sum readout keeps `__name__`; the arithmetic drops it. + // The stored sum evaluation keeps `__name__`; the arithmetic drops it. ("sum_over_time(m[5m]) * 2", 150. * 2.), ] { assert_eq!( @@ -545,13 +541,20 @@ fn per_series_scalar_arithmetic_rejects_label_sets_equal_without_the_name() { } fn with_vector_match( - mut dag: PostAsapDAG, - kind: planner_types::pre_asap::VectorMatchKind, + mut dag: PhysicalASAPDAG, + kind: planner_types::ir::operator::VectorMatchKind, labels: &[&str], -) -> PostAsapDAG { +) -> PhysicalASAPDAG { for node in &mut dag.nodes { - if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { - operator.vector_match = Some(planner_types::pre_asap::VectorMatch { + if let PhysicalASAPOperatorPayload::Relational { + operator: + planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator, + return_bool: _, + }, + } = &mut node.payload + { + operator.vector_match = Some(planner_types::ir::operator::VectorMatch { kind: kind.clone(), labels: labels.iter().map(|l| l.to_string()).collect(), grouping: None, @@ -565,7 +568,7 @@ fn with_vector_match( // the result's labels; a duplicate match group on either side is an error. #[test] fn per_series_vector_matching_follows_on_and_ignoring() { - use planner_types::pre_asap::VectorMatchKind; + use planner_types::ir::operator::VectorMatchKind; let samples = counter("a", "api", 10., 10.) .chain(counter("b", "api", 5., 5.)) .chain(counter("b", "db", 5., 5.)) @@ -621,44 +624,43 @@ fn population_sums_and_averages_are_compensated() { } } -// A bare count over stored Count-Min state compiles to a Planner readout that +// A bare count over stored Count-Min state compiles to a Planner evaluation that // returns the sketch's total update weight, including colliding items. #[test] -fn stored_count_min_bare_count_compiles_to_a_readout() { - use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; - use asap_physical_operators::summary_kernels::CountMinSketchAccumulator; - let root = Rc::new(lower_with("count(up)", AccuracyTarget::Epsilon(0.02))); - let dag = - asap_aware_mapping::SketchAlgorithmStrategy::new(&asap_aware_mapping::DefaultCostModel) - .replacements(&TargetSubDAG::new(&root)) - .into_iter() - .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => { - let dag = compile_post_asap_dag(&node).ok()?; - let bare_count = dag.nodes.iter().any(|n| { - matches!( - &n.payload, - PostAsapOperatorPayload::SummaryEstimate { - query: SketchStatistic::PointCount { value: None, .. } - } - ) - }); - let count_min = dag.nodes.iter().any(|n| { - matches!(&n.payload, PostAsapOperatorPayload::SummaryAgg { +fn stored_count_min_bare_count_compiles_to_a_evaluation() { + use asap_executor::summary_kernels::CountMinSketchAccumulator; + use asap_logical_optimizer::{Replacement, ReplacementStrategy, TargetSubDAG}; + let root = lower_with("count(up)", AccuracyTarget::Epsilon(0.02)); + let dag = asap_logical_optimizer::ASAPStrategies::default() + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::SubDAG(node) => { + let dag = compile_physical_asap_dag(&node).ok()?; + let bare_count = dag.nodes.iter().any(|n| { + matches!( + &n.payload, + PhysicalASAPOperatorPayload::SummaryEstimate { + query: SketchStatistic::PointCount { value: None, .. } + } + ) + }); + let count_min = dag.nodes.iter().any(|n| { + matches!(&n.payload, PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) - }); - (bare_count && count_min).then_some(dag) - } - _ => None, - }) - .expect("Planner lists a Count-Min candidate for count(up)"); + }); + (bare_count && count_min).then_some(dag) + } + _ => None, + }) + .expect("Planner lists a Count-Min candidate for count(up)"); let state = dag .nodes .iter() - .find(|n| matches!(n.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|n| matches!(n.payload, PhysicalASAPOperatorPayload::SummaryAgg { .. })) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } = &state.payload @@ -675,7 +677,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { u64::from(state.id.0), InputContract::bounded(schema.clone()), )]), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], ) .unwrap(); let program: CompiledPhysicalDAG = diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/executor/tests/physical_dag.rs similarity index 76% rename from crates/asap-physical-operators/tests/physical_dag.rs rename to crates/executor/tests/physical_dag.rs index e2ad735d6..82659f2b3 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/executor/tests/physical_dag.rs @@ -1,5 +1,5 @@ //! Acceptance tests use the library directly, without either backend engine. -use asap_physical_operators::{ +use asap_executor::{ dag::{ operators::{Expression, Operator, Reduction, SortKey}, values::{Batch, SchemaRef, Value}, @@ -8,15 +8,19 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; -use planner_types::{ - post_asap::{ExactKind, ExactParams, Field, FieldDataType, Schema}, - pre_asap::DataType, +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, }; +use planner_types::ir::schema::{DataType, ExactKind, ExactParams, Field, FieldDataType}; +use planner_types::ir::BinaryOperator; +use planner_types::ir::ScalarExpr; use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: fields .iter() .map(|(name, dtype, nullable)| Field { @@ -117,7 +121,7 @@ fn grouped_sort_limit_across_batches() { // The same computation runs in either engine scope with fresh per-run state. #[test] -fn summary_construction_merge_and_readout_at_both_phases() { +fn summary_construction_merge_and_evaluation_at_both_phases() { let schema = schema(&[("v", DataType::Float64, false)]); let batches = (1..=20) .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) @@ -140,11 +144,11 @@ fn summary_construction_merge_and_readout_at_both_phases() { dag.add( 4, vec![3], - Operator::readout( + Operator::evaluation( state, 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_executor::operators::SummaryEvaluation::Exact( + asap_executor::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Sum, lookback_ms: None, }, @@ -295,11 +299,11 @@ fn binding_rejects_unsupported_operations() { vec![], ) .unwrap(); - assert!(Operator::readout( + assert!(Operator::evaluation( sum.schema(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( - planner_types::post_asap::SketchStatistic::Quantile { q: 0.5 } + asap_executor::operators::SummaryEvaluation::Sketch( + planner_types::ir::schema::SketchStatistic::Quantile { q: 0.5 } ) ) .is_err()); @@ -309,7 +313,8 @@ fn binding_rejects_unsupported_operations() { // KLL is one family example: precomputation changes input sources, not operators. #[test] fn kll_raw_partial_and_precomputed_are_native_dags() { - use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + use planner_types::ir::schema::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("value", DataType::Float64, false)]); let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), @@ -396,11 +401,11 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { dag.add( 5, vec![4], - Operator::readout( + Operator::evaluation( state.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( - planner_types::post_asap::SketchStatistic::Quantile { q: 0.5 }, + asap_executor::operators::SummaryEvaluation::Sketch( + planner_types::ir::schema::SketchStatistic::Quantile { q: 0.5 }, ), ) .unwrap(), @@ -419,13 +424,13 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { // Exact state must match its declared family; a mislabeled state is rejected. #[test] fn exact_state_and_family_validation() { - use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + use asap_executor::summary_kernels::exact::ExactAccumulator; let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); acc.update(None, 7., 0); - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "state".into(), @@ -452,11 +457,11 @@ fn exact_state_and_family_validation() { dag.add( 1, vec![0], - Operator::readout( + Operator::evaluation( schema.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_executor::operators::SummaryEvaluation::Exact( + asap_executor::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Sum, lookback_ms: None, }, @@ -484,42 +489,49 @@ fn exact_state_and_family_validation() { // Planner binding rejects unknown computation instead of accepting a fallback. #[test] fn bind_post_asap_before_execution() { - use asap_physical_operators::dag::planner::bind; - use planner_types::{ - post_asap::{ - EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDAG, PostAsapDAGEdge, - PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, - WindowEdgeCompatibility, - }, - pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, - }; - use std::{collections::BTreeMap, rc::Rc}; + use asap_executor::dag::planner::bind; + use planner_types::ir::properties::ExecutionDataState; + use planner_types::ir::scalar::{ArithmeticOpKind, ScalarValue}; + use std::collections::BTreeMap; let schema = schema(&[("value", DataType::Float64, false)]); - let node = |id, payload| PostAsapDAGNode { - id: PostAsapNodeId(id), + let node = |id, payload| PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(id), payload, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, }; - let mut dag = PostAsapDAG { + let mut dag = PhysicalASAPDAG { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::ir::scalar::ScalarValue::Float64(1.), + )]], + schema: (*schema).clone(), + }, }, ), node( 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Project { - cols: vec![ProjectItem { + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Project { + cols: vec![planner_types::ir::export::WireProjectItem { alias: None, - expr: QueryExpr::Arithmetic { + expr: planner_types::ir::export::WireScalarExpr::Arithmetic { + semantics: planner_types::ir::ExprSemantics::Sql, op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(0)), - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), + left: Box::new(planner_types::ir::export::WireScalarExpr::Column( + 0, + )), + right: Box::new( + planner_types::ir::export::WireScalarExpr::Literal( + ScalarValue::Float64(2.), + ), + ), }, }], qualifier: None, @@ -527,18 +539,18 @@ fn bind_post_asap_before_execution() { }, ), ], - edges: vec![PostAsapDAGEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + edges: vec![PhysicalASAPDAGEdge { + producer: planner_types::ir::export::LogicalASAPNodeId(0), + consumer: planner_types::ir::export::LogicalASAPNodeId(1), role: EdgeRole::Input, intermediate_schema: (*schema).clone(), data_state: ExecutionDataState::QUERY_ROWS, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }], - root: PostAsapNodeId(1), + roots: vec![planner_types::ir::export::LogicalASAPNodeId(1)], }; - let sources = || -> BTreeMap> { + let sources = || -> BTreeMap> { BTreeMap::from([( 0, Box::new( @@ -547,7 +559,7 @@ fn bind_post_asap_before_execution() { vec![Batch::try_new(schema.clone(), vec![vec![Value::Float64(1.)]]).unwrap()], ) .unwrap(), - ) as asap_physical_operators::dag::planner::Source<'static>, + ) as asap_executor::dag::planner::Source<'static>, )]) }; let native = bind(&dag, sources(), &[1]).unwrap(); @@ -555,17 +567,15 @@ fn bind_post_asap_before_execution() { // A literal Fallback needs no deployment input. let literal = bind(&dag, BTreeMap::new(), &[1]).unwrap(); assert_eq!(floats(&run(&literal, 1, query()), 0), vec![3.]); - dag.nodes[1].payload = PostAsapOperatorPayload::Value { - operation: ValueOperation::Extension { - name: "unknown".into(), - }, + dag.nodes[1].payload = PhysicalASAPOperatorPayload::Extension { + name: "unsupported".into(), }; assert!(bind(&dag, sources(), &[1]).is_err()); } // A completed empty population has an exact zero count, with integer output. #[test] -fn empty_exact_count_is_an_integer_state_readout() { +fn empty_exact_count_is_an_integer_state_evaluation() { let input = schema(&[("value", DataType::Float64, false)]); let build = Operator::summary_build( input.clone(), @@ -575,11 +585,11 @@ fn empty_exact_count_is_an_integer_state_readout() { vec![], ) .unwrap(); - let read = Operator::readout( + let read = Operator::evaluation( build.schema(), 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_executor::operators::SummaryEvaluation::Exact( + asap_executor::summary_kernels::exact::ExactEvaluation { statistic: Statistic::Count, lookback_ms: None, }, @@ -597,14 +607,8 @@ fn empty_exact_count_is_an_integer_state_readout() { // A deployment source cannot pass a different row shape to bound expressions. #[test] fn source_batches_must_match_the_bound_schema() { - use asap_physical_operators::dag::{self, PhysicalOperator}; - use planner_types::{ - post_asap::{ - ExecutionDataState, PostAsapDAG, PostAsapDAGNode, PostAsapNodeId, - PostAsapOperatorPayload, - }, - pre_asap::QueryExpr, - }; + use asap_executor::dag::{self, PhysicalOperator}; + use planner_types::ir::properties::ExecutionDataState; use std::{cell::Cell, collections::BTreeMap, rc::Rc}; struct WrongSource { schema: SchemaRef, @@ -637,18 +641,24 @@ fn source_batches_must_match_the_bound_schema() { } let expected = schema(&[("value", DataType::Float64, false)]); let starts = Rc::new(Cell::new(0)); - let plan = PostAsapDAG { - nodes: vec![PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(1.), + let plan = PhysicalASAPDAG { + nodes: vec![PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(0), + payload: PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::ir::scalar::ScalarValue::Float64(1.), + )]], + schema: (*expected).clone(), + }, }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*expected).clone(), guarantee: None, }], edges: vec![], - root: PostAsapNodeId(0), + roots: vec![planner_types::ir::export::LogicalASAPNodeId(0)], }; let source = Box::new(WrongSource { schema: expected, @@ -706,72 +716,90 @@ fn extrema_preserve_numeric_values_in_the_presence_of_nan() { // Planner wire nodes, including grouping and edge roles, are executable at either phase. #[test] fn planner_semijoin_sort_limit_contract_at_both_phases() { - use asap_physical_operators::dag::planner::{bind, Source}; - use planner_types::{ - post_asap::*, - pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, - }; - use std::{collections::BTreeMap, rc::Rc}; + use asap_executor::dag::planner::{bind, Source}; + use planner_types::ir::operator::{GroupKeys, JoinKind}; + use planner_types::ir::properties::*; + use planner_types::ir::scalar::CompareOpKind; + use planner_types::ir::schema::*; + use std::collections::BTreeMap; let rows_schema = schema(&[ ("group", DataType::Utf8, false), ("key", DataType::Utf8, false), ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &asap_physical_operators::values::SchemaRef| PostAsapDAGNode { - id: PostAsapNodeId(id), + let node = |id, payload, schema: &asap_executor::values::SchemaRef| PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(id), payload, output_schema: (**schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, }; - let edge = |producer, consumer, role, schema: &asap_physical_operators::values::SchemaRef| { - PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + let edge = + |producer, consumer, role, schema: &asap_executor::values::SchemaRef| PhysicalASAPDAGEdge { + producer: planner_types::ir::export::LogicalASAPNodeId(producer), + consumer: planner_types::ir::export::LogicalASAPNodeId(consumer), role, intermediate_schema: (**schema).clone(), data_state: ExecutionDataState::QUERY_ROWS, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, - } - }; + }; let groups = GroupKeys::by(vec![0]); - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes: vec![ node( 0, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::ir::scalar::ScalarValue::Float64(0.), + )]], + schema: (*rows_schema).clone(), + }, }, &rows_schema, ), node( 1, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::promql_scalar(0.), + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Values { + rows: vec![vec![planner_types::ir::export::WireScalarExpr::Literal( + planner_types::ir::scalar::ScalarValue::Float64(0.), + )]], + schema: (*keys_schema).clone(), + }, }, &keys_schema, ), node( 2, - PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pruning: None, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(1)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(3)), - })), + PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::Join { + join_kind: JoinKind::Semi, + pred: planner_types::ir::export::WirePredicate( + planner_types::ir::export::WireScalarExpr::Compare { + left: Box::new(planner_types::ir::export::WireScalarExpr::Column( + 1, + )), + op: CompareOpKind::Eq, + right: Box::new(planner_types::ir::export::WireScalarExpr::Column( + 3, + )), + semantics: planner_types::ir::ExprSemantics::Sql, + }, + ), + }, }, &rows_schema, ), node( 3, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Sort { - keys: vec![SortKey { - expr: QueryExpr::Column(2), + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Sort { + keys: vec![planner_types::ir::export::WireSortKey { + expr: planner_types::ir::export::WireScalarExpr::Column(2), ascending: false, nulls_first: false, }], @@ -782,9 +810,9 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ), node( 4, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Limit { + n: Some(1), offset: 0, partition_by: groups, }, @@ -799,7 +827,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { edge(2, 3, EdgeRole::Input, &rows_schema), edge(3, 4, EdgeRole::Input, &rows_schema), ], - root: PostAsapNodeId(4), + roots: vec![planner_types::ir::export::LogicalASAPNodeId(4)], }; let text = |v: &str| Value::Utf8(v.into()); for (phase, scope) in [ @@ -861,9 +889,9 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { // Planner scalar signatures, collection access and null predicates share native execution. #[test] fn planner_expressions_preserve_collection_and_nullable_types() { - use asap_physical_operators::dag::expressions::CompiledExpression; - use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; - use std::rc::Rc; + use asap_executor::dag::expressions::CompiledExpression; + use planner_types::ir::scalar::{CompareOpKind, ScalarValue}; + let input_schema = schema(&[( "items", DataType::Map { @@ -873,11 +901,11 @@ fn planner_expressions_preserve_collection_and_nullable_types() { }, false, )]); - let access = QueryExpr::FunctionCall { + let access = ScalarExpr::FunctionCall { name: "asap_element_access".into(), args: vec![ - QueryExpr::Column(0), - QueryExpr::Literal(ScalarValue::Utf8("count".into())), + ScalarExpr::Column(0), + ScalarExpr::Literal(ScalarValue::Utf8("count".into())), ], }; let project = Operator::project( @@ -910,10 +938,11 @@ fn planner_expressions_preserve_collection_and_nullable_types() { .unwrap(); let projected = project.schema(); dag.add(1, vec![0], project).unwrap(); - let predicate = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let predicate = ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Ge, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(1))), }; dag.add( 2, @@ -927,9 +956,9 @@ fn planner_expressions_preserve_collection_and_nullable_types() { .unwrap(); let rows = run(&dag, 2, query()); assert!(matches!(rows.as_slice(),[row] if matches!(row.as_slice(),[Value::Int64(7)]))); - let unknown = QueryExpr::FunctionCall { + let unknown = ScalarExpr::FunctionCall { name: "unregistered_function".into(), - args: vec![QueryExpr::Column(0)], + args: vec![ScalarExpr::Column(0)], }; assert!(CompiledExpression::compile(&unknown, &input_schema).is_err()); } @@ -937,14 +966,17 @@ fn planner_expressions_preserve_collection_and_nullable_types() { // Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. #[test] fn native_relational_join_kinds_preserve_unmatched_rows() { - use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; - use std::rc::Rc; + use planner_types::ir::operator::JoinKind; + use planner_types::ir::scalar::CompareOpKind; + use planner_types::ir::Predicate; + let input = schema(&[("key", DataType::Int64, true)]); - let predicate = Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let predicate = Predicate(ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })); + right: Box::new(ScalarExpr::Column(1)), + }); for (kind, count) in [ (JoinKind::Inner, 1), (JoinKind::Left, 3), @@ -1017,7 +1049,8 @@ fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scop } } fn assert_weighted_rate_topk(count_sketch: bool) { - use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + use planner_types::ir::schema::{SketchAlgorithm, SketchKind, SketchParams}; + let raw = schema(&[ ("service", DataType::Utf8, false), ("job", DataType::Utf8, false), @@ -1048,7 +1081,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { } let rates = Operator::window( raw.clone(), - planner_types::pre_asap::AggIntent::Rate, + planner_types::ir::operator::AggIntent::Rate, 3, 4, vec![0, 1, 2], @@ -1084,7 +1117,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { ("service", DataType::Utf8, false), ("score", DataType::Float64, false), ]); - let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); + let evaluation = Operator::keyed_evaluation(build.schema(), 1, 8, output.clone()).unwrap(); let mut dag = PhysicalDAG::default(); dag.add( 0, @@ -1094,7 +1127,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { .unwrap(); dag.add(1, vec![0], rates).unwrap(); dag.add(2, vec![1], build).unwrap(); - dag.add(3, vec![2], readout).unwrap(); + dag.add(3, vec![2], evaluation).unwrap(); dag.add( 4, vec![3], @@ -1138,20 +1171,18 @@ fn assert_weighted_rate_topk(count_sketch: bool) { // The grouped temporal reducer's sample schema must survive physical Sort/Limit binding. #[test] fn grouped_temporal_schema_compiles_and_executes_topk() { - use asap_physical_operators::physical_planner::{ + use asap_executor::physical_planner::{ compile_node, CompiledPhysicalDAG, InputContract, Source, }; - use planner_types::post_asap::{ - ExecutionDataState, PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, - ValueOperation, - }; - use planner_types::pre_asap::{ - aggregate_output_schema, AggIntent, Field, GroupKeys, QueryExpr, Reduction as IrReduction, - Schema as IrSchema, - }; + use planner_types::ir::export::{PhysicalASAPDAGNode, PhysicalASAPOperatorPayload}; + use planner_types::ir::properties::ExecutionDataState; + + use planner_types::ir::operator::{AggIntent, GroupKeys, Reduction as IrReduction}; + use planner_types::ir::schema::{aggregate_output_schema, Schema as IrSchema}; + let grouped = IrSchema::new(vec![ - Field::plain("job", DataType::Utf8, false), - Field::plain("sum", DataType::Float64, false), + planner_types::ir::schema::Field::plain("job", DataType::Utf8, false), + planner_types::ir::schema::Field::plain("sum", DataType::Float64, false), ]); let output = aggregate_output_schema( &grouped, @@ -1167,15 +1198,18 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { .map(|c| { ( c.name.as_str(), - c.dtype.plain().unwrap().clone(), + c.plain_dtype().unwrap().clone(), c.nullable, ) }) .collect::>(), ); - let node = |id, operation| PostAsapDAGNode { - id: PostAsapNodeId(id), - payload: PostAsapOperatorPayload::Value { operation }, + let node = |id, operation| PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(id), + payload: PhysicalASAPOperatorPayload::Relational { + operator: operation, + }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*input).clone(), guarantee: None, @@ -1184,8 +1218,8 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { &node( 1, ValueOperation::Sort { - keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(1), + keys: vec![planner_types::ir::export::WireSortKey { + expr: planner_types::ir::export::WireScalarExpr::Column(1), ascending: false, nulls_first: false, }], @@ -1199,7 +1233,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { &node( 2, ValueOperation::Limit { - n: 1, + n: Some(1), offset: 0, partition_by: GroupKeys::none(), }, @@ -1248,36 +1282,34 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { // A certified candidate set must have authoritative values for every key, including after recovery. #[test] fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { - use asap_physical_operators::physical_planner::{ + use asap_executor::physical_planner::{ compile_node, CompiledPhysicalDAG, InputContract, Source, }; - use planner_types::{ - post_asap::*, - pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}, - }; - use std::{collections::BTreeMap, rc::Rc}; + use planner_types::ir::operator::JoinKind; + use planner_types::ir::properties::*; + use planner_types::ir::scalar::CompareOpKind; + use planner_types::ir::schema::*; + use std::collections::BTreeMap; let schema = schema(&[("key", DataType::Utf8, false)]); for certified in [false, true] { - let node = PostAsapDAGNode { - id: PostAsapNodeId(2), + let node = PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(2), output_schema: (*schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })), - pruning: certified.then_some(CandidateCompleteness::Certified { - guarantee: ResultGuarantee { - metric: ErrorMetric::TopKMembership, - bound: BoundExpr::Zero, - failure_probability: ProbabilityExpr::Constant { value: 0.01 }, - provenance: vec![], - }, - }), + payload: PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::Join { + join_kind: JoinKind::Semi, + pred: planner_types::ir::export::WirePredicate( + planner_types::ir::export::WireScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(planner_types::ir::export::WireScalarExpr::Column(0)), + op: CompareOpKind::Eq, + right: Box::new(planner_types::ir::export::WireScalarExpr::Column(1)), + }, + ), + }, }, }; let dag = CompiledPhysicalDAG::from_operators( @@ -1290,7 +1322,12 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { 2, ( vec![0, 1], - compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(), + if certified { + Operator::certified_semi_join(schema.clone(), schema.clone(), vec![(0, 0)]) + .unwrap() + } else { + compile_node(&node, &[schema.clone(), schema.clone()]).unwrap() + }, ), )] .into(), @@ -1341,7 +1378,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { while let Some(batch) = stream.next().await { rows.extend(batch?.rows().iter().cloned()); } - Ok::<_, asap_physical_operators::Error>(rows) + Ok::<_, asap_executor::Error>(rows) }); if certified && !complete { assert!(result @@ -1360,30 +1397,34 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { // Precompute arithmetic must match population/window identities, never zip arrival order. #[test] fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { - use asap_physical_operators::physical_planner::{ + use asap_executor::physical_planner::{ compile_node, CompiledPhysicalDAG, InputContract, Source, }; - use planner_types::{ - post_asap::*, - pre_asap::{ArithmeticOpKind, BinaryOpKind}, - }; + use planner_types::ir::operator::BinaryOpKind; + use planner_types::ir::properties::*; + use planner_types::ir::scalar::ArithmeticOpKind; + use planner_types::ir::schema::*; use std::collections::BTreeMap; let input = schema(&[ ("population", DataType::Utf8, false), ("time", DataType::Timestamp, false), ("value", DataType::Float64, false), ]); - let node = PostAsapDAGNode { - id: PostAsapNodeId(2), + let node = PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(2), output_schema: (*input).clone(), output_state: ExecutionDataState::INGESTION_ROWS, guarantee: None, - payload: PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, + payload: PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, }, }, }; @@ -1448,7 +1489,7 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { while let Some(batch) = stream.next().await { rows.extend(batch?.rows().iter().cloned()); } - Ok::<_, asap_physical_operators::Error>(rows) + Ok::<_, asap_executor::Error>(rows) }); match expected { Some(values) => assert_eq!(floats(&result.unwrap(), 2), values), diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/executor/tests/physical_plan_recovery.rs similarity index 87% rename from crates/asap-physical-operators/tests/physical_plan_recovery.rs rename to crates/executor/tests/physical_plan_recovery.rs index fda1b502a..68a6e4672 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/executor/tests/physical_plan_recovery.rs @@ -1,19 +1,16 @@ //! Deserialized physical plans recover selected operators without logical lowering. //! Deployments choose the encoding; JSON is used here only as a test format. -use asap_physical_operators::{ +use asap_executor::{ operators::{Operator, SortKey}, physical_planner::{CompiledPhysicalDAG, InputContract}, }; -use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::DataType, -}; +use planner_types::ir::schema::{DataType, Field, FieldDataType}; use std::{collections::BTreeMap, sync::Arc}; fn sorted() -> CompiledPhysicalDAG { - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -74,7 +71,7 @@ fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { #[test] fn candidate_recovery_preserves_materialization_boundary() { - use asap_physical_operators::physical_planner::PhysicalASAPDAG; + use asap_executor::physical_planner::CompiledPhysicalPlan; let precompute = sorted(); let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); let query = CompiledPhysicalDAG::from_operators( @@ -89,13 +86,13 @@ fn candidate_recovery_preserves_materialization_boundary() { vec![2], ) .unwrap(); - let candidate = PhysicalASAPDAG { + let candidate = CompiledPhysicalPlan { precompute: Some(precompute), query, materialized_outputs: BTreeMap::from([(1, output)]), }; let bytes = serde_json::to_vec(&candidate).unwrap(); - let restored = serde_json::from_slice::(&bytes).unwrap(); + let restored = serde_json::from_slice::(&bytes).unwrap(); assert_eq!(restored.precompute.as_ref().unwrap().roots(), &[1]); assert_eq!(restored.query.roots(), &[2]); assert_eq!(serde_json::to_vec(&restored).unwrap(), bytes); @@ -103,6 +100,7 @@ fn candidate_recovery_preserves_materialization_boundary() { wire["materialized_outputs"]["1"]["schema"]["fields"][0]["dtype"] = serde_json::json!({"Plain":"utf8"}); assert!( - serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()).is_err() + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) + .is_err() ); } diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/executor/tests/physical_semantics.rs similarity index 88% rename from crates/asap-physical-operators/tests/physical_semantics.rs rename to crates/executor/tests/physical_semantics.rs index 56b62590b..a078cfbdb 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/executor/tests/physical_semantics.rs @@ -1,7 +1,7 @@ //! Contract tests inspired by DataFusion's limit, sort and join test matrices. //! Expectations follow ASAP's IR (notably row-count and IEEE NaN equality). //! Reference: apache/datafusion e2ca7f3, physical-plan/src/{limit.rs,sorts/sort.rs}. -use asap_physical_operators::{ +use asap_executor::{ expressions::CompiledExpression, operators::{Expression, Operator, Reduction, SortKey}, plan::PhysicalDAG, @@ -9,16 +9,19 @@ use asap_physical_operators::{ values::{Batch, SchemaRef, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, -}; -use std::{rc::Rc, sync::Arc}; +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{PhysicalASAPDAGNode, PhysicalASAPOperatorPayload}; +use planner_types::ir::operator::JoinKind; +use planner_types::ir::scalar::CompareOpKind; +use planner_types::ir::schema::{DataType, Field, FieldDataType}; +use planner_types::ir::Predicate; +use planner_types::ir::ScalarExpr; +use std::sync::Arc; fn schema(fields: &[(&str, DataType, bool)]) -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: fields .iter() .map(|(name, dtype, nullable)| Field { @@ -74,11 +77,12 @@ fn keys(rows: &[Vec]) -> Vec>> { .collect() } fn eq_predicate() -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + Predicate(ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })) + right: Box::new(ScalarExpr::Column(1)), + }) } fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { let input = schema(&[("key", DataType::Float64, true)]); @@ -335,26 +339,29 @@ fn aggregate_empty_and_all_null_follow_asap_contract() { fn projection_rejects_expression_bound_to_another_schema() { let original = schema(&[("a", DataType::Int64, false), ("b", DataType::Int64, false)]); let current = schema(&[("a", DataType::Int64, false)]); - let expr = CompiledExpression::compile(&QueryExpr::Column(1), &original).unwrap(); + let expr = CompiledExpression::compile(&ScalarExpr::Column(1), &original).unwrap(); assert!(Operator::project(current, vec![("b".into(), Expression::planner(expr))]).is_err()); } // A valid Planner MIN/MAX schema must bind even for a non-null input column. #[test] fn global_extrema_bind_with_planner_derived_schema() { - use asap_physical_operators::physical_planner::compile_node; - use planner_types::{ - post_asap::*, - pre_asap::{AggIntent, Field, GroupKeys, Reduction as PlanReduction}, - }; + use asap_executor::physical_planner::compile_node; + use planner_types::ir::operator::{AggIntent, GroupKeys, Reduction as PlanReduction}; + use planner_types::ir::properties::*; + use planner_types::ir::schema::*; let input = schema(&[("v", DataType::Int64, false)]); for measure in [ AggIntent::Min { col: Some(0) }, AggIntent::Max { col: Some(0) }, ] { let planner_input = - planner_types::pre_asap::Schema::new(vec![Field::plain("v", DataType::Int64, false)]); - let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( + planner_types::ir::schema::Schema::new(vec![planner_types::ir::schema::Field::plain( + "v", + DataType::Int64, + false, + )]); + let derived = planner_types::ir::schema::aggregate_output_schema( &planner_input, &PlanReduction::Reduce(GroupKeys::by(vec![])), std::slice::from_ref(&measure), @@ -364,19 +371,20 @@ fn global_extrema_bind_with_planner_derived_schema() { let result = derived.fields[0].clone(); let output = schema(&[( &result.name, - result.dtype.plain().unwrap().clone(), + result.plain_dtype().unwrap().clone(), result.nullable, )]); - let node = PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::Exact(ExactOperation::Aggregate { + let node = PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(1), + payload: PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Aggregate { reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), measures: vec![measure], output_names: vec![result.name], filters: vec![], having: None, - }), + }, }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*output).clone(), @@ -407,10 +415,11 @@ fn planner_comparisons_handle_nan_without_execution_errors() { CompareOpKind::Gt, CompareOpKind::Ge, ] { - let expression = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let expression = ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(ScalarExpr::Column(0)), op: op.clone(), - right: Rc::new(QueryExpr::Column(1)), + right: Box::new(ScalarExpr::Column(1)), }; let compiled = CompiledExpression::compile(&expression, &input).unwrap(); for row in [ @@ -474,10 +483,11 @@ fn mixed_numeric_comparisons_preserve_large_integer_precision() { ("a", DataType::Int64, false), ("b", DataType::Float64, false), ]); - let expr = QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let expr = ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Column(1)), + right: Box::new(ScalarExpr::Column(1)), }; let compiled = CompiledExpression::compile(&expr, &input).unwrap(); for (a, b, expected) in [ @@ -498,11 +508,11 @@ fn boolean_truth_tables_agree_between_expression_paths() { for and in [true, false] { for a in [None, Some(false), Some(true)] { for b in [None, Some(false), Some(true)] { - let parts = vec![QueryExpr::Column(0), QueryExpr::Column(1)]; + let parts = vec![ScalarExpr::Column(0), ScalarExpr::Column(1)]; let planner = if and { - QueryExpr::BoolAnd(parts) + ScalarExpr::BoolAnd(parts) } else { - QueryExpr::BoolOr(parts) + ScalarExpr::BoolOr(parts) }; let native = if and { Expression::And( @@ -543,8 +553,9 @@ fn boolean_truth_tables_agree_between_expression_paths() { // Partial/final execution must agree with one build for an uncompacted KLL population. #[test] -fn kll_partial_merge_and_multiple_readouts_preserve_population() { - use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; +fn kll_partial_merge_and_multiple_evaluations_preserve_population() { + use planner_types::ir::schema::{SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("v", DataType::Float64, false)]); let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), @@ -588,11 +599,11 @@ fn kll_partial_merge_and_multiple_readouts_preserve_population() { dag.add( id, vec![build], - Operator::readout( + Operator::evaluation( state.clone(), 0, - asap_physical_operators::operators::ReadoutQuery::Sketch( - planner_types::post_asap::SketchStatistic::Quantile { q }, + asap_executor::operators::SummaryEvaluation::Sketch( + planner_types::ir::schema::SketchStatistic::Quantile { q }, ), ) .unwrap(), @@ -611,8 +622,8 @@ fn kll_partial_merge_and_multiple_readouts_preserve_population() { )); for (pair, expected) in outputs.chunks(2).zip([0., 64., 127.]) { let value = |batches: &[Result< - asap_physical_operators::runtime::SharedValue, - asap_physical_operators::Error, + asap_executor::runtime::SharedValue, + asap_executor::Error, >]| { assert_eq!(batches.len(), 1); match batches[0].as_ref().unwrap().rows()[0][0] { @@ -631,7 +642,7 @@ fn kll_partial_merge_and_multiple_readouts_preserve_population() { // Retained zero-column rows still own Vec headers and must consume the output budget. #[test] fn zero_column_output_obeys_memory_limit() { - use asap_physical_operators::Error; + use asap_executor::Error; let input = schema(&[]); let batch = Batch::try_new(input.clone(), vec![vec![]; 200]).unwrap(); let mut dag = PhysicalDAG::default(); @@ -660,8 +671,9 @@ fn zero_column_output_obeys_memory_limit() { // Empty exact-state finalization must preserve ordinary global MIN/MAX null semantics. #[test] fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { - use asap_physical_operators::Statistic; - use planner_types::post_asap::{ExactKind, ExactParams}; + use asap_executor::Statistic; + use planner_types::ir::schema::{ExactKind, ExactParams}; + let input = schema(&[("v", DataType::Float64, false)]); for (kind, params, statistic) in [ (ExactKind::Min, ExactParams::Min, Statistic::Min), @@ -683,11 +695,11 @@ fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { dag.add( 2, vec![1], - Operator::readout( + Operator::evaluation( state, 0, - asap_physical_operators::operators::ReadoutQuery::Exact( - asap_physical_operators::summary_kernels::exact::ExactReadout { + asap_executor::operators::SummaryEvaluation::Exact( + asap_executor::summary_kernels::exact::ExactEvaluation { statistic, lookback_ms: None, }, diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/executor/tests/plan_properties.rs similarity index 71% rename from crates/asap-physical-operators/tests/plan_properties.rs rename to crates/executor/tests/plan_properties.rs index e3dd7b5fc..7a629e1d6 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/executor/tests/plan_properties.rs @@ -1,5 +1,5 @@ //! Finite-input contracts are validated before source execution. -use asap_physical_operators::{ +use asap_executor::{ operators::{Operator, SortKey}, plan::{Boundedness, Emission, PhysicalDAG}, runtime::{Limits, OutputStream, RunContext, Scope}, @@ -7,10 +7,8 @@ use asap_physical_operators::{ values::{Batch, SchemaRef}, Error, }; -use planner_types::{ - post_asap::{Field, FieldDataType, Schema}, - pre_asap::{DataType, QueryExpr, Source}, -}; +use planner_types::ir::operator::Source; +use planner_types::ir::schema::{DataType, Field, FieldDataType, Schema}; use std::sync::{ atomic::{AtomicUsize, Ordering}, Arc, @@ -35,9 +33,9 @@ impl RawSource for DeclaredSource { // A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. #[test] fn blocking_inputs_require_an_explicit_finite_source() { - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "v".into(), @@ -67,11 +65,20 @@ fn blocking_inputs_require_an_explicit_finite_source() { ) .unwrap(); let scan = registry - .bind(&QueryExpr::Scan { - source: identity, - schema: Schema::new(vec![Field::plain("v", DataType::Int64, false)]), - predicates: vec![], - }) + .bind( + &planner_types::ir::OperatorNode::new_shared(planner_types::ir::Operator::NonASAP( + planner_types::ir::NonASAPOp::Scan { + source: identity, + schema: Schema::new(vec![planner_types::ir::schema::Field::plain( + "v", + DataType::Int64, + false, + )]), + predicates: vec![], + }, + )) + .unwrap(), + ) .unwrap(); let mut dag = PhysicalDAG::default(); dag.add(0, vec![], scan).unwrap(); @@ -112,16 +119,16 @@ fn blocking_inputs_require_an_explicit_finite_source() { } } -// Kernel support must not be mistaken for executable native state/readout support. +// Kernel support must not be mistaken for executable native state/evaluation support. #[test] fn summary_capability_levels_are_distinct() { - use asap_physical_operators::{ - capability::{validate_native_family, validate_sketch_readout, validate_summary_kernel}, - planner::post_asap::SketchStatistic, + use asap_executor::{ + capability::{validate_native_family, validate_sketch_evaluation, validate_summary_kernel}, + planner::ir::schema::SketchStatistic, }; - use planner_types::{ - post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate}, - pre_asap::ColumnRef, + use planner_types::ir::scalar::ColumnRef; + use planner_types::ir::schema::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate, }; let grouping = GroupingStrategy::default(); let cms = FieldDataType::Sketch( @@ -135,10 +142,10 @@ fn summary_capability_levels_are_distinct() { grouping.clone(), ); let update = SummaryUpdate { - item: Some(planner_types::post_asap::SummaryInputExpr::Column( + item: Some(planner_types::ir::schema::SummaryInputExpr::Column( ColumnRef::Named("host".into()), )), - weight: planner_types::post_asap::SummaryInputExpr::Constant(1.0), + weight: planner_types::ir::schema::SummaryInputExpr::Constant(1.0), weight_domain: Default::default(), }; assert!(validate_summary_kernel(&cms, &update, &grouping).is_ok()); @@ -148,8 +155,8 @@ fn summary_capability_levels_are_distinct() { key: ColumnRef::SampleValue, value: None, }; - assert!(validate_sketch_readout(&cms, &bare_count).is_ok()); - assert!(validate_sketch_readout( + assert!(validate_sketch_evaluation(&cms, &bare_count).is_ok()); + assert!(validate_sketch_evaluation( &cms, &SketchStatistic::PointCount { key: ColumnRef::Named("host".into()), @@ -162,7 +169,7 @@ fn summary_capability_levels_are_distinct() { grouping, ); assert!(validate_native_family(&kll).is_ok()); - assert!(validate_sketch_readout(&kll, &SketchStatistic::Quantile { q: 1.5 }).is_err()); - assert!(validate_sketch_readout(&kll, &SketchStatistic::Cardinality).is_err()); - assert!(validate_sketch_readout(&kll, &SketchStatistic::Quantile { q: 0.5 }).is_ok()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Quantile { q: 1.5 }).is_err()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Cardinality).is_err()); + assert!(validate_sketch_evaluation(&kll, &SketchStatistic::Quantile { q: 0.5 }).is_ok()); } diff --git a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs b/crates/executor/tests/planspace_series_identity_heap.rs similarity index 76% rename from crates/asap-physical-operators/tests/planspace_series_identity_heap.rs rename to crates/executor/tests/planspace_series_identity_heap.rs index 1f03b61c4..ce9fee082 100644 --- a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs +++ b/crates/executor/tests/planspace_series_identity_heap.rs @@ -2,18 +2,24 @@ //! Planner's search space: `enumerate_candidate_dags_for_root` lists //! current-series TopK heaps without a caller-side series-identity pass, cost //! ranking, or workload Cartesian expansion. Placement variants are not listed. -use asap_aware_mapping::{ - accuracy::{AccuracyEvidenceProvider, DefaultAccuracyModel, PropagationStats}, - cost_model::DefaultCostModel, - replacement::{default_strategies_with_evidence, ReplacementProvenance}, - search_workload_with_targets, Proposals, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +mod common; +use common::compile_physical_asap_dag; +use planner_types::ir::OperatorNode; + +use asap_executor::physical_planner::promql_rows::{ + compile_current_series_evaluation, SERIES_IDENTITY_COLUMN, }; -use asap_physical_operators::physical_planner::promql_rows::{ - compile_current_series_readout, SERIES_IDENTITY_COLUMN, +use asap_logical_optimizer::{ + accuracy::AccuracyEvidenceProvider, accuracy::DefaultAccuracyModel, accuracy::PropagationStats, + pass1::replacement::default_strategies_with_evidence, + pass1::replacement::ReplacementProvenance, search_workload_with_targets, Proposals, + ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::cost::cost_model::DefaultCostModel; use planner_types::{ - post_asap::*, - pre_asap::QueryExpr, + ir::properties::*, + ir::schema::*, types::AccuracyTarget, workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, @@ -25,7 +31,7 @@ use std::rc::Rc; struct Evidence; impl AccuracyEvidenceProvider for Evidence { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &OperatorNode) -> Option { Some(1000) } fn propagation_stats( @@ -64,7 +70,7 @@ impl ReplacementStrategy for LogicalOnly { } } -fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { +fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -89,14 +95,12 @@ fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { ..Default::default() }), }; - Rc::new( - asap_frontend_promql::lower_promql_workload(&workload, 0) - .unwrap() - .remove(0), - ) + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0) } -type InventoryDAG = Vec<(usize, Rc)>; +type InventoryDAG = Vec<(usize, Rc)>; /// Candidate DAGs for query 1 of a two-query workload, with and without /// whole-root proposals. Query 0 is a bystander that must not multiply them. @@ -109,12 +113,11 @@ fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec ), (1, lower(query, &accuracy), Some(accuracy)), ]; - let full = default_strategies_with_evidence(&DefaultCostModel, &Evidence); - let logical: Vec> = - default_strategies_with_evidence(&DefaultCostModel, &Evidence) - .into_iter() - .map(|strategy| Box::new(LogicalOnly(strategy)) as Box) - .collect(); + let full = default_strategies_with_evidence(&Evidence); + let logical: Vec> = default_strategies_with_evidence(&Evidence) + .into_iter() + .map(|strategy| Box::new(LogicalOnly(strategy)) as Box) + .collect(); let enumerate = |strategies: &[Box]| { search_workload_with_targets(roots.clone(), strategies, &DefaultAccuracyModel) .enumerate_candidate_dags_for_root(&1, 65_536) @@ -124,12 +127,14 @@ fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec (enumerate(&full), enumerate(&logical)) } +/// Whether an operator below the root carries series identity. A top-k root +/// returns its selected series' identity whatever realizes it. fn carries_identity(dag: &InventoryDAG) -> bool { dag.iter().any(|(_, root)| { - compile_post_asap_dag(root) - .unwrap() - .nodes + let dag = compile_physical_asap_dag(root).unwrap(); + dag.nodes .iter() + .filter(|node| !dag.roots.contains(&node.id)) .any(|node| { node.output_schema .fields @@ -140,7 +145,10 @@ fn carries_identity(dag: &InventoryDAG) -> bool { } /// Shared acceptance checks; returns the added identity-carrying alternatives. -fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec> { +fn added_alternatives( + query: &str, + accuracy: AccuracyTarget, +) -> Vec> { let (full, logical) = inventories(query, accuracy); for (index, dag) in full.iter().enumerate() { assert_eq!(dag.len(), 1, "one root per candidate, no workload product"); @@ -159,15 +167,18 @@ fn added_alternatives(query: &str, accuracy: AccuracyTarget) -> Vec asap_aware_mapping::CandidateLogicalASAPDAGs<&'static str> { +fn grouped_rate_space() -> asap_logical_optimizer::CandidateLogicalASAPDAGs<&'static str> { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -40,26 +47,20 @@ fn grouped_rate_space() -> asap_aware_mapping::CandidateLogicalASAPDAGs<&'static ..Default::default() }), }; - let root = Rc::new( - asap_frontend_promql::lower_promql_workload(&workload, 0) - .unwrap() - .remove(0), - ); - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) - .unwrap(), - ); + let root = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let root = asap_executor::physical_planner::promql_rows::with_series_identity(&root).unwrap(); search_workload(vec![("grouped-rate", root)]) } -fn grouped_rate() -> PostAsapDAG { +fn grouped_rate() -> PhysicalASAPDAG { let space = grouped_rate_space(); - let selected = space - .global_selection(&DefaultCostModel) + let selected = global_selection(&space, &DefaultCostModel) .assemble_selected_query(&space.roots[0].1) .unwrap() .unwrap(); - compile_post_asap_dag(&selected).unwrap() + compile_physical_asap_dag(&selected).unwrap() } fn run(plan: &CompiledPhysicalDAG, inputs: BTreeMap, scope: Scope) -> Vec { let sources = inputs @@ -81,7 +82,7 @@ fn run(plan: &CompiledPhysicalDAG, inputs: BTreeMap, scope: Scope) - }) } -/// Rate readouts and grouped Sum can run together during bounded precompute; +/// Rate evaluations and grouped Sum can run together during bounded precompute; /// storing per-series rates instead leaves the same Sum in the query DAG. #[test] fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { @@ -92,22 +93,20 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } ) }) .unwrap(); - let readout = dag + let evaluation = dag .nodes .iter() .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } + PhysicalASAPOperatorPayload::FinalizeExactAccumulator ) && dag .edges .iter() @@ -116,7 +115,7 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { .unwrap(); let input_schema = Arc::new(state.output_schema.clone()); let (family, update, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family, input, grouping, @@ -137,9 +136,9 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { let state = accumulator.into_accumulator(); expected_rate_sum += state .as_any() - .downcast_ref::() + .downcast_ref::() .unwrap() - .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .evaluation(asap_executor::Statistic::Rate, range_ms, None) .unwrap() .unwrap(); let summary = Value::Summary { @@ -169,10 +168,10 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { }) .collect(); let batch = Batch::try_new(input_schema.clone(), rows).unwrap(); - let root = u64::from(dag.root.0); + let root = u64::from(dag.roots[0].0); let state_id = u64::from(state.id.0); - let rate_id = u64::from(readout.id.0); - let frontiers = asap_physical_operators::physical_planner::enumerate_frontiers( + let rate_id = u64::from(evaluation.id.0); + let frontiers = asap_executor::physical_planner::enumerate_frontiers( &dag, &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), &[root], @@ -183,15 +182,13 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { assert!(frontiers.contains(&vec![rate_id])); assert!(frontiers.contains(&vec![root])); assert!(!frontiers.contains(&vec![root, rate_id])); - assert!( - asap_physical_operators::physical_planner::enumerate_frontiers( - &dag, - &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), - &[root], - 1, - ) - .is_err() - ); + assert!(asap_executor::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 1, + ) + .is_err()); let candidates = compile_candidates( &dag, BTreeMap::from([(state_id, InputContract::bounded(input_schema))]), @@ -256,15 +253,13 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { InputContract::bounded(Arc::new(state.output_schema.clone())), )]); for frontier in [vec![rate_id, rate_id], vec![root, rate_id], vec![999]] { - assert!( - asap_physical_operators::physical_planner::compile_candidate( - &dag, - contracts.clone(), - &[root], - &frontier - ) - .is_err() - ); + assert!(asap_executor::physical_planner::compile_candidate( + &dag, + contracts.clone(), + &[root], + &frontier + ) + .is_err()); } let inventory = compile_candidates( &dag, @@ -365,9 +360,9 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { let rate_of_sum = wrong_order .into_accumulator() .as_any() - .downcast_ref::() + .downcast_ref::() .unwrap() - .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .evaluation(asap_executor::Statistic::Rate, range_ms, None) .unwrap() .unwrap(); assert_ne!( @@ -376,11 +371,11 @@ fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { ); } -/// Enumerated frontiers include both grouped-result and per-series readout +/// Enumerated frontiers include both grouped-result and per-series evaluation /// persistence; an explicit Rate-state input retains its original semantics. #[test] fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { - use asap_physical_operators::physical_planner::enumerate_frontiers; + use asap_executor::physical_planner::enumerate_frontiers; let dag = grouped_rate(); let state = dag .nodes @@ -388,7 +383,7 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } @@ -399,7 +394,7 @@ fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { u64::from(state.id.0), InputContract::bounded(Arc::new(state.output_schema.clone())), )]); - let roots = [u64::from(dag.root.0)]; + let roots = [u64::from(dag.roots[0].0)]; let frontiers = enumerate_frontiers(&dag, &inputs, &roots, 4096).unwrap(); let candidates = compile_candidates(&dag, inputs.clone(), &roots, &frontiers) .into_iter() @@ -422,11 +417,11 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { let mut executed = 0; for forest in inventory.candidates { let root = &forest[0].1; - let dag = compile_post_asap_dag(root).unwrap(); + let dag = compile_physical_asap_dag(root).unwrap(); let Some(state) = dag.nodes.iter().find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } @@ -440,25 +435,25 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. } ) }) .map(|node| u64::from(node.id.0)) - .unwrap_or(u64::from(dag.root.0)); + .unwrap_or(u64::from(dag.roots[0].0)); let physical_asap_dags = compile_candidates( &dag, BTreeMap::from([( u64::from(state.id.0), InputContract::bounded(Arc::new(state.output_schema.clone())), )]), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], &[vec![], vec![boundary]], ); let (family, input, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family, input, grouping, @@ -558,14 +553,14 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { /// The per-frontier lowering used before compile-once cuts: each boundary /// choice lowers the precompute and query DAGs from the logical DAG again. fn recompiled_candidate( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: &BTreeMap, roots: &[u64], frontier: &[u64], -) -> Result { - use asap_physical_operators::plan::Emission; +) -> Result { + use asap_executor::plan::Emission; if frontier.is_empty() { - return Ok(PhysicalASAPDAG { + return Ok(CompiledPhysicalPlan { precompute: None, query: compile(dag, inputs.clone(), roots)?, materialized_outputs: BTreeMap::new(), @@ -580,7 +575,7 @@ fn recompiled_candidate( } let mut query_inputs = inputs.clone(); query_inputs.extend(materialized_outputs.clone()); - Ok(PhysicalASAPDAG { + Ok(CompiledPhysicalPlan { precompute: Some(precompute), query: compile(dag, query_inputs, roots)?, materialized_outputs, @@ -588,7 +583,7 @@ fn recompiled_candidate( } fn assert_cuts_match_recompilation( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, inputs: BTreeMap, roots: &[u64], min_frontiers: usize, @@ -615,13 +610,13 @@ fn grouped_rate_cuts_equal_per_frontier_compilation() { let state = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, PhysicalASAPOperatorPayload::SummaryAgg { .. })) .unwrap(); let inputs = BTreeMap::from([( u64::from(state.id.0), InputContract::bounded(Arc::new(state.output_schema.clone())), )]); - assert_cuts_match_recompilation(&dag, inputs, &[u64::from(dag.root.0)], 3); + assert_cuts_match_recompilation(&dag, inputs, &[u64::from(dag.roots[0].0)], 3); } /// Cuts of a DAG whose nodes lower to helper operators (current-series @@ -655,26 +650,32 @@ fn population_topk_cuts_equal_per_frontier_compilation() { let original = asap_frontend_promql::lower_promql_workload(&workload, 0) .unwrap() .remove(0); - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity(&original) - .unwrap(), - ); - let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( - std::slice::from_ref(&root), - ) - .candidate(&root) - .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let root = + asap_executor::physical_planner::promql_rows::with_series_identity(&original).unwrap(); + let selected = + asap_logical_optimizer::pass1::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let dag = compile_physical_asap_dag(&selected).unwrap(); let raw = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + }) .unwrap(); let inputs = BTreeMap::from([( u64::from(raw.id.0), InputContract::bounded(Arc::new(raw.output_schema.clone())), )]); - let roots = [u64::from(dag.root.0)]; + let roots = [u64::from(dag.roots[0].0)]; let compiled = compile(&dag, inputs.clone(), &roots).unwrap(); // The root reads its population through a Sort helper numbered by the root. let helper = u64::MAX - (roots[0] << 16); @@ -694,24 +695,22 @@ fn cut_candidate_rejects_invalid_frontiers() { let state = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .find(|node| matches!(node.payload, PhysicalASAPOperatorPayload::SummaryAgg { .. })) .unwrap(); - let readout = dag + let evaluation = dag .nodes .iter() .find(|node| { matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator - } + PhysicalASAPOperatorPayload::FinalizeExactAccumulator ) }) .unwrap(); let (state_id, rate_id, root) = ( u64::from(state.id.0), - u64::from(readout.id.0), - u64::from(dag.root.0), + u64::from(evaluation.id.0), + u64::from(dag.roots[0].0), ); let inputs = BTreeMap::from([( state_id, diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/executor/tests/precompute_population.rs similarity index 77% rename from crates/asap-physical-operators/tests/precompute_population.rs rename to crates/executor/tests/precompute_population.rs index 5b91c6e72..e5d35fb8b 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/executor/tests/precompute_population.rs @@ -1,5 +1,5 @@ //! Persisted precompute DAGs preserve group/window identity and execute state-to-state computation. -use asap_physical_operators::{ +use asap_executor::{ factory::create_planner_accumulator, operators::Operator, physical_planner::{precompute, CompiledPhysicalDAG, Source}, @@ -8,20 +8,24 @@ use asap_physical_operators::{ Statistic, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; -use planner_types::{ - post_asap::*, - pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, }; +use planner_types::ir::operator::{BinaryOpKind, GroupKeys, Reduction}; +use planner_types::ir::properties::*; +use planner_types::ir::scalar::{ArithmeticOpKind, ColumnRef}; +use planner_types::ir::schema::{DataType, *}; +use planner_types::ir::BinaryOperator; use std::{collections::BTreeMap, sync::Arc}; // Typed series identity survives finalization and derived precompute through population metadata. #[test] fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = |dtype| Schema { - closed: true, + let schema = |dtype| planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -41,7 +45,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { value_schema.time_index = Some(1); value_schema.fields.push(Field { table: None, - name: planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY.into(), + name: planner_types::ir::schema::PROMQL_SERIES_IDENTITY.into(), dtype: FieldDataType::Plain(DataType::Utf8), nullable: false, }); @@ -50,39 +54,44 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (SummaryInputExpr::Constant(1.), 4.), ] { let nodes = vec![ - PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(0), + payload: PhysicalASAPOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: state_schema.clone(), guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, + PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(1), + payload: PhysicalASAPOperatorPayload::FinalizeExactAccumulator, output_state: ExecutionDataState::INGESTION_ROWS, output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, + PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(2), + payload: PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, }, }, output_state: ExecutionDataState::INGESTION_ROWS, output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(3), - payload: PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(3), + payload: PhysicalASAPOperatorPayload::SummaryAgg { family: family.clone(), input: SummaryUpdate { weight, @@ -104,9 +113,9 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (2, 3, EdgeRole::Input), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + .map(|(producer, consumer, role)| PhysicalASAPDAGEdge { + producer: planner_types::ir::export::LogicalASAPNodeId(producer), + consumer: planner_types::ir::export::LogicalASAPNodeId(consumer), role, intermediate_schema: nodes[producer as usize].output_schema.clone(), data_state: nodes[producer as usize].output_state, @@ -114,10 +123,10 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes, edges, - root: PostAsapNodeId(3), + roots: vec![planner_types::ir::export::LogicalASAPNodeId(3)], }; // Identity metadata must remain one non-null Utf8 column. for mutation in 0..3 { @@ -131,7 +140,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { assert!(precompute::compile(&invalid_identity, &[0], &[3]).is_err()); } let mut invalid_grouping = dag.clone(); - let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = + let PhysicalASAPOperatorPayload::SummaryAgg { reduction, .. } = &mut invalid_grouping.nodes[3].payload else { unreachable!() @@ -209,9 +218,9 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { assert_eq!( state .as_any() - .downcast_ref::() + .downcast_ref::() .unwrap() - .readout(Statistic::Sum, None, None) + .evaluation(Statistic::Sum, None, None) .unwrap() .unwrap(), expected @@ -221,9 +230,9 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { } fn logical_schema(family: FieldDataType) -> Schema { - Schema { - closed: true, + planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -238,34 +247,35 @@ fn state_dag( target: Option, merge: bool, ) -> CompiledPhysicalDAG { - let mut nodes = vec![PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + let mut nodes = vec![PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(0), + payload: PhysicalASAPOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: logical_schema(family.clone()), guarantee: None, }]; if merge { - nodes.push(PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::SummaryMerge, + nodes.push(PhysicalASAPDAGNode { + id: planner_types::ir::export::LogicalASAPNodeId(1), + payload: PhysicalASAPOperatorPayload::SummaryMerge, ..nodes[0].clone() }); } let read_id = nodes.len() as u32; - nodes.push(PostAsapDAGNode { - id: PostAsapNodeId(read_id), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, + nodes.push(PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(read_id), + payload: PhysicalASAPOperatorPayload::FinalizeExactAccumulator, output_state: ExecutionDataState::INGESTION_ROWS, output_schema: logical_schema(FieldDataType::Plain(DataType::Float64)), guarantee: None, }); if let Some(target) = target { - nodes.push(PostAsapDAGNode { - id: PostAsapNodeId(nodes.len() as u32), - payload: PostAsapOperatorPayload::SummaryAgg { + nodes.push(PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(nodes.len() as u32), + payload: PhysicalASAPOperatorPayload::SummaryAgg { family: target.clone(), input: SummaryUpdate::column(ColumnRef::SampleValue), reduction: Reduction::by(vec![]), @@ -278,7 +288,7 @@ fn state_dag( }); } let edges = (1..nodes.len()) - .map(|i| PostAsapDAGEdge { + .map(|i| PhysicalASAPDAGEdge { producer: nodes[i - 1].id, consumer: nodes[i].id, role: EdgeRole::Input, @@ -290,7 +300,11 @@ fn state_dag( .collect(); let root = nodes.last().unwrap().id; precompute::compile( - &PostAsapDAG { nodes, edges, root }, + &PhysicalASAPDAG { + nodes, + edges, + roots: vec![root], + }, &[0], &[u64::from(root.0)], ) @@ -299,9 +313,9 @@ fn state_dag( fn native_run( program: &CompiledPhysicalDAG, family: FieldDataType, - states: Vec>, + states: Vec>, context: RunContext, -) -> Result>, asap_physical_operators::Error> { +) -> Result>, asap_executor::Error> { let program = serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) .unwrap(); @@ -344,11 +358,11 @@ fn ingestion_context(limits: Limits) -> RunContext { ) .unwrap() } -fn sum_state(value: f64) -> Arc { - let mut state = asap_physical_operators::summary_kernels::exact::ExactAccumulator::new( - planner_types::post_asap::FieldDataType::ExactAggregate( - planner_types::post_asap::ExactKind::Sum, - planner_types::post_asap::ExactParams::Sum, +fn sum_state(value: f64) -> Arc { + let mut state = asap_executor::summary_kernels::exact::ExactAccumulator::new( + planner_types::ir::schema::FieldDataType::ExactAggregate( + planner_types::ir::schema::ExactKind::Sum, + planner_types::ir::schema::ExactParams::Sum, ), false, ) @@ -374,7 +388,7 @@ fn explicit_merge_changes_pane_cardinality() { .iter() .map(|row| match row[2] { Value::Float64(v) => v, - _ => panic!("numeric readout expected"), + _ => panic!("numeric evaluation expected"), }) .collect::>(); assert_eq!(values, expected); @@ -415,7 +429,7 @@ fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { // An exact count must not silently lose units when exposed through Float64 rows. #[test] fn precompute_count_conversion_checks_precision() { - use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + use asap_executor::summary_kernels::exact::ExactAccumulator; let family = FieldDataType::ExactAggregate(ExactKind::Count, ExactParams::Count); let program = state_dag(family.clone(), None, false); for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/executor/tests/promql_binary.rs similarity index 78% rename from crates/asap-physical-operators/tests/promql_binary.rs rename to crates/executor/tests/promql_binary.rs index 4eec88aa1..f1ad19c90 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/executor/tests/promql_binary.rs @@ -1,24 +1,26 @@ //! Binary computation must be fully compiled before deployment binds values. -use asap_physical_operators::{ +use asap_executor::{ operators::Operator, physical_planner::{compile_node, CompiledPhysicalDAG, InputContract, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, SchemaRef, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::{ - post_asap::{ - BinaryOperator, ExecutionDataState, Field, FieldDataType, PostAsapDAGNode, PostAsapNodeId, - PostAsapOperatorPayload, Schema, - }, - pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, }; +use planner_types::ir::operator::BinaryOpKind; +use planner_types::ir::properties::ExecutionDataState; +use planner_types::ir::scalar::ArithmeticOpKind; +use planner_types::ir::schema::{DataType, Field, FieldDataType}; +use planner_types::ir::BinaryOperator; use std::{collections::BTreeMap, sync::Arc}; fn schema() -> SchemaRef { - Arc::new(Schema { - closed: true, + Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, @@ -61,10 +63,19 @@ fn program() -> CompiledPhysicalDAG { }) } fn program_for(operator: BinaryOperator) -> CompiledPhysicalDAG { + program_for_bool(operator, false) +} +fn program_for_bool(operator: BinaryOperator, return_bool: bool) -> CompiledPhysicalDAG { let schema = schema(); - let node = PostAsapDAGNode { - id: PostAsapNodeId(2), - payload: PostAsapOperatorPayload::Binary { operator }, + let node = PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(2), + payload: PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator, + return_bool, + }, + }, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, @@ -85,14 +96,14 @@ fn program_for(operator: BinaryOperator) -> CompiledPhysicalDAG { fn evaluate( left: Vec>, right: Vec>, -) -> Result>, asap_physical_operators::Error> { +) -> Result>, asap_executor::Error> { evaluate_with(program(), left, right) } fn evaluate_with( physical_dag: CompiledPhysicalDAG, left: Vec>, right: Vec>, -) -> Result>, asap_physical_operators::Error> { +) -> Result>, asap_executor::Error> { let sources = [left, right] .into_iter() .enumerate() @@ -152,16 +163,19 @@ fn duplicate_matching_identity_is_rejected() { // Scalar broadcasting and comparison filtering keep the vector operand's value. #[test] fn scalar_broadcast_and_bool_comparison_are_distinct() { - use asap_physical_operators::physical_planner::promql_values; - use planner_types::pre_asap::CompareOpKind; + use asap_executor::physical_planner::promql_values; + use planner_types::ir::scalar::CompareOpKind; for return_bool in [false, true] { let physical_dag = promql_values::compile_binary( - &BinaryOperator { - kind: BinaryOpKind::Compare(CompareOpKind::Lt), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, + &asap_executor::expressions::binary::BinaryOperator::from_logical( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Lt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + ), return_bool, true, false, @@ -276,8 +290,8 @@ fn binary_obeys_memory_and_cancellation() { }); assert!(matches!( (cancel, result), - (true, Err(asap_physical_operators::Error::Cancelled)) - | (false, Err(asap_physical_operators::Error::MemoryLimit)) + (true, Err(asap_executor::Error::Cancelled)) + | (false, Err(asap_executor::Error::MemoryLimit)) )); } } @@ -285,12 +299,15 @@ fn binary_obeys_memory_and_cancellation() { // A `bool` comparison over label-map vectors yields 1 or 0 and drops the name. #[test] fn label_map_bool_comparison_drops_the_name() { - let program = program_for(BinaryOperator { - kind: BinaryOpKind::CompareBool(planner_types::pre_asap::CompareOpKind::Gt), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }); + let program = program_for_bool( + BinaryOperator { + kind: BinaryOpKind::Compare(planner_types::ir::scalar::CompareOpKind::Gt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + true, + ); let rows = evaluate_with( program, vec![row("a", "api", 6.)], @@ -309,24 +326,23 @@ fn label_map_bool_comparison_drops_the_name() { assert!(matches!(row[1], Value::Float64(v) if v == 1.)); } -// Stored temporal readouts drop metric names before filter comparisons and set matching. +// Stored temporal evaluations drop metric names before filter comparisons and set matching. #[test] -fn stored_series_readouts_support_filters_and_sets() { - use asap_physical_operators::{ - physical_planner::compile, summary_kernels::exact::ExactAccumulator, - }; - use planner_types::post_asap::*; - use planner_types::pre_asap::{ - schema::PROMQL_SERIES_IDENTITY, CompareOpKind, PromQLVectorSetOpKind, - }; +fn stored_series_evaluations_support_filters_and_sets() { + use asap_executor::{physical_planner::compile, summary_kernels::exact::ExactAccumulator}; + use planner_types::ir::operator::PromQLVectorSetOpKind; + use planner_types::ir::scalar::CompareOpKind; + use planner_types::ir::schema::PROMQL_SERIES_IDENTITY; + use planner_types::{ir::properties::*, ir::schema::*}; + for (exact_kind, params) in [ (ExactKind::Sum, ExactParams::Sum), (ExactKind::Count, ExactParams::Count), ] { let family = FieldDataType::ExactAggregate(exact_kind.clone(), params); - let state_schema = Arc::new(Schema { - closed: true, + let state_schema = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, @@ -351,19 +367,21 @@ fn stored_series_readouts_support_filters_and_sets() { BinaryOpKind::Set(PromQLVectorSetOpKind::Or), ] { let nodes = (0..5) - .map(|id| PostAsapDAGNode { - id: PostAsapNodeId(id), + .map(|id| PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(id), payload: match id { - 0 | 1 => PostAsapOperatorPayload::SummaryMerge, - 2 | 3 => PostAsapOperatorPayload::Value { - operation: ValueOperation::FinalizeExactAccumulator, - }, - _ => PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - kind: kind.clone(), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, + 0 | 1 => PhysicalASAPOperatorPayload::SummaryMerge, + 2 | 3 => PhysicalASAPOperatorPayload::FinalizeExactAccumulator, + _ => PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::BinaryOp { + operator: BinaryOperator { + kind: kind.clone(), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, }, }, }, @@ -387,9 +405,9 @@ fn stored_series_readouts_support_filters_and_sets() { (3, 4, EdgeRole::Right), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + .map(|(producer, consumer, role)| PhysicalASAPDAGEdge { + producer: planner_types::ir::export::LogicalASAPNodeId(producer), + consumer: planner_types::ir::export::LogicalASAPNodeId(consumer), role, intermediate_schema: nodes[producer as usize].output_schema.clone(), data_state: nodes[producer as usize].output_state, @@ -397,10 +415,10 @@ fn stored_series_readouts_support_filters_and_sets() { window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes, edges, - root: PostAsapNodeId(4), + roots: vec![planner_types::ir::export::LogicalASAPNodeId(4)], }; let physical_dag = compile( &dag, diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/executor/tests/promql_fallback.rs similarity index 90% rename from crates/asap-physical-operators/tests/promql_fallback.rs rename to crates/executor/tests/promql_fallback.rs index 45b3526f6..b6f1b5789 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/executor/tests/promql_fallback.rs @@ -1,27 +1,34 @@ //! A retained PromQL sub-DAG (`Fallback`) compiles from its typed expression. //! The deployment supplies only its selector's raw series; expected values are //! hand-computed with Prometheus semantics. -use asap_physical_operators::{ +mod common; +use asap_executor::{ operators::Operator, physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDAG, InputContract}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; +use common::compile_physical_asap_dag; use futures::{executor::block_on, StreamExt}; -use planner_types::{ - post_asap::{execution_data_state::lift_plain, *}, - pre_asap::QueryExpr, - types::AccuracyTarget, - workload::*, -}; +use planner_types::ir::export::PhysicalASAPDAG; +use planner_types::physical::execution_data_state::lift_plain; +use planner_types::types::AccuracyTarget; +use planner_types::workload::*; use std::{collections::BTreeMap, rc::Rc}; /// Bare selectors look back one ingestion interval: 60s. -fn parse(query: &str) -> QueryExpr { +fn parse(query: &str) -> Rc { parse_with(query, AccuracyTarget::Exact) } -fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +fn parse_with(query: &str, accuracy: AccuracyTarget) -> Rc { + match parse_root(query, accuracy) { + planner_types::ir::QueryRoot::Operator(node) => node, + _ => panic!("expected operator query"), + } +} + +fn parse_root(query: &str, accuracy: AccuracyTarget) -> planner_types::ir::QueryRoot { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -46,24 +53,18 @@ fn parse_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { ..Default::default() }), }; - asap_frontend_promql::lower_promql_workload(&workload, 0) + asap_frontend_promql::lower_promql_query_workload(&workload, 0) .unwrap() .remove(0) } -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { promql_rows::with_series_identity(&parse(query)).unwrap() } /// The whole query retained as one pre-ASAP node. -fn fallback_dag(expression: QueryExpr) -> PostAsapDAG { - let schema = lift_plain(&expression.output_schema().unwrap()); - compile_post_asap_dag(&Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(expression)), - schema, - guarantee: None, - })) - .unwrap() +fn fallback_dag(expression: Rc) -> PhysicalASAPDAG { + compile_physical_asap_dag(&expression).unwrap() } /// `(labels, seconds, value)`. `labels` is `k=v,...`, or a bare `job` value. @@ -82,13 +83,14 @@ fn labels(spec: &str) -> BTreeMap { } /// The metric a selector reads. -fn metric(selector: &QueryExpr) -> String { - match selector { - QueryExpr::Scan { - source: planner_types::pre_asap::Source::TimeSeries { metric }, +fn metric(selector: &planner_types::ir::OperatorNode) -> String { + match selector.expect_non_asap() { + planner_types::ir::NonASAPOp::Scan { + source: planner_types::ir::operator::Source::TimeSeries { metric }, .. } => metric.clone(), - QueryExpr::TimeRange { child, .. } | QueryExpr::TimeShift { child, .. } => metric(child), + planner_types::ir::NonASAPOp::TimeRange { child, .. } + | planner_types::ir::NonASAPOp::TimeShift { child, .. } => metric(child), other => panic!("not a selector: {other:?}"), } } @@ -99,8 +101,11 @@ fn compile_query(query: &str) -> Result { } /// Compile a DAG whose root is the Fallback computing `expression`. -fn compile_dag(expression: &QueryExpr, dag: &PostAsapDAG) -> Result { - let root = u64::from(dag.root.0); +fn compile_dag( + expression: &planner_types::ir::OperatorNode, + dag: &PhysicalASAPDAG, +) -> Result { + let root = u64::from(dag.roots[0].0); let inputs = promql_fallback::raw_series(expression) .map_err(|e| e.to_string())? .into_iter() @@ -124,14 +129,26 @@ fn evaluate( metrics: &[(&str, &[Sample])], at: i64, ) -> Result, i64, f64)>, String> { - let expression = lower(query); - evaluate_dag(&expression, &fallback_dag(expression.clone()), metrics, at) + match parse_root(query, AccuracyTarget::Exact) { + planner_types::ir::QueryRoot::Operator(expression) => { + let expression = + promql_rows::with_series_identity(&expression).map_err(|e| e.to_string())?; + evaluate_dag(&expression, &fallback_dag(expression.clone()), metrics, at) + } + planner_types::ir::QueryRoot::Scalar(expr) => { + let expr = expr + .map_operator_refs(&mut |node| promql_rows::with_series_identity(node).unwrap()); + let (program, selectors) = + promql_fallback::compile_scalar_root(&expr).map_err(|e| e.to_string())?; + execute_program(program, selectors, metrics, at, None) + } + } } #[allow(clippy::type_complexity)] fn evaluate_dag( - expression: &QueryExpr, - dag: &PostAsapDAG, + expression: &planner_types::ir::OperatorNode, + dag: &PhysicalASAPDAG, metrics: &[(&str, &[Sample])], at: i64, ) -> Result, i64, f64)>, String> { @@ -140,15 +157,26 @@ fn evaluate_dag( #[allow(clippy::type_complexity)] fn evaluate_dag_with_range( - expression: &QueryExpr, - dag: &PostAsapDAG, + expression: &planner_types::ir::OperatorNode, + dag: &PhysicalASAPDAG, metrics: &[(&str, &[Sample])], at: i64, bounds: Option<(i64, i64)>, ) -> Result, i64, f64)>, String> { let program = compile_dag(expression, dag)?; - let mut sources = BTreeMap::new(); let selectors = promql_fallback::raw_series(expression).unwrap(); + execute_program(program, selectors, metrics, at, bounds) +} + +#[allow(clippy::type_complexity)] +fn execute_program( + program: CompiledPhysicalDAG, + selectors: Vec, + metrics: &[(&str, &[Sample])], + at: i64, + bounds: Option<(i64, i64)>, +) -> Result, i64, f64)>, String> { + let mut sources = BTreeMap::new(); for (i, (selector, schema)) in selectors.into_iter().enumerate() { let name = metric(&selector); let rows = metrics @@ -423,12 +451,15 @@ fn dense_subquery_grids_are_rejected() { fn raw_series_contract_is_explicit() { let expression = lower("rate(m[5m])"); let dag = fallback_dag(expression.clone()); - let root = u64::from(dag.root.0); + let root = u64::from(dag.roots[0].0); let [(selector, schema)] = promql_fallback::raw_series(&expression) .unwrap() .try_into() .unwrap(); - assert!(matches!(selector, QueryExpr::TimeRange { .. })); + assert!(matches!( + selector.expect_non_asap(), + planner_types::ir::NonASAPOp::TimeRange { .. } + )); let missing = compile(&dag, BTreeMap::new(), &[root]).err().unwrap(); assert!(missing.to_string().contains("raw series input")); let mut wrong = (*schema).clone(); @@ -445,54 +476,31 @@ fn raw_series_contract_is_explicit() { // A consumed bare selector is raw range rows for its consumer; it is not // turned into instant selection. let selector = lower("m"); - let schema = lift_plain(&selector.output_schema().unwrap()); - let node = |id, payload| PostAsapDAGNode { - id: PostAsapNodeId(id), - payload, - output_state: ExecutionDataState::QUERY_ROWS, - output_schema: schema.clone(), - guarantee: None, - }; - let consumed = PostAsapDAG { - nodes: vec![ - node( - 0, - PostAsapOperatorPayload::Fallback { - expression: selector.clone(), - }, - ), - node( - 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, - offset: 0, - partition_by: Default::default(), - }, - }, - ), - ], - edges: vec![PostAsapDAGEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), - role: EdgeRole::Input, - intermediate_schema: schema.clone(), - data_state: ExecutionDataState::QUERY_ROWS, - grouping: GroupingEdgeCompatibility::NotApplicable, - window: WindowEdgeCompatibility::NotApplicable, - }], - root: PostAsapNodeId(1), - }; + let _schema = lift_plain(&selector.schema.clone()); + let consumed = planner_types::ir::OperatorNode::new_shared( + planner_types::ir::Operator::NonASAP(planner_types::ir::NonASAPOp::Limit { + n: Some(1), + offset: 0, + partition_by: Default::default(), + child: selector.clone(), + }), + ) + .unwrap(); + let consumed = fallback_dag(consumed.clone()); let raw = promql_fallback::raw_series(&selector).unwrap().remove(0).1; assert!(compile( &consumed, BTreeMap::from([( - promql_fallback::raw_series_input(0, 0), + promql_fallback::raw_series_input(u64::from(consumed.roots[0].0), 0), InputContract::bounded(raw) )]), - &[1], + &consumed + .roots + .iter() + .map(|r| u64::from(r.0)) + .collect::>() ) - .is_err()); + .is_ok()); // Implicit subquery resolution belongs to the deployment's evaluation interval. assert!(compile_query("max_over_time(m[5m:])").is_err()); } @@ -1221,19 +1229,19 @@ fn histogram_quantile_rejects_equal_output_label_sets() { // for an approximate target, and the selected DAG compiles and executes. #[test] fn histogram_quantile_selection_keeps_the_exact_fallback() { - use asap_aware_mapping::{ - accuracy::DefaultAccuracyModel, cost_model::DefaultCostModel, default_strategies, - search_workload_with_targets, Replacement, + use asap_logical_optimizer::{ + accuracy::DefaultAccuracyModel, default_strategies, search_workload_with_targets, + Replacement, }; + use asap_plan_selection::cost::cost_model::DefaultCostModel; let samples = buckets(&[("job=a", HISTOGRAM)]); for target in [AccuracyTarget::Exact, AccuracyTarget::Epsilon(0.01)] { for query in [ "histogram_quantile(0.5, x_bucket)", "histogram_quantile(0.5, sum by (le, job) (x_bucket))", ] { - let root = Rc::new( - promql_rows::with_series_identity(&parse_with(query, target.clone())).unwrap(), - ); + let root = + promql_rows::with_series_identity(&parse_with(query, target.clone())).unwrap(); let space = search_workload_with_targets( vec![(query, root.clone(), Some(target.clone()))], &default_strategies(), @@ -1243,16 +1251,17 @@ fn histogram_quantile_selection_keeps_the_exact_fallback() { let candidates = &space.candidates_for_target(planned).unwrap().candidates; assert!( candidates.iter().all(|c| matches!(&c.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::KeepPreAsap(e) if **e == *root))), + Replacement::SubDAG(node) if !node.contains_asap() && node.operator == root.operator)), "{query}: {candidates:?}" ); - let selected = space - .global_selection(&DefaultCostModel) - .assemble_selected_dag(planned) - .unwrap() - .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let selected = asap_plan_selection::candidate_selection::global_selection( + &space, + &DefaultCostModel, + ) + .assemble_selected_dag(planned) + .unwrap() + .unwrap(); + let dag = compile_physical_asap_dag(&selected).unwrap(); let rows = evaluate_dag(&root, &dag, &[("x_bucket", &samples)], 60).unwrap(); let values: Vec<_> = rows.iter().map(|(_, _, v)| *v).collect(); assert_eq!(values, vec![1.75], "{query} {target:?}"); @@ -1285,7 +1294,7 @@ fn nonfinite_literals_round_trip_in_plans() { ] { let expression = lower(query); let json = serde_json::to_vec(&expression).unwrap(); - let restored: QueryExpr = serde_json::from_slice(&json).unwrap(); + let restored: Rc = serde_json::from_slice(&json).unwrap(); let result = evaluate_dag(&restored, &fallback_dag(restored.clone()), &[], 60).unwrap(); assert_eq!(result.len(), 1); if expected.is_nan() { @@ -1579,7 +1588,7 @@ fn subquery_label_uniqueness_is_checked_per_evaluation_step() { #[test] fn logical_nonfinite_quantile_parameter_round_trips() { let expression = lower("histogram_quantile(NaN, x_bucket)"); - let restored: QueryExpr = + let restored: Rc = serde_json::from_slice(&serde_json::to_vec(&expression).unwrap()).unwrap(); let samples = buckets(&[("job=a", HISTOGRAM)]); let result = evaluate_dag( @@ -1592,3 +1601,56 @@ fn logical_nonfinite_quantile_parameter_round_trips() { assert_eq!(result.len(), 1); assert!(result[0].2.is_nan()); } + +/// The proposal's pointwise projections preserve names only for unary minus. +#[test] +fn pointwise_projection_names_and_dynamic_parameters() { + let samples = [("job=a", 300, -2.5)]; + assert_eq!( + labeled("-m", &[("m", &samples)], 300), + [("__name__=m,job=a".into(), 2.5)] + ); + assert_eq!( + labeled("abs(m)", &[("m", &samples)], 300), + [("job=a".into(), 2.5)] + ); + assert_eq!( + run("round(m, scalar(vector(2)))", &samples, 300).unwrap(), + [("a".into(), 300_000, -2.0)] + ); + assert_eq!( + run("clamp(m, time()-301, time())", &samples, 300).unwrap(), + [("a".into(), 300_000, -1.0)] + ); + assert!(run("clamp(m, 2, 1)", &samples, 300).unwrap().is_empty()); + assert_eq!( + run("year(m)", &[("a", 300, 0.0)], 300).unwrap(), + [("a".into(), 300_000, 1970.0)] + ); + assert_eq!( + run("hour()", &[], 3600).unwrap(), + [("".into(), 3_600_000, 1.0)] + ); +} + +/// Execute every PromQL root/conversion example in the scalar design document. +#[test] +fn scalar_design_document_examples_execute() { + let samples = [("job=a", 300, 1.0), ("job=b", 300, 2.0)]; + for (query, expected) in [ + ("2", 2.0), + ("time()", 300.0), + ("vector(time())", 300.0), + ("scalar(sum(up)) + 1", 4.0), + ] { + let root = parse_root(query, AccuracyTarget::Exact); + root.validate_structure().unwrap(); + let output = evaluate(query, &[("up", &samples)], 300).unwrap(); + assert_eq!(output.len(), 1, "{query}"); + assert_eq!(output[0].2, expected, "{query}"); + } + assert_eq!( + labeled("up * 2", &[("up", &samples)], 300), + [("job=a".into(), 2.0), ("job=b".into(), 4.0)] + ); +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/executor/tests/promql_values.rs similarity index 94% rename from crates/asap-physical-operators/tests/promql_values.rs rename to crates/executor/tests/promql_values.rs index e7deb78b1..4644d8052 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/executor/tests/promql_values.rs @@ -1,12 +1,16 @@ //! Compile, persist and rebind dynamic-label computation without deployment lowering. -use asap_physical_operators::{ +use asap_executor::expressions::binary::{BinaryOpKind, BinaryOperator}; + +use asap_executor::{ operators::Operator, physical_planner::{promql_values::*, CompiledPhysicalDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::{AggIntent, ColumnRef, GroupKeys}; +use planner_types::ir::operator::{AggIntent, GroupKeys}; +use planner_types::ir::scalar::ColumnRef; + use std::collections::BTreeMap; fn row(labels: &[(&str, &str)], value: f64) -> Vec { @@ -27,7 +31,7 @@ fn run(dag: CompiledPhysicalDAG, rows: Vec>) -> Vec> { fn run_inputs( dag: CompiledPhysicalDAG, batches: Vec, -) -> Result>, asap_physical_operators::Error> { +) -> Result>, asap_executor::Error> { let dag = serde_json::from_slice::(&serde_json::to_vec(&dag).unwrap()).unwrap(); let sources = batches @@ -207,16 +211,13 @@ fn histogram_quantile_keeps_each_label_group() { // Linking an ensemble preserves its shared producer and every selected operator. #[test] fn composed_ensemble_shares_a_producer_across_roots() { - use asap_physical_operators::{ + use asap_executor::{ physical_planner::InputContract, plan::{PhysicalOperator, PlanProperties}, runtime::{Input, OutputStream}, values::SchemaRef, }; - use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{ArithmeticOpKind, BinaryOpKind}, - }; + use planner_types::ir::scalar::ArithmeticOpKind; struct Counted { source: Operator, starts: std::rc::Rc>, @@ -241,7 +242,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { &'a self, inputs: Vec>, context: RunContext, - ) -> Result, asap_physical_operators::Error> { + ) -> Result, asap_executor::Error> { self.starts.set(self.starts.get() + 1); self.source.start(inputs, context) } @@ -325,10 +326,7 @@ fn compiled_constant_needs_no_deployment_source() { // arithmetic or bool comparisons remove the metric name. #[test] fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { - use planner_types::{ - post_asap::BinaryOperator, - pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}, - }; + use planner_types::ir::scalar::{ArithmeticOpKind, CompareOpKind}; for left_scalar in [false, true] { for names in [["a", "a"], ["a", "b"]] { for (kind, return_bool) in [ @@ -398,12 +396,12 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { ); } -// Persisted exact readout DAGs, rather than the storage adapter, merge panes, +// Persisted exact evaluation graphs, rather than the storage adapter, merge panes, // finalize each population, and preserve the requested metric-name semantics. #[test] -fn exact_state_readouts_recover_and_finalize_panes() { - use asap_physical_operators::factory::create_planner_accumulator; - use planner_types::post_asap::*; +fn exact_state_evaluations_recover_and_finalize_panes() { + use asap_executor::factory::create_planner_accumulator; + use planner_types::ir::schema::*; use std::sync::Arc; for (kind, params, expected) in [ (ExactKind::Sum, ExactParams::Sum, 12.), @@ -436,7 +434,7 @@ fn exact_state_readouts_recover_and_finalize_panes() { }) .collect(); let output = run_inputs( - compile_exact_readout(family.clone(), 60_000, preserve).unwrap(), + compile_exact_evaluation(family.clone(), 60_000, preserve).unwrap(), vec![Batch::try_new(exact_state_schema(family.clone()).unwrap(), rows).unwrap()], ) .unwrap(); @@ -452,8 +450,8 @@ fn exact_state_readouts_recover_and_finalize_panes() { #[test] fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { - use asap_physical_operators::factory::create_planner_accumulator; - use planner_types::post_asap::*; + use asap_executor::factory::create_planner_accumulator; + use planner_types::ir::schema::*; use std::sync::Arc; for (kind, params, expected) in [ (ExactKind::Rate, ExactParams::Rate, 1.), @@ -483,7 +481,7 @@ fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { }) .collect(); let output = run_inputs( - compile_exact_readout(family.clone(), 60_000, false).unwrap(), + compile_exact_evaluation(family.clone(), 60_000, false).unwrap(), vec![Batch::try_new(exact_state_schema(family).unwrap(), rows).unwrap()], ) .unwrap(); diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/executor/tests/raw_scan.rs similarity index 76% rename from crates/asap-physical-operators/tests/raw_scan.rs rename to crates/executor/tests/raw_scan.rs index cf56a8682..6db0c6dfa 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/executor/tests/raw_scan.rs @@ -1,31 +1,38 @@ //! Scan acceptance uses the public connector contract and Planner physical DAGs. -use asap_physical_operators::dag::{ +use asap_executor::dag::{ planner::bind_with_data_sources, scan::{DataSources, MemorySource, RawSource}, values::{Batch, SchemaRef, Value}, Error, Limits, OutputStream, RunContext, Scope, }; use futures::{executor::block_on, stream, StreamExt}; -use planner_types::pre_asap::Schema; -use planner_types::{ - post_asap::*, - pre_asap::{DataType, Field, GroupKeys, Predicate, QueryExpr, Source}, +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, }; +use planner_types::ir::operator::{GroupKeys, Source}; +use planner_types::ir::properties::*; +use planner_types::ir::schema::{DataType, Field, *}; +use planner_types::ir::Predicate; use std::{ collections::BTreeMap, - rc::Rc, sync::{ atomic::{AtomicUsize, Ordering}, Arc, }, }; -fn fixture() -> (QueryExpr, SchemaRef, Vec) { +fn fixture() -> (planner_types::ir::NonASAPOp, SchemaRef, Vec) { let schema = - planner_types::pre_asap::Schema::new(vec![Field::plain("value", DataType::Int64, true)]); - let output = Arc::new(Schema { - closed: true, + planner_types::ir::schema::Schema::new(vec![planner_types::ir::schema::Field::plain( + "value", + DataType::Int64, + true, + )]); + let output = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -34,13 +41,13 @@ fn fixture() -> (QueryExpr, SchemaRef, Vec) { }], time_index: None, }); - let scan = QueryExpr::Scan { + let scan = planner_types::ir::NonASAPOp::Scan { source: Source::Table { table_ref: "numbers".into(), }, - predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( - QueryExpr::Column(0), - ))))], + predicates: vec![Predicate(planner_types::ir::ScalarExpr::IsNotNull( + Box::new(planner_types::ir::ScalarExpr::Column(0)), + ))], schema, }; let batches = vec![ @@ -57,32 +64,44 @@ fn fixture() -> (QueryExpr, SchemaRef, Vec) { ]; (scan, output, batches) } -fn plan(scan: QueryExpr, schema: &SchemaRef, state: ExecutionDataState) -> PostAsapDAG { - let node = |id, payload| PostAsapDAGNode { - id: PostAsapNodeId(id), +fn plan( + scan: planner_types::ir::NonASAPOp, + schema: &SchemaRef, + state: ExecutionDataState, +) -> PhysicalASAPDAG { + let node = |id, payload| PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(id), payload, output_state: state, output_schema: (**schema).clone(), guarantee: None, }; - let edge = |producer, consumer| PostAsapDAGEdge { - producer: PostAsapNodeId(producer), - consumer: PostAsapNodeId(consumer), + let edge = |producer, consumer| PhysicalASAPDAGEdge { + producer: planner_types::ir::export::LogicalASAPNodeId(producer), + consumer: planner_types::ir::export::LogicalASAPNodeId(consumer), role: EdgeRole::Input, intermediate_schema: (**schema).clone(), data_state: state, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }; - PostAsapDAG { + PhysicalASAPDAG { nodes: vec![ - node(0, PostAsapOperatorPayload::Fallback { expression: scan }), + node( + 0, + PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::from_op(&scan, &mut |_| { + panic!("no plan refs") + }), + }, + ), node( 1, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Sort { - keys: vec![planner_types::pre_asap::SortKey { - expr: QueryExpr::Column(0), + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Sort { + keys: vec![planner_types::ir::export::WireSortKey { + expr: planner_types::ir::export::WireScalarExpr::Column(0), ascending: false, nulls_first: false, }], @@ -92,9 +111,9 @@ fn plan(scan: QueryExpr, schema: &SchemaRef, state: ExecutionDataState) -> PostA ), node( 2, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 2, + PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Limit { + n: Some(2), offset: 0, partition_by: GroupKeys::by(vec![]), }, @@ -102,7 +121,7 @@ fn plan(scan: QueryExpr, schema: &SchemaRef, state: ExecutionDataState) -> PostA ), ], edges: vec![edge(0, 1), edge(1, 2)], - root: PostAsapNodeId(2), + roots: vec![planner_types::ir::export::LogicalASAPNodeId(2)], } } fn registry(source: Arc) -> DataSources { @@ -173,8 +192,8 @@ struct CountingSource { fail: bool, } impl RawSource for CountingSource { - fn boundedness(&self) -> asap_physical_operators::plan::Boundedness { - asap_physical_operators::plan::Boundedness::Bounded + fn boundedness(&self) -> asap_executor::plan::Boundedness { + asap_executor::plan::Boundedness::Bounded } fn schema(&self) -> SchemaRef { self.schema.clone() @@ -223,17 +242,27 @@ fn lazy_open_shared_producer_and_cancellation() { #[test] fn binding_errors_and_reader_errors_are_not_empty_results() { let (mut scan, schema, _) = fixture(); - assert!(DataSources::default().bind(&scan).is_err()); + assert!(DataSources::default() + .bind(&planner_types::ir::OperatorNode::with_schema( + planner_types::ir::Operator::NonASAP(scan.clone()), + scan.output_schema().unwrap() + )) + .is_err()); let opened = Arc::new(AtomicUsize::new(0)); let sources = registry(Arc::new(CountingSource { schema: schema.clone(), opened: opened.clone(), fail: true, })); - if let QueryExpr::Scan { predicates, .. } = &mut scan { - predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + if let planner_types::ir::NonASAPOp::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(planner_types::ir::ScalarExpr::Column(0))); } - assert!(sources.bind(&scan).is_err()); + assert!(sources + .bind(&planner_types::ir::OperatorNode::with_schema( + planner_types::ir::Operator::NonASAP(scan.clone()), + scan.output_schema().unwrap() + )) + .is_err()); assert_eq!(opened.load(Ordering::SeqCst), 0); let (scan, _, _) = fixture(); let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); @@ -296,20 +325,24 @@ fn schema_drift_and_memory_limits_fail_the_scan() { // An empty table is a valid empty scan; nullable comparisons retain only TRUE. #[test] fn empty_sources_and_three_valued_predicates() { - use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + use planner_types::ir::scalar::{CompareOpKind, ScalarValue}; + let (mut scan, schema, batches) = fixture(); - if let QueryExpr::Scan { + if let planner_types::ir::NonASAPOp::Scan { predicates, source, .. } = &mut scan { *source = Source::TimeSeries { metric: "samples".into(), }; - *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + *predicates = vec![Predicate(planner_types::ir::ScalarExpr::Compare { + semantics: planner_types::ir::ExprSemantics::Sql, + left: Box::new(planner_types::ir::ScalarExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), - }))]; + right: Box::new(planner_types::ir::ScalarExpr::Literal(ScalarValue::Int64( + 2, + ))), + })]; } for (batches, expected) in [(vec![], 0), (batches, 2)] { let mut sources = DataSources::default(); @@ -337,7 +370,7 @@ fn empty_sources_and_three_valued_predicates() { // A physical candidate can be compiled once without readers and rebound per run. #[test] fn compile_without_readers_and_rebind_inputs() { - use asap_physical_operators::{ + use asap_executor::{ operators::Operator, physical_planner::{compile, InputContract, Source}, }; @@ -373,7 +406,7 @@ fn compile_without_readers_and_rebind_inputs() { // Input boundedness must be proved during compilation, before readers exist. #[test] fn compilation_rejects_unknown_boundedness_for_sort() { - use asap_physical_operators::{ + use asap_executor::{ physical_planner::{compile, InputContract}, plan::{Boundedness, Emission, PlanProperties}, }; diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/executor/tests/summary_projection.rs similarity index 76% rename from crates/asap-physical-operators/tests/summary_projection.rs rename to crates/executor/tests/summary_projection.rs index 8cac1cbda..cb44f5d6f 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/executor/tests/summary_projection.rs @@ -1,5 +1,5 @@ //! Opaque state travels through a retained physical projection without scalar decoding. -use asap_physical_operators::{ +use asap_executor::{ expressions::Expression, factory::create_planner_accumulator, operators::Operator, @@ -8,11 +8,14 @@ use asap_physical_operators::{ values::{Batch, Value}, }; use futures::{executor::block_on, StreamExt}; -use planner_types::pre_asap::Schema; -use planner_types::{ - post_asap::*, - pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, +use planner_types::ir::export::NonASAPOpKind as ValueOperation; +use planner_types::ir::export::{ + EdgeRole, GroupingEdgeCompatibility, PhysicalASAPDAG, PhysicalASAPDAGEdge, PhysicalASAPDAGNode, + PhysicalASAPOperatorPayload, WindowEdgeCompatibility, }; +use planner_types::ir::properties::*; +use planner_types::ir::scalar::ColumnRef; +use planner_types::ir::schema::{DataType, *}; use std::{collections::BTreeMap, sync::Arc}; // A Post-ASAP projection may reorder/rename summary columns; recovery must retain @@ -20,9 +23,9 @@ use std::{collections::BTreeMap, sync::Arc}; #[test] fn post_asap_summary_projection_survives_recovery() { let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let schema = Arc::new(Schema { - closed: true, + let schema = Arc::new(planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![ Field { table: None, @@ -39,9 +42,9 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }); - let output = Schema { - closed: true, + let output = planner_types::ir::schema::Schema { unique_keys: vec![], + closed: false, fields: vec![ schema.fields[1].clone(), Field { @@ -51,24 +54,26 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }; - let dag = PostAsapDAG { + let dag = PhysicalASAPDAG { nodes: vec![ - PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: PostAsapOperatorPayload::SummaryMerge, + PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(0), + payload: PhysicalASAPOperatorPayload::SummaryMerge, output_schema: (*schema).clone(), output_state: ExecutionDataState::INGESTION_SUMMARY, guarantee: None, }, - PostAsapDAGNode { - id: PostAsapNodeId(1), - payload: PostAsapOperatorPayload::Value { - operation: ValueOperation::Project { + PhysicalASAPDAGNode { + coverage: None, + id: planner_types::ir::export::LogicalASAPNodeId(1), + payload: PhysicalASAPOperatorPayload::Relational { + operator: ValueOperation::Project { cols: vec![1, 0] .into_iter() - .map(|index| ProjectItem { + .map(|index| planner_types::ir::export::WireProjectItem { alias: None, - expr: QueryExpr::Column(index), + expr: planner_types::ir::export::WireScalarExpr::Column(index), }) .collect(), qualifier: None, @@ -79,16 +84,16 @@ fn post_asap_summary_projection_survives_recovery() { guarantee: None, }, ], - edges: vec![PostAsapDAGEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), + edges: vec![PhysicalASAPDAGEdge { + producer: planner_types::ir::export::LogicalASAPNodeId(0), + consumer: planner_types::ir::export::LogicalASAPNodeId(1), role: EdgeRole::Input, intermediate_schema: (*schema).clone(), data_state: ExecutionDataState::INGESTION_SUMMARY, grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }], - root: PostAsapNodeId(1), + roots: vec![planner_types::ir::export::LogicalASAPNodeId(1)], }; let program = compile( &dag, diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/executor/tests/weighted_topk_binding.rs similarity index 78% rename from crates/asap-physical-operators/tests/weighted_topk_binding.rs rename to crates/executor/tests/weighted_topk_binding.rs index fe79948d6..e82d0ce65 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/executor/tests/weighted_topk_binding.rs @@ -1,27 +1,26 @@ //! Planner output binds directly to the shared runtime at a declared rate-value frontier. -use asap_aware_mapping::{ - accuracy::{ - AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, - }, - cost_model::DefaultCostModel, - Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, -}; -use asap_physical_operators::dag::{ +mod common; +use asap_executor::dag::{ operators::Operator, planner::{compile, InputContract, Source}, values::{Batch, Value}, Limits, RunContext, Scope, }; -use futures::{executor::block_on, StreamExt}; -use planner_types::{ - post_asap::*, - pre_asap::{DataType, QueryExpr}, - types::AccuracyTarget, +use asap_logical_optimizer::{ + accuracy::AccuracyEvidenceProvider, accuracy::DefaultAccuracyModel, + accuracy::EqualSplitAllocator, accuracy::PropagationStats, ASAPStrategies, Replacement, + ReplacementStrategy, TargetSubDAG, }; +use common::compile_physical_asap_dag; +use futures::{executor::block_on, StreamExt}; +use planner_types::ir::export::{PhysicalASAPDAG, PhysicalASAPOperatorPayload}; +use planner_types::ir::properties::*; +use planner_types::ir::schema::{DataType, *}; +use planner_types::types::AccuracyTarget; use std::{collections::BTreeMap, rc::Rc, sync::Arc}; struct Evidence; impl AccuracyEvidenceProvider for Evidence { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &planner_types::ir::OperatorNode) -> Option { Some(1000) } fn propagation_stats( @@ -53,25 +52,22 @@ fn planner_weighted_topk_binds_at_either_deployment_phase() { #[test] fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { assert_weighted_binding( - &asap_aware_mapping::accuracy::NoAccuracyEvidence, + &asap_logical_optimizer::accuracy::NoAccuracyEvidence, SketchAlgorithm::CmsWithHeap, ); assert_weighted_binding( - &asap_aware_mapping::accuracy::NoAccuracyEvidence, + &asap_logical_optimizer::accuracy::NoAccuracyEvidence, SketchAlgorithm::CountSketchWithHeap, ); } fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: SketchAlgorithm) { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.1), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.1), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, evidence, @@ -80,7 +76,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if candidate.rationale.contains(&format!("{algorithm:?}")) => { Some(node) @@ -88,8 +84,8 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S _ => None, }) .unwrap(); - let dag = compile_post_asap_dag(&plan).unwrap(); - let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:FieldDataType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let dag = compile_physical_asap_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PhysicalASAPOperatorPayload::SummaryAgg{family:FieldDataType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); let rate_id = dag .edges .iter() @@ -156,7 +152,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S let compiled = compile( &placed, BTreeMap::from([(rate_id.0 as u64, InputContract::bounded(rates.clone()))]), - &[dag.root.0 as u64], + &[dag.roots[0].0 as u64], ) .unwrap(); let physical_dag = compiled @@ -166,7 +162,7 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S let output = block_on(async { let mut output = Vec::new(); let mut stream = physical_dag - .execute(&[dag.root.0 as u64], context) + .execute(&[dag.roots[0].0 as u64], context) .unwrap() .remove(0); while let Some(batch) = stream.next().await { @@ -203,7 +199,7 @@ use planner_types::workload::{ pub fn lower_promql( query: &str, accuracy: AccuracyTarget, -) -> Result { +) -> Result, asap_frontend_promql::PromqlError> { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -248,21 +244,19 @@ fn rate_updates_cannot_enter_integer_heap_factory() { ); let input = SummaryUpdate { item: Some(SummaryInputExpr::Column( - planner_types::pre_asap::ColumnRef::Named("service".into()), + planner_types::ir::scalar::ColumnRef::Named("service".into()), )), - weight: SummaryInputExpr::Column(planner_types::pre_asap::ColumnRef::SampleValue), + weight: SummaryInputExpr::Column(planner_types::ir::scalar::ColumnRef::SampleValue), weight_domain: WeightDomain::NonNegative { proof: NonNegativeWeightProof::ResetAwareCounterDerivative, }, }; - assert!( - asap_physical_operators::factory::create_planner_accumulator( - &family, - &input, - &Default::default() - ) - .is_err() - ); + assert!(asap_executor::factory::create_planner_accumulator( + &family, + &input, + &Default::default() + ) + .is_err()); } /// A catalog-resolved per-series rate can feed a heap sketch directly, without @@ -272,44 +266,46 @@ fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { check_direct_rate_topk(false); } -// Unreferenced labels still distinguish series throughout Rate and heap readout. +// Unreferenced labels still distinguish series throughout Rate and heap evaluation. #[test] fn direct_rate_topk_preserves_dynamic_unreferenced_labels() { check_direct_rate_topk(true); } fn check_direct_rate_topk(dynamic: bool) { - use asap_physical_operators::physical_planner::promql_rows::{ + use asap_executor::physical_planner::promql_rows::{ decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, }; let mut logical = lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); - fn resolve_catalog(node: &mut QueryExpr) { - match node { - QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { - resolve_catalog(Rc::make_mut(child)) - } - QueryExpr::Scan { schema, .. } => { + fn resolve_catalog(node: &mut planner_types::ir::OperatorNode) { + match &mut node.operator { + planner_types::ir::Operator::NonASAP( + planner_types::ir::NonASAPOp::Aggregate { child, .. } + | planner_types::ir::NonASAPOp::TimeRange { child, .. }, + ) => resolve_catalog(Rc::make_mut(child)), + planner_types::ir::Operator::NonASAP(planner_types::ir::NonASAPOp::Scan { + schema, + .. + }) => { schema.closed = true; - schema - .fields - .push(planner_types::pre_asap::schema::Field::plain( - "service", - DataType::Utf8, - false, - )); + schema.fields.push(planner_types::ir::schema::Field::plain( + "service", + DataType::Utf8, + false, + )); } _ => panic!("unexpected input shape: {node:?}"), } + node.schema = node.operator.output_schema().unwrap(); } if dynamic { logical = with_series_identity(&logical).unwrap(); } else { - resolve_catalog(&mut logical); + resolve_catalog(Rc::make_mut(&mut logical)); } - let root = Rc::new(logical); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let root = logical; + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &Evidence, @@ -322,7 +318,7 @@ fn check_direct_rate_topk(dynamic: bool) { let candidate = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if candidate.rationale.contains(&format!("{algorithm:?}")) => { Some(node) @@ -332,31 +328,28 @@ fn check_direct_rate_topk(dynamic: bool) { .unwrap_or_else(|| panic!("missing {algorithm:?} over direct Rate")); if dynamic { let (source, ranked) = - asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking( - candidate, - ) - .unwrap(); + asap_executor::physical_planner::promql_rows::compile_rate_ranking(candidate) + .unwrap(); assert!(matches!( - source.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + source.operator, + planner_types::ir::Operator::ASAP( + planner_types::ir::ASAPOp::FinalizeExactAccumulator { .. } + ) )); assert_eq!(ranked.input_contracts().count(), 1); let encoded = String::from_utf8(serde_json::to_vec(&ranked).unwrap()).unwrap(); assert!(encoded.contains("KeyedSummaryBuild")); - assert!(encoded.contains("KeyedReadout")); + assert!(encoded.contains("KeyedEvaluation")); assert!( !encoded.contains("\"Rate\""), - "Rate must be supplied by its exact stored-state readout" + "Rate must be supplied by its exact stored-state evaluation" ); } - let dag = compile_post_asap_dag(candidate).unwrap(); + let dag = compile_physical_asap_dag(candidate).unwrap(); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); let build = dag.nodes.iter().find(|node| matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); let input_id = dag .edges .iter() @@ -377,8 +370,8 @@ fn check_direct_rate_topk(dynamic: bool) { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } + PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } } ) }) @@ -390,14 +383,13 @@ fn check_direct_rate_topk(dynamic: bool) { u64::from(raw.id.0), InputContract::bounded(raw_schema.clone()), )]), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], ) .unwrap(); let bytes = serde_json::to_vec(&raw_compiled).unwrap(); - let raw_compiled = serde_json::from_slice::< - asap_physical_operators::physical_planner::CompiledPhysicalDAG, - >(&bytes) - .unwrap(); + let raw_compiled = + serde_json::from_slice::(&bytes) + .unwrap(); // Each evaluation receives a complete raw window. A reset, a stopped // series and an expired leader must not retain last run's heap weights. for (end, series, expected) in [ @@ -475,7 +467,7 @@ fn check_direct_rate_topk(dynamic: bool) { let mut raw_scores = block_on(async { let mut scores = Vec::new(); let mut stream = physical_dag - .execute(&[u64::from(dag.root.0)], context) + .execute(&[u64::from(dag.roots[0].0)], context) .unwrap() .remove(0); while let Some(batch) = stream.next().await { @@ -525,7 +517,7 @@ fn check_direct_rate_topk(dynamic: bool) { u64::from(input_id.0), InputContract::bounded(schema.clone()), )]), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], ) .unwrap(); for (time, values, expected) in [ @@ -591,7 +583,7 @@ fn check_direct_rate_topk(dynamic: bool) { let mut scores = block_on(async { let mut scores = vec![]; let mut stream = physical_dag - .execute(&[u64::from(dag.root.0)], context) + .execute(&[u64::from(dag.roots[0].0)], context) .unwrap() .remove(0); while let Some(batch) = stream.next().await { @@ -629,13 +621,12 @@ fn check_direct_rate_topk(dynamic: bool) { // CountSketch; a raw metric does not establish the non-negative CMS contract. #[test] fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { - use asap_physical_operators::physical_planner::promql_rows::{ + use asap_executor::physical_planner::promql_rows::{ decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, }; let logical = lower_promql("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)).unwrap(); let root = Rc::new(with_series_identity(&logical).unwrap()); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &Evidence, @@ -649,21 +640,21 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { let selected = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CountSketchWithHeap") => { + Replacement::SubDAG(node) if candidate.rationale.contains("CountSketchWithHeap") => { Some(node) } _ => None, }) .expect("signed spatial TopK must expose CountSketch with heap"); - let dag = compile_post_asap_dag(selected).unwrap(); + let dag = compile_physical_asap_dag(selected).unwrap(); let raw = dag .nodes .iter() .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::TimeRange { .. } + PhysicalASAPOperatorPayload::Relational { + operator: planner_types::ir::export::NonASAPOpKind::TimeRange { .. } } ) }) @@ -672,19 +663,17 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { let program = compile( &dag, BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], ) .unwrap(); let snapshot_program = - asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( - selected, - ) - .unwrap(); + asap_executor::physical_planner::promql_rows::compile_current_series_evaluation(selected) + .unwrap(); let encoded: serde_json::Value = serde_json::from_slice(&serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); assert!(!encoded.to_string().contains("CurrentSeries")); assert!(encoded.to_string().contains("KeyedSummaryBuild")); - assert!(encoded.to_string().contains("KeyedReadout")); + assert!(encoded.to_string().contains("KeyedEvaluation")); for (values, expected, score) in [ ([100., 20.], "a", 100.), ([1., 20.], "b", 20.), @@ -759,113 +748,64 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { } } -/// Deployment-side lifecycle choice: every summary state of `candidate` is -/// continuously maintained, and the chosen lifecycles set execution timing. -fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDAG { - use asap_aware_mapping::{ - cost_model::{Cost, CostModel}, - enumerate_summary_maintenance_lifecycles, CostRate, Horizon, - SummaryMaintenanceCapabilities, SummaryMaintenanceLifecycleCapabilities, - SummaryMaintenanceLifecycleCostInputs, WorkloadDemand, - }; - use planner_types::workload::{ - DataArrival, Rate, RepeatedDemand, RepeatingEntry, RepetitionInterval, - }; - struct Costed; - impl CostModel for Costed { - fn rank_candidates( - &self, - _: &planner_types::pre_asap::agg_intent::AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - fn summary_maintenance_lifecycle_cost_inputs( - &self, - _: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(Cost(10.)), - maintenance_cost_per_update: Some(Cost(1.)), - summary_read_cost: Some(Cost(1.)), - retention_cost_rate: Some(CostRate(0.1)), - retirement_cost: Some(Cost(1.)), - } - } - fn summary_maintenance_capabilities( - &self, - _: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - SummaryMaintenanceCapabilities { - incremental_update: true, - merge: true, - delete: true, - } - } - } - const NOW_MS: u64 = 1_000_000; - let queries = QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: None, - repeating_queries: Some(vec![RepeatingEntry { - query: Query("topk by(job)(2, rate(m[1m]))".into()), - demand: RepeatedDemand::FixedInterval(RepetitionInterval(60_000)), - requirements: QueryRequirements::default(), - predictability: Predictability::Predictable { known_at: None }, - time_selection: TimeSelection::default(), - }]), - }; - let data = DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - ingestion_rate: WorkloadEvidence { - value: Some(Rate(1.)), - source: planner_types::workload::EvidenceSource::Observed, - observed_at_ms: Some(NOW_MS), - valid_for_ms: Some(60_000), - }, - ..Default::default() - }; - let lifecycles = enumerate_summary_maintenance_lifecycles( - Rc::clone(candidate), - WorkloadDemand::new_with_data(&queries, &data, &[0]), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &Costed, +/// Every summary state of `candidate` maintained at ingestion time, the +/// materialization a deployment would assign for a continuously served query: +/// each `SummaryAgg` and every input it consumes run at ingestion time, the +/// rest at query time. The phases are assigned on the exported DAG because +/// the candidate pins its finalize boundary to query time. +fn continuously_maintained_dag(candidate: &Rc) -> PhysicalASAPDAG { + use planner_types::ir::{apply_materialization_timings, MaterializationAssignment, TimingMemo}; + let timed = apply_materialization_timings( + candidate, + &MaterializationAssignment::all_query_time(), + &mut TimingMemo::new(), ) .unwrap(); - let choices = lifecycles - .deployments() + let dag = planner_types::ir::export::compile_physical_asap_dag(&timed).unwrap(); + let mut pending: Vec<_> = dag + .nodes .iter() - .map(|deployment| { - ( - deployment.post_asap_node_id, - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - ) + .filter(|node| matches!(node.payload, PhysicalASAPOperatorPayload::SummaryAgg { .. })) + .map(|node| node.id) + .collect(); + let mut ingestion = std::collections::HashSet::new(); + while let Some(id) = pending.pop() { + if ingestion.insert(id) { + pending.extend( + dag.edges + .iter() + .filter(|edge| edge.consumer == id) + .map(|edge| edge.producer), + ); + } + } + let phases = dag + .nodes + .iter() + .map(|node| { + let timing = if ingestion.contains(&node.id) { + ExecutionTiming::IngestionTime + } else { + ExecutionTiming::QueryTime + }; + (node.id, timing) }) - .collect::>(); - lifecycles - .select(&choices) - .unwrap() - .execution_timed_dag() - .unwrap() + .collect(); + dag.with_execution_phases(&phases).unwrap() } // A maintained heap over finalized per-series Rate is the fixed-window -// placement: lifecycle timing, not a separate candidate, puts it in precompute. +// placement: materialization timing, not a separate candidate, puts it in precompute. #[test] -fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { - use asap_physical_operators::physical_planner::{ - compile_candidate, promql_rows::with_series_identity, - }; +fn maintained_rate_heap_compiles_fixed_window_precompute() { + use asap_executor::physical_planner::{compile_candidate, promql_rows::with_series_identity}; let root = Rc::new( with_series_identity( &lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(), ) .unwrap(), ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &Evidence, @@ -874,7 +814,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .filter_map(|candidate| match candidate.replacement { - Replacement::Summary(root) if candidate.rationale.contains("WithHeap") => Some(root), + Replacement::SubDAG(root) if candidate.rationale.contains("WithHeap") => Some(root), _ => None, }) .collect::>(); @@ -887,7 +827,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, _), .. } @@ -900,7 +840,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(..), .. } @@ -914,18 +854,22 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { u64::from(state.id.0), InputContract::bounded(Arc::new(state.output_schema.clone())), )]), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], &[u64::from(heap.id.0)], ) .unwrap(); - let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&dag).unwrap(); + let exported = + asap_executor::physical_planner::promql_rows::compile_fixed_window_rate_aggregation( + &dag, + ) + .unwrap(); assert_eq!( serde_json::to_vec(&exported).unwrap(), serde_json::to_vec(&physical).unwrap() ); // Execute the selected split across a state serialization boundary. // Each run builds fresh weights from that window's counters. - let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDAG, + let execute = |plan: &asap_executor::physical_planner::CompiledPhysicalDAG, input: Batch, scope: Scope| { let id = plan.input_contracts().next().unwrap().0; @@ -949,7 +893,7 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { }) }; let (family, input, grouping) = match &state.payload { - PostAsapOperatorPayload::SummaryAgg { + PhysicalASAPOperatorPayload::SummaryAgg { family, input, grouping, @@ -975,10 +919,8 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { .zip(["a", "b", "c"]) .map(|(samples, label)| { let mut accumulator = - asap_physical_operators::factory::create_planner_accumulator( - family, input, grouping, - ) - .unwrap(); + asap_executor::factory::create_planner_accumulator(family, input, grouping) + .unwrap(); for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { accumulator.update_single(value, end - 60_000 + offset); } diff --git a/crates/frontend-common/Cargo.toml b/crates/frontend-common/Cargo.toml new file mode 100644 index 000000000..18f351e6c --- /dev/null +++ b/crates/frontend-common/Cargo.toml @@ -0,0 +1,11 @@ +[package] +name = "asap-frontend-common" +version = "0.1.0" +edition = "2021" + +# Shared front-end layer: the name-based `UnresolvedOp` tree every front end +# emits, and the resolver that binds it into the unified `OperatorNode` IR. +[dependencies] +asap-types = { path = "../types" } +serde = { version = "1", features = ["derive", "rc"] } +thiserror = "2" diff --git a/crates/frontend-common/src/lib.rs b/crates/frontend-common/src/lib.rs new file mode 100644 index 000000000..47684f668 --- /dev/null +++ b/crates/frontend-common/src/lib.rs @@ -0,0 +1,23 @@ +//! `asap-frontend-common` — the front-end-facing, name-based operator tree +//! and its resolver into the unified IR. +//! +//! A front end builds an [`UnresolvedOp`] tree (column references are +//! name-based [`ColumnRef`](asap_types::ir::scalar::ColumnRef)s) during its +//! own `interpret` step and calls [`resolve_root`], which binds every +//! reference to a positional `ColumnId` and returns the +//! [`OperatorNode`](asap_types::ir::OperatorNode) DAG. +//! +//! - [`unresolved`] — [`UnresolvedOp`] / [`UnresolvedScalar`]: the tree. +//! - [`schema_resolver`] — [`SchemaResolver`]: builds the binding schema of a +//! schemaless (PromQL) leaf from the names the query references. +//! - [`resolve`] — [`resolve_root`]: the bottom-up binding walk. + +pub mod resolve; +pub mod schema_resolver; +pub mod unresolved; + +pub use resolve::{resolve_expr, resolve_root, resolve_scalar_root, ResolveDAGError}; +pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; +pub use unresolved::{ + UnresolvedOp, UnresolvedPredicate, UnresolvedProjectItem, UnresolvedScalar, UnresolvedSortKey, +}; diff --git a/crates/frontend-common/src/resolve.rs b/crates/frontend-common/src/resolve.rs new file mode 100644 index 000000000..388203b8b --- /dev/null +++ b/crates/frontend-common/src/resolve.rs @@ -0,0 +1,1188 @@ +//! Resolve a front-end-emitted [`UnresolvedOp`] tree into the unified IR +//! ([`Rc`]): a single, shape-preserving, bottom-up walk that +//! binds every [`ColumnRef`] to a positional `ColumnId`. +//! +//! Every structural decision (reduction choice, window folds, heavy-hitter +//! recognition, ...) is the front end's; what is left here is the mechanical, +//! schema-dependent substitution. Children are resolved first; each child +//! becomes an `OperatorNode` whose derived `.schema` is the scope the parent's +//! own references resolve against, so a `JOIN`'s concatenated schema and a +//! cross-series aggregate's frozen-closed output bind to the right positions. +//! +//! Scope boundaries: `Join` / `SetOp` sides and the operators referenced from +//! scalar positions (`scalar(v)`, subqueries) are each bound as a root in +//! their own scope. A `BinaryOp` side is too, but additionally inherits the +//! label names its enclosing scope references (issue #52): the `job` in +//! `sum by (job)(a or b)` appears in neither side's own matchers. + +use asap_types::ir::schema::aggregate_schema::aggregate_output_schema; +use std::rc::Rc; + +use thiserror::Error; + +use asap_types::ir::operator::operator_properties::ConcatDiscriminatorKey; +use asap_types::ir::operator::{AggIntent, GroupKeys, Reduction}; +use asap_types::ir::scalar::column_resolution::resolve_group_keys_promql; +use asap_types::ir::scalar::{resolve_column_ref, resolve_column_refs, ColumnRef, ResolveError}; +use asap_types::ir::schema::{ColumnId, Schema, SchemaDerivationError}; +use asap_types::ir::{NonASAPOp, OperatorNode, Predicate, ProjectItem, ScalarExpr, SortKey}; + +use crate::schema_resolver::{collect_referenced_columns, SchemaResolver}; +use crate::unresolved::{UnresolvedOp, UnresolvedScalar, UnresolvedSortKey}; + +/// Errors from resolving an [`UnresolvedOp`] tree. +#[derive(Debug, Error)] +pub enum ResolveDAGError { + /// A column reference did not resolve against its in-scope schema. + #[error("column resolution failed: {0}")] + Resolve(#[from] ResolveError), + /// Deriving the schema of an already-resolved child failed (needed to + /// resolve positional column references against it). + #[error("schema derivation failed: {0}")] + Schema(#[from] SchemaDerivationError), +} + +use asap_types::ir::canonicalize::canonicalize; + +/// Resolve the whole tree rooted at `tree`: bind every `ColumnRef` to a +/// `ColumnId` via the [`SchemaResolver`], then canonicalize the result. +pub fn resolve_root(tree: &UnresolvedOp) -> Result, ResolveDAGError> { + resolve_root_with_inherited(tree, &[]) +} + +/// [`resolve_root`] with label names inherited from an enclosing scope seeded +/// into the leaf schema (a `BinaryOp` side, a scalar operand's operator). +fn resolve_root_with_inherited( + tree: &UnresolvedOp, + inherited: &[String], +) -> Result, ResolveDAGError> { + let fallback = SchemaResolver::new().resolve_schema_with_inherited(tree, inherited); + let root = resolve(tree, &fallback)?; + let root = canonicalize(root)?; + root.validate_structure()?; + Ok(root) +} + +/// Bind `tree` as a root in its own scope, inheriting from `enclosing` the +/// label names `tree` does not reference itself (issue #52). +fn resolve_nested_root( + tree: &UnresolvedOp, + enclosing: &Schema, +) -> Result, ResolveDAGError> { + let own = collect_referenced_columns(tree); + let inherited: Vec = inherited_names(enclosing) + .into_iter() + .filter(|n| !own.contains(n)) + .collect(); + resolve_root_with_inherited(tree, &inherited) +} + +fn node(op: NonASAPOp) -> Result, ResolveDAGError> { + Ok(OperatorNode::new_shared( + asap_types::ir::Operator::NonASAP(op), + )?) +} + +/// The generic substitution walk. `fallback` is the usage-derived schema a +/// schemaless `Scan` in this scope binds to. +fn resolve(tree: &UnresolvedOp, fallback: &Schema) -> Result, ResolveDAGError> { + use UnresolvedOp as U; + let expr = |e: &UnresolvedScalar, schema: &Schema| resolve_expr_in(e, schema, fallback); + let pred = |p: &UnresolvedScalar, schema: &Schema| { + Ok::<_, ResolveDAGError>(Predicate(expr(p, schema)?)) + }; + let sort_keys = |keys: &[UnresolvedSortKey], schema: &Schema| { + keys.iter() + .map(|k| { + Ok::<_, ResolveDAGError>(SortKey { + expr: expr(&k.expr, schema)?, + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + }) + .collect::, _>>() + }; + match tree { + U::Scan { + source, + predicates, + schema, + } => { + let schema = schema.clone().unwrap_or_else(|| fallback.clone()); + let predicates = predicates + .iter() + .map(|p| pred(&p.0, &schema)) + .collect::, _>>()?; + node(NonASAPOp::Scan { + source: source.clone(), + predicates, + schema, + }) + } + + // Row expressions have no input-column scope. + U::Values { rows, schema } => { + let empty = Schema::new(Vec::new()); + let rows = rows + .iter() + .map(|row| { + row.iter() + .map(|e| expr(e, &empty)) + .collect::, _>>() + }) + .collect::, _>>()?; + node(NonASAPOp::Values { + rows, + schema: schema.clone(), + }) + } + + // A scalar at an operator position has no child scope; in practice a + // literal, so `fallback` is never consulted for a column here. + U::PromqlScalarOp { + child, + scalar, + op, + scalar_left, + return_bool, + } => { + let child = resolve(child, fallback)?; + let child = if child.schema.closed { + child + } else { + asap_types::ir::schema_support::with_promql_series_identity(&child) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + }; + let scalar = resolve_expr(scalar, &Schema::default())?; + lower_scalar_vector(child, scalar, op, *scalar_left, *return_bool) + } + U::PromqlMap { + child, + sample, + drop_metric_name, + } => { + let child = resolve(child, fallback)?; + let child = if child.schema.closed { + child + } else { + asap_types::ir::schema_support::with_promql_series_identity(&child) + .map_err(SchemaDerivationError::InvalidScalarSignature)? + }; + let sample = resolve_expr(sample, &child.schema)?; + project_sample(child, sample, *drop_metric_name) + } + U::PromqlVectorFromScalar(inner) => { + node(NonASAPOp::PromqlVectorFromScalar(expr(inner, fallback)?)) + } + + U::PromqlRelabel { dst, value, child } => { + let child = resolve(child, fallback)?; + let value = expr(value, &child.schema)?; + node(NonASAPOp::PromqlRelabel { + dst: dst.clone(), + value, + child, + }) + } + + U::PromqlInfoEnrich { selector, child } => node(NonASAPOp::PromqlInfoEnrich { + selector: selector.clone(), + child: resolve(child, fallback)?, + }), + + U::PromqlSeriesSample { by, kind, child } => { + let child = resolve(child, fallback)?; + let by = resolve_group_keys(by, &child.schema)?; + node(NonASAPOp::PromqlSeriesSample { + by, + kind: *kind, + child, + }) + } + + U::Filter { pred: p, child } => { + let child = resolve(child, fallback)?; + let pred = pred(&p.0, &child.schema)?; + node(NonASAPOp::Filter { pred, child }) + } + + U::Project { + cols, + qualifier, + child, + } => { + let child = resolve(child, fallback)?; + let cols = cols + .iter() + .map(|item| { + Ok::<_, ResolveDAGError>(ProjectItem { + alias: item.alias.clone(), + expr: expr(&item.expr, &child.schema)?, + }) + }) + .collect::, _>>()?; + node(NonASAPOp::Project { + cols, + qualifier: qualifier.clone(), + child, + }) + } + + U::Aggregate { + reduction, + measures, + output_names, + filters, + having, + child, + } => { + let child = resolve(child, fallback)?; + let reduction = resolve_reduction(reduction, &child.schema)?; + let measures = measures + .iter() + .map(|m| resolve_agg_intent(m, &child.schema)) + .collect::, ResolveError>>()?; + let filters = filters + .iter() + .map(|p| p.as_ref().map(|p| pred(&p.0, &child.schema)).transpose()) + .collect::, _>>()?; + // HAVING is evaluated over the aggregate's own output. + let having = having + .as_ref() + .map(|h| { + let out_schema = aggregate_output_schema( + &child.schema, + &reduction, + &measures, + output_names, + )?; + pred(&h.0, &out_schema) + }) + .transpose()?; + node(NonASAPOp::Aggregate { + reduction, + measures, + output_names: output_names.clone(), + filters, + having, + child, + }) + } + + U::Dedup { cols, child } => { + let child = resolve(child, fallback)?; + let cols = resolve_column_refs(cols, &child.schema)?; + node(NonASAPOp::Dedup { cols, child }) + } + + U::Concat { + children, + discriminator_unique_key, + } => { + let children = children + .iter() + .map(|c| resolve(c, fallback)) + .collect::, _>>()?; + // Resolved against the first branch's own output schema — the one + // `output_schema`'s `Concat` arm derives the merged schema from. + let discriminator_unique_key = discriminator_unique_key + .as_ref() + .map(|key| { + let schema = &children + .first() + .ok_or(SchemaDerivationError::EmptyConcat)? + .schema; + Ok::<_, ResolveDAGError>(ConcatDiscriminatorKey::new( + resolve_column_ref(key.discriminator(), schema)?, + resolve_column_refs(key.inner_key(), schema)?, + )) + }) + .transpose()?; + node(NonASAPOp::Concat { + children, + discriminator_unique_key, + }) + } + + U::Join { + kind, + pred: p, + left, + right, + } => { + // Each branch is bound independently (different leaves / label + // sets); the predicate sees left ++ right. + let left = resolve_root_with_inherited(left, &[])?; + let right = resolve_root_with_inherited(right, &[])?; + let mut concat = left.schema.clone(); + concat.fields.extend(right.schema.fields.iter().cloned()); + let pred = pred(&p.0, &concat)?; + node(NonASAPOp::Join { + kind: kind.clone(), + pred, + left, + right, + }) + } + + U::SetOp { + kind, + all, + left, + right, + } => node(NonASAPOp::SetOp { + kind: kind.clone(), + all: *all, + left: resolve_root_with_inherited(left, &[])?, + right: resolve_root_with_inherited(right, &[])?, + }), + + U::Sort { + keys, + partition_by, + child, + } => { + let child = resolve(child, fallback)?; + let keys = sort_keys(keys, &child.schema)?; + let partition_by = resolve_group_keys(partition_by, &child.schema)?; + node(NonASAPOp::Sort { + keys, + partition_by, + child, + }) + } + + U::Limit { + n, + offset, + partition_by, + child, + } => { + let child = resolve(child, fallback)?; + let partition_by = resolve_group_keys(partition_by, &child.schema)?; + node(NonASAPOp::Limit { + n: *n, + offset: *offset, + partition_by, + child, + }) + } + + U::PromqlSubquery { + range, + resolution, + child, + } => node(NonASAPOp::PromqlSubquery { + range: *range, + resolution: *resolution, + child: resolve(child, fallback)?, + }), + + U::TimeRange { range, kind, child } => node(NonASAPOp::TimeRange { + range: *range, + kind: *kind, + child: resolve(child, fallback)?, + }), + + U::TimeShift { shift, child } => node(NonASAPOp::TimeShift { + shift: *shift, + child: resolve(child, fallback)?, + }), + + U::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + child, + } => { + let child = resolve(child, fallback)?; + let args = args + .iter() + .map(|a| expr(a, &child.schema)) + .collect::, _>>()?; + let partition_by = resolve_group_keys(partition_by, &child.schema)?; + let order_by = sort_keys(order_by, &child.schema)?; + node(NonASAPOp::SQLWindowFunc { + func: func.clone(), + args, + partition_by, + order_by, + frame: frame.clone(), + output_name: output_name.clone(), + child, + }) + } + + U::BinaryOp { + operator, + return_bool, + lhs, + rhs, + } => { + // The two sides may scan different metrics with different label + // sets, so each resolves against its OWN bound schema — but still + // sees the label names the enclosing scope references (issue #52). + // The inherited set is computed over the whole `BinaryOp`, so one + // side's own labels are not conjured into the other. + let own = collect_referenced_columns(tree); + let inherited: Vec = inherited_names(fallback) + .into_iter() + .filter(|n| !own.contains(n)) + .collect(); + node(NonASAPOp::BinaryOp { + operator: operator.clone(), + return_bool: *return_bool, + lhs: resolve_root_with_inherited(lhs, &inherited)?, + rhs: resolve_root_with_inherited(rhs, &inherited)?, + }) + } + } +} + +/// The label names an enclosing scope's schema carries beyond the `(ts, +/// value)` floor. +fn inherited_names(schema: &Schema) -> Vec { + schema + .fields + .iter() + .filter(|c| c.name != "ts" && c.name != "value") + .map(|c| c.name.clone()) + .collect() +} + +/// Resolve a name-based scalar expression against `schema`. Operators it +/// reads (`scalar(v)`, subqueries) are bound as roots in their own scope, +/// inheriting `schema`'s label names. +pub fn resolve_expr( + expr: &UnresolvedScalar, + schema: &Schema, +) -> Result { + resolve_expr_in(expr, schema, schema) +} + +/// [`resolve_expr`] where the operators the expression reads inherit from +/// `enclosing` (the owning root's fallback schema) rather than from `schema`. +fn resolve_expr_in( + expr: &UnresolvedScalar, + schema: &Schema, + enclosing: &Schema, +) -> Result { + use UnresolvedScalar as S; + let bx = |e: &UnresolvedScalar| -> Result, ResolveDAGError> { + Ok(Box::new(resolve_expr_in(e, schema, enclosing)?)) + }; + let each = |es: &[UnresolvedScalar]| -> Result, ResolveDAGError> { + es.iter() + .map(|e| resolve_expr_in(e, schema, enclosing)) + .collect() + }; + let op = |o: &UnresolvedOp| resolve_nested_root(o, enclosing); + Ok(match expr { + S::Column(c) => ScalarExpr::Column(resolve_column_ref(c, schema)?), + S::Literal(s) => ScalarExpr::Literal(s.clone()), + S::EvalTimestamp => ScalarExpr::EvalTimestamp, + S::CurrentTimestamp => ScalarExpr::CurrentTimestamp, + S::Negative { expr, semantics } => ScalarExpr::Negative { + expr: bx(expr)?, + semantics: *semantics, + }, + S::Compare { + left, + op, + right, + semantics, + } => ScalarExpr::Compare { + left: bx(left)?, + op: op.clone(), + right: bx(right)?, + semantics: *semantics, + }, + S::BoolAnd(v) => ScalarExpr::BoolAnd(each(v)?), + S::BoolOr(v) => ScalarExpr::BoolOr(each(v)?), + S::Not(e) => ScalarExpr::Not(bx(e)?), + S::IsNull(e) => ScalarExpr::IsNull(bx(e)?), + S::IsNotNull(e) => ScalarExpr::IsNotNull(bx(e)?), + S::Cast { expr, to, try_cast } => ScalarExpr::Cast { + expr: bx(expr)?, + to: to.clone(), + try_cast: *try_cast, + }, + S::InList { + expr, + list, + negated, + } => ScalarExpr::InList { + expr: bx(expr)?, + list: each(list)?, + negated: *negated, + }, + S::FunctionCall { name, args } => ScalarExpr::FunctionCall { + name: name.clone(), + args: each(args)?, + }, + S::Arithmetic { + op, + left, + right, + semantics, + } => ScalarExpr::Arithmetic { + op: op.clone(), + left: bx(left)?, + right: bx(right)?, + semantics: *semantics, + }, + S::Case { + operand, + branches, + else_expr, + } => ScalarExpr::Case { + operand: operand.as_deref().map(bx).transpose()?, + branches: branches + .iter() + .map(|(w, t)| { + Ok(( + resolve_expr_in(w, schema, enclosing)?, + resolve_expr_in(t, schema, enclosing)?, + )) + }) + .collect::, ResolveDAGError>>()?, + else_expr: else_expr.as_deref().map(bx).transpose()?, + }, + S::PromqlScalarFromVector(o) => ScalarExpr::PromqlScalarFromVector(op(o)?), + S::ScalarSubquery(o) => ScalarExpr::ScalarSubquery(op(o)?), + S::Exists { subquery, negated } => ScalarExpr::Exists { + subquery: op(subquery)?, + negated: *negated, + }, + S::InSubquery { + expr, + subquery, + negated, + } => ScalarExpr::InSubquery { + expr: bx(expr)?, + subquery: op(subquery)?, + negated: *negated, + }, + }) +} + +/// Resolve name-based group keys positionally, preserving `by`/`without`. +fn resolve_group_keys( + keys: &GroupKeys, + schema: &Schema, +) -> Result, ResolveError> { + let ids = resolve_column_refs(keys.keys(), schema)?; + Ok(if keys.is_without() { + GroupKeys::without(ids) + } else { + GroupKeys::by(ids) + }) +} + +/// Resolve a name-based reduction. Uses [`resolve_group_keys_promql`] rather +/// than the strict [`resolve_group_keys`]: a key absent from a **closed** +/// schema (the output of a nested cross-series aggregate that collapsed the +/// label) is provably absent from every row, so PromQL drops it from the +/// grouping rather than rejecting the query (issue #53) — `sum(sum by (group) +/// (m)) by (job)`. SQL `GROUP BY` keys are always present, so the lenient +/// path is a no-op difference there. +fn resolve_reduction( + reduction: &Reduction, + schema: &Schema, +) -> Result, ResolveError> { + Ok(match reduction { + Reduction::Reduce(by) => { + let ids = resolve_group_keys_promql(by.keys(), schema)?; + Reduction::Reduce(if by.is_without() { + GroupKeys::without(ids) + } else { + GroupKeys::by(ids) + }) + } + Reduction::PerEntity => Reduction::PerEntity, + }) +} + +/// Resolve a name-based aggregate intent: every `col: Option` +/// resolves to `Option` (`None` stays `None`, the sample-value +/// convention); every other field carries through unchanged. +fn resolve_agg_intent( + intent: &AggIntent, + schema: &Schema, +) -> Result, ResolveError> { + let col = |c: &Option| -> Result, ResolveError> { + c.as_ref() + .map(|r| resolve_column_ref(r, schema)) + .transpose() + }; + Ok(match intent { + AggIntent::Count { accuracy } => AggIntent::Count { + accuracy: accuracy.clone(), + }, + AggIntent::PearsonCorr { left, right } => AggIntent::PearsonCorr { + left: resolve_column_ref(left, schema)?, + right: resolve_column_ref(right, schema)?, + }, + AggIntent::Sum { col: c } => AggIntent::Sum { col: col(c)? }, + AggIntent::Min { col: c } => AggIntent::Min { col: col(c)? }, + AggIntent::Max { col: c } => AggIntent::Max { col: col(c)? }, + AggIntent::Avg { col: c } => AggIntent::Avg { col: col(c)? }, + AggIntent::StdDev { col: c, population } => AggIntent::StdDev { + col: col(c)?, + population: *population, + }, + AggIntent::Variance { col: c, population } => AggIntent::Variance { + col: col(c)?, + population: *population, + }, + AggIntent::Quantile { + col: c, + q, + accuracy, + } => AggIntent::Quantile { + col: col(c)?, + q: *q, + accuracy: accuracy.clone(), + }, + AggIntent::TopK { k, accuracy } => AggIntent::TopK { + k: *k, + accuracy: accuracy.clone(), + }, + AggIntent::Cardinality { cols, accuracy } => AggIntent::Cardinality { + cols: cols + .iter() + .map(|c| resolve_column_ref(c, schema)) + .collect::>()?, + accuracy: accuracy.clone(), + }, + AggIntent::FrequencyL2 { col: c, accuracy } => AggIntent::FrequencyL2 { + col: col(c)?, + accuracy: accuracy.clone(), + }, + AggIntent::FrequencyEntropy { col: c, accuracy } => AggIntent::FrequencyEntropy { + col: col(c)?, + accuracy: accuracy.clone(), + }, + AggIntent::Rate => AggIntent::Rate, + AggIntent::IRate => AggIntent::IRate, + AggIntent::Increase => AggIntent::Increase, + AggIntent::Changes => AggIntent::Changes, + AggIntent::Delta => AggIntent::Delta, + AggIntent::IDelta => AggIntent::IDelta, + AggIntent::Deriv => AggIntent::Deriv, + AggIntent::Resets => AggIntent::Resets, + AggIntent::PredictLinear { seconds } => AggIntent::PredictLinear { seconds: *seconds }, + AggIntent::DoubleExpSmoothing { smoothing, trend } => AggIntent::DoubleExpSmoothing { + smoothing: *smoothing, + trend: *trend, + }, + AggIntent::HistogramCount => AggIntent::HistogramCount, + AggIntent::HistogramSum => AggIntent::HistogramSum, + AggIntent::HistogramAvg => AggIntent::HistogramAvg, + AggIntent::HistogramStdDev => AggIntent::HistogramStdDev, + AggIntent::HistogramStdVar => AggIntent::HistogramStdVar, + AggIntent::HistogramFraction { lower, upper } => AggIntent::HistogramFraction { + lower: *lower, + upper: *upper, + }, + AggIntent::HistogramQuantile { q, le } => AggIntent::HistogramQuantile { + q: *q, + le: resolve_column_ref(le, schema)?, + }, + AggIntent::Math(f) => AggIntent::Math(f.clone()), + AggIntent::Absent => AggIntent::Absent, + AggIntent::AbsentOverTime => AggIntent::AbsentOverTime, + AggIntent::PresentOverTime => AggIntent::PresentOverTime, + AggIntent::TimeFn(f) => AggIntent::TimeFn(*f), + AggIntent::Group => AggIntent::Group, + AggIntent::CountValues { label } => AggIntent::CountValues { + label: label.clone(), + }, + AggIntent::LastOverTime => AggIntent::LastOverTime, + AggIntent::FirstOverTime => AggIntent::FirstOverTime, + AggIntent::MadOverTime => AggIntent::MadOverTime, + AggIntent::TsOfMinOverTime => AggIntent::TsOfMinOverTime, + AggIntent::TsOfMaxOverTime => AggIntent::TsOfMaxOverTime, + AggIntent::TsOfFirstOverTime => AggIntent::TsOfFirstOverTime, + AggIntent::TsOfLastOverTime => AggIntent::TsOfLastOverTime, + AggIntent::Extension { ext_kind, payload } => AggIntent::Extension { + ext_kind: ext_kind.clone(), + payload: payload.clone(), + }, + }) +} + +/// Resolve a standalone scalar in an empty column scope; plan reads retain their own scope. +pub fn resolve_scalar_root(tree: &UnresolvedScalar) -> Result { + let resolved = resolve_expr(tree, &Schema::default())?; + resolved.scalar_type(&Schema::default())?; + Ok(resolved) +} + +fn lower_scalar_vector( + child: Rc, + scalar: ScalarExpr, + op: &asap_types::ir::operator::BinaryOpKind, + scalar_left: bool, + return_bool: bool, +) -> Result, ResolveDAGError> { + use asap_types::ir::operator::BinaryOpKind; + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::schema::DataType; + use asap_types::ir::ExprSemantics; + let value = child + .schema + .column_id("value") + .or_else(|| { + child + .schema + .fields + .iter() + .enumerate() + .filter(|(i, f)| { + Some(*i) != child.schema.time_index + && matches!(f.plain_dtype(), Some(DataType::Float64 | DataType::Int64)) + }) + .map(|(i, _)| i) + .next_back() + }) + .ok_or_else(|| { + SchemaDerivationError::InvalidScalarSignature("vector has no numeric sample".into()) + })?; + let sample = ScalarExpr::Column(value); + let (left, right) = if scalar_left { + (scalar, sample) + } else { + (sample, scalar) + }; + let semantics = ExprSemantics::Promql; + let return_bool = return_bool || matches!(op, BinaryOpKind::CompareBool(_)); + let computed = match op { + BinaryOpKind::Arithmetic(op) => ScalarExpr::Arithmetic { + op: op.clone(), + left: Box::new(left), + right: Box::new(right), + semantics, + }, + BinaryOpKind::Compare(op) | BinaryOpKind::CompareBool(op) => { + let predicate = ScalarExpr::Compare { + op: op.clone(), + left: Box::new(left), + right: Box::new(right), + semantics, + }; + if !return_bool { + return node(NonASAPOp::Filter { + child, + pred: Predicate(predicate), + }); + } + ScalarExpr::Case { + operand: None, + branches: vec![(predicate, ScalarExpr::Literal(ScalarValue::Float64(1.0)))], + else_expr: Some(Box::new(ScalarExpr::Literal(ScalarValue::Float64(0.0)))), + } + } + BinaryOpKind::Set(_) => { + return Err(SchemaDerivationError::InvalidScalarSignature( + "set operators require two vectors".into(), + ) + .into()) + } + }; + project_sample(child, computed, true) +} + +fn project_sample( + child: Rc, + computed: ScalarExpr, + drop_metric_name: bool, +) -> Result, ResolveDAGError> { + let value = asap_types::ir::scalar::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + &child.schema, + )?; + let cols = child + .schema + .fields + .iter() + .enumerate() + .filter(|(_, f)| !drop_metric_name || f.name != "__name__") + .map(|(i, f)| { + let expr = if i == value { + computed.clone() + } else if drop_metric_name && f.name == asap_types::ir::schema::PROMQL_SERIES_IDENTITY { + ScalarExpr::FunctionCall { + name: "promql_drop_metric_name".into(), + args: vec![ScalarExpr::Column(i)], + } + } else { + ScalarExpr::Column(i) + }; + ProjectItem { + alias: Some(f.name.clone()), + expr, + } + }) + .collect(); + node(NonASAPOp::Project { + child, + cols, + qualifier: None, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::unresolved::UnresolvedPredicate; + use asap_types::ir::operator::{ + BinaryOpKind, JoinKind, PromQLVectorSetOpKind, Source, VectorMatch, + }; + use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; + use asap_types::ir::schema::{DataType, Field}; + use asap_types::ir::BinaryOperator; + use asap_types::ir::ExprSemantics; + use asap_types::types::AccuracyTarget; + + fn scan(metric: &str) -> UnresolvedOp { + UnresolvedOp::Scan { + source: Source::TimeSeries { + metric: metric.into(), + }, + predicates: vec![], + schema: None, + } + } + + fn named(n: &str) -> UnresolvedScalar { + UnresolvedScalar::Column(ColumnRef::Named(n.into())) + } + + fn eq_lit(col: UnresolvedScalar, v: &str) -> UnresolvedScalar { + UnresolvedScalar::Compare { + left: Box::new(col), + op: CompareOpKind::Eq, + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Utf8(v.into()))), + semantics: ExprSemantics::Promql, + } + } + + fn binary(kind: BinaryOpKind, vector_match: Option) -> BinaryOperator { + BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + } + } + + // Both sides resolve with qualifiers; an unknown right input is an error. + #[test] + fn resolve_pearson_corr_inputs() { + let schema = Schema::new(vec![ + Field::plain("x", DataType::Float64, true).with_table("a"), + Field::plain("x", DataType::Float64, true).with_table("b"), + ]); + let intent = AggIntent::PearsonCorr { + left: ColumnRef::Qualified { + table: "a".into(), + name: "x".into(), + }, + right: ColumnRef::Qualified { + table: "b".into(), + name: "x".into(), + }, + }; + assert_eq!( + resolve_agg_intent(&intent, &schema).unwrap(), + AggIntent::PearsonCorr { left: 0, right: 1 } + ); + let missing = AggIntent::PearsonCorr { + left: ColumnRef::Qualified { + table: "a".into(), + name: "x".into(), + }, + right: ColumnRef::Named("missing".into()), + }; + assert!(resolve_agg_intent(&missing, &schema).is_err()); + } + + // Every leg resolves independently, qualifiers included; one unknown leg + // fails rather than silently shortening the tuple. + #[test] + fn resolve_distinct_tuple_columns() { + let schema = Schema::new(vec![ + Field::plain("k", DataType::Int64, true).with_table("a"), + Field::plain("k", DataType::Int64, true).with_table("b"), + ]); + let qualified = |table: &str| ColumnRef::Qualified { + table: table.into(), + name: "k".into(), + }; + let intent = AggIntent::Cardinality { + cols: vec![qualified("b"), qualified("a")], + accuracy: AccuracyTarget::Exact, + }; + assert_eq!( + resolve_agg_intent(&intent, &schema).unwrap(), + AggIntent::Cardinality { + cols: vec![1, 0], + accuracy: AccuracyTarget::Exact, + } + ); + let missing = AggIntent::Cardinality { + cols: vec![qualified("a"), ColumnRef::Named("missing".into())], + accuracy: AccuracyTarget::Exact, + }; + assert!(resolve_agg_intent(&missing, &schema).is_err()); + } + + // ` > `: the bridged literal comes through unchanged, the + // vector side binds positionally, the `VectorMatch` survives untouched, and + // the node's schema follows the vector side. + #[test] + fn scalar_comparison_preserves_vector_values_and_labels() { + let unresolved = UnresolvedOp::PromqlScalarOp { + child: Rc::new(scan("up")), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(1.0)), + op: BinaryOpKind::Compare(CompareOpKind::Gt), + scalar_left: true, + return_bool: false, + }; + let resolved = resolve_root(&unresolved).unwrap(); + let NonASAPOp::Filter { + child, + pred: Predicate(ScalarExpr::Compare { left, right, .. }), + } = resolved.expect_non_asap() + else { + panic!("expected Filter") + }; + assert_eq!(**left, ScalarExpr::literal_f64(1.0)); + assert_eq!( + **right, + ScalarExpr::Column(child.schema.column_id("value").unwrap()) + ); + assert_eq!(resolved.schema, child.schema); + assert!(resolved.schema.has_promql_series_identity()); + } + + // A `Concat` discriminator column referenced nowhere else, over a + // schemaless first branch, resolves to the branch's own positional ids. + #[test] + fn resolve_root_seeds_and_resolves_an_otherwise_unreferenced_discriminator_column() { + let unresolved = UnresolvedOp::concat_with_discriminator( + vec![scan("m"), scan("m")], + ColumnRef::Named("phi".into()), + vec![ColumnRef::Named("host".into())], + ); + + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Concat { + children, + discriminator_unique_key, + } = resolved.expect_non_asap() + else { + panic!("expected a resolved Concat, got {resolved:?}"); + }; + let schema = &children[0].schema; + let key = discriminator_unique_key + .as_ref() + .expect("discriminator key survives resolution"); + assert_eq!(*key.discriminator(), schema.column_id("phi").unwrap()); + assert_eq!( + key.inner_key().to_vec(), + vec![schema.column_id("host").unwrap()] + ); + } + + // `sum by (job)(a or b)`: each `BinaryOp` side binds in its own scope but + // inherits the enclosing aggregate's group key (issue #52). + #[test] + fn binary_op_sides_inherit_enclosing_group_keys() { + let unresolved = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("job".into())]), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(UnresolvedOp::BinaryOp { + operator: binary(BinaryOpKind::Set(PromQLVectorSetOpKind::Or), None), + return_bool: false, + lhs: Rc::new(scan("a")), + rhs: Rc::new(scan("b")), + }), + }; + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Aggregate { + reduction, child, .. + } = resolved.expect_non_asap() + else { + panic!("expected Aggregate"); + }; + let NonASAPOp::BinaryOp { lhs, rhs, .. } = child.expect_non_asap() else { + panic!("expected BinaryOp"); + }; + let job = lhs.schema.column_id("job").expect("lhs sees job"); + assert_eq!(rhs.schema.column_id("job"), Some(job)); + assert_eq!(reduction.expect_reduce().keys(), &[job]); + assert_eq!(resolved.schema.fields[0].name, "job"); + } + + // HAVING binds against the aggregate's output, not its input. + #[test] + fn having_resolves_against_aggregate_output() { + let input = Schema::new(vec![ + Field::plain("k", DataType::Utf8, false), + Field::plain("v", DataType::Float64, false), + ]); + let unresolved = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("k".into())]), + measures: vec![AggIntent::Sum { + col: Some(ColumnRef::Named("v".into())), + }], + output_names: vec!["total".into()], + filters: vec![], + having: Some(UnresolvedPredicate(UnresolvedScalar::Compare { + left: Box::new(named("total")), + op: CompareOpKind::Gt, + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Float64(1.0))), + semantics: ExprSemantics::Sql, + })), + child: Rc::new(UnresolvedOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Some(input), + }), + }; + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Aggregate { + having: Some(Predicate(ScalarExpr::Compare { left, .. })), + .. + } = resolved.expect_non_asap() + else { + panic!("expected Aggregate with HAVING"); + }; + assert_eq!(**left, ScalarExpr::Column(1)); + assert_eq!(resolved.schema.fields[1].name, "total"); + } + + // A join predicate binds against left ++ right; a qualified reference + // picks the right side even when both inputs share the column name. + #[test] + fn join_predicate_resolves_against_left_then_right() { + let side = |table: &str| UnresolvedOp::Scan { + source: Source::Table { + table_ref: table.into(), + }, + predicates: vec![], + schema: Some(Schema::new(vec![ + Field::plain("k", DataType::Int64, false).with_table(table) + ])), + }; + let qualified = |table: &str| { + UnresolvedScalar::Column(ColumnRef::Qualified { + table: table.into(), + name: "k".into(), + }) + }; + let unresolved = UnresolvedOp::Join { + kind: JoinKind::Inner, + pred: UnresolvedPredicate(UnresolvedScalar::Compare { + left: Box::new(qualified("b")), + op: CompareOpKind::Eq, + right: Box::new(qualified("a")), + semantics: ExprSemantics::Sql, + }), + left: Rc::new(side("a")), + right: Rc::new(side("b")), + }; + let resolved = resolve_root(&unresolved).expect("resolves"); + let NonASAPOp::Join { + pred: Predicate(ScalarExpr::Compare { left, right, .. }), + .. + } = resolved.expect_non_asap() + else { + panic!("expected Join"); + }; + assert_eq!(**left, ScalarExpr::Column(1)); + assert_eq!(**right, ScalarExpr::Column(0)); + } + + // `m * scalar(x{a="1"})`: the operator inside the scalar operand is bound + // as a root in its own scope — its matcher label seeds its own leaf, not + // the vector side's. + #[test] + fn scalar_from_vector_operand_binds_in_its_own_scope() { + let x = UnresolvedOp::Scan { + source: Source::TimeSeries { metric: "x".into() }, + predicates: vec![UnresolvedPredicate(eq_lit(named("a"), "1"))], + schema: None, + }; + let unresolved = UnresolvedOp::PromqlScalarOp { + child: Rc::new(scan("m")), + scalar: UnresolvedScalar::PromqlScalarFromVector(Rc::new(x)), + op: BinaryOpKind::Arithmetic(asap_types::ir::scalar::ArithmeticOpKind::Mul), + scalar_left: false, + return_bool: false, + }; + let resolved = resolve_root(&unresolved).unwrap(); + let NonASAPOp::Project { + child: lhs, cols, .. + } = resolved.expect_non_asap() + else { + panic!("expected Project") + }; + assert!(lhs.schema.column_id("a").is_none()); + let ScalarExpr::Arithmetic { right, .. } = &cols[1].expr else { + panic!("expected arithmetic") + }; + let ScalarExpr::PromqlScalarFromVector(inner) = right.as_ref() else { + panic!("expected scalar(v)") + }; + let a = inner + .schema + .column_id("a") + .expect("own matcher label seeded"); + let NonASAPOp::Scan { predicates, .. } = inner.expect_non_asap() else { + panic!("expected Scan"); + }; + let Predicate(ScalarExpr::Compare { left, .. }) = &predicates[0] else { + panic!("expected Compare"); + }; + assert_eq!(**left, ScalarExpr::Column(a)); + assert_eq!(resolved.schema.fields.len(), lhs.schema.fields.len()); + } + + // PromQL grouping drops a key provably absent from a closed input (#53): + // `sum(sum by (group)(m)) by (job)`. + #[test] + fn nested_aggregate_drops_absent_promql_group_key() { + let inner = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("group".into())]), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(scan("m")), + }; + let outer = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![ColumnRef::Named("job".into())]), + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(inner), + }; + let resolved = resolve_root(&outer).expect("resolves"); + let NonASAPOp::Aggregate { reduction, .. } = resolved.expect_non_asap() else { + panic!("expected Aggregate"); + }; + assert!(reduction.expect_reduce().keys().is_empty()); + } +} diff --git a/crates/frontend-common/src/schema_resolver.rs b/crates/frontend-common/src/schema_resolver.rs new file mode 100644 index 000000000..1f08542fb --- /dev/null +++ b/crates/frontend-common/src/schema_resolver.rs @@ -0,0 +1,445 @@ +//! The **SchemaResolver** — name resolution as an explicit pass. +//! +//! [`SchemaResolver::resolve_schema`] produces the complete, self-contained +//! [`Schema`] every `ColumnId` in a schemaless leaf's scope indexes into, so +//! positional resolution in [`resolve`](crate::resolve) is total. +//! +//! The default [`UsageDerivedCatalog`] knows nothing — every schema is derived +//! purely from the query's own usage. That is the honest state for the +//! observability domain (metric label sets are open-ended). A registry-backed +//! `SchemaCatalog` is future work; only the catalog impl swaps when it lands. + +use asap_types::ir::operator::{AggIntent, GroupKeys, Reduction}; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{DataType, Field, Schema}; + +use crate::unresolved::{UnresolvedOp, UnresolvedScalar}; + +/// The DB / source-schema metadata source — resolves a source (metric / +/// table) name to its known columns. Distinct from `Scan.schema`, which is +/// the *resolved* binding schema this feeds. Even a registry-backed PromQL +/// catalog yields an **open** schema: a metric's labels are per-series and +/// time-varying, so the registry is a superset hint, not a per-row contract. +pub trait SchemaCatalog { + /// Columns known for `source`. `None` when unknown — the resolver then + /// falls back to a usage-derived column set. + fn columns_for(&self, source: &str) -> Option>; +} + +/// The default catalog: knows nothing. +pub struct UsageDerivedCatalog; + +impl SchemaCatalog for UsageDerivedCatalog { + fn columns_for(&self, _source: &str) -> Option> { + None + } +} + +/// The explicit name-resolution pass. +pub struct SchemaResolver { + catalog: C, +} + +impl Default for SchemaResolver { + fn default() -> Self { + Self::new() + } +} + +impl SchemaResolver { + pub fn new() -> Self { + Self { + catalog: UsageDerivedCatalog, + } + } +} + +impl SchemaResolver { + pub fn with_catalog(catalog: C) -> Self { + Self { catalog } + } + + /// The complete [`Schema`] in scope for a query rooted at `tree`: the + /// time axis, the synthetic `value` column, and one column per distinct + /// name referenced anywhere in the tree. + pub fn resolve_schema(&self, tree: &UnresolvedOp) -> Schema { + self.resolve_schema_with_inherited(tree, &[]) + } + + /// Like [`resolve_schema`](Self::resolve_schema), but also seeds + /// `inherited` label names referenced by an **enclosing** scope rather + /// than by `tree` itself. This is how an independently-bound `BinaryOp` + /// side still sees an outer aggregate's group keys — the `__name__` / + /// `job` in `sum by (__name__)(a or b)`, which appear in neither side's + /// own matchers (issue #52). + pub fn resolve_schema_with_inherited( + &self, + tree: &UnresolvedOp, + inherited: &[String], + ) -> Schema { + let mut columns: Vec = leftmost_scan_name(tree) + .and_then(|name| self.catalog.columns_for(name)) + .unwrap_or_else(default_leaf_columns); + + // Ensure the (ts, value) floor is present. + for floor in default_leaf_columns() { + if !columns.iter().any(|c| c.name == floor.name) { + columns.push(floor); + } + } + + // One column per referenced-but-unknown name, plus the inherited ones. + let referenced = collect_referenced_columns(tree); + for name in referenced.iter().chain(inherited) { + if !columns.iter().any(|c| c.name == *name) { + columns.push(Field::plain(name.clone(), DataType::Utf8, true)); + } + } + + let time_index = columns.iter().position(|c| c.name == "ts"); + Schema { + fields: columns, + time_index, + unique_keys: Vec::new(), + // Usage-derived (schemaless PromQL): the metric's full label set is + // open and runtime-only, so this lists only what the query references. + closed: false, + } + } +} + +/// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. +fn default_leaf_columns() -> Vec { + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ] +} + +/// Push a `ColumnRef`'s bare name (the schema-seedable identifier). `Qualified` +/// collapses to its `name`; `SampleValue`/`Wildcard` carry no name. +fn push_ref_name(c: &ColumnRef, out: &mut Vec) { + match c { + ColumnRef::Named(n) => out.push(n.clone()), + ColumnRef::Qualified { name, .. } => out.push(name.clone()), + ColumnRef::SampleValue | ColumnRef::Wildcard => {} + } +} + +/// The leftmost `Scan`'s source name, following the relational skeleton only +/// (never the operators referenced from scalar positions: those are bound in +/// their own scope). +fn leftmost_scan_name(tree: &UnresolvedOp) -> Option<&str> { + use asap_types::ir::operator::Source; + use UnresolvedOp as U; + match tree { + U::Scan { source, .. } => Some(match source { + Source::TimeSeries { metric } => metric.as_str(), + Source::Table { table_ref } => table_ref.as_str(), + }), + U::Values { .. } | U::PromqlVectorFromScalar(_) => None, + U::PromqlMap { child, .. } + | U::PromqlScalarOp { child, .. } + | U::PromqlRelabel { child, .. } + | U::PromqlInfoEnrich { child, .. } + | U::PromqlSeriesSample { child, .. } + | U::Filter { child, .. } + | U::Project { child, .. } + | U::Aggregate { child, .. } + | U::Dedup { child, .. } + | U::Sort { child, .. } + | U::Limit { child, .. } + | U::PromqlSubquery { child, .. } + | U::TimeRange { child, .. } + | U::TimeShift { child, .. } + | U::SQLWindowFunc { child, .. } => leftmost_scan_name(child), + U::Concat { children, .. } => children.first().and_then(|c| leftmost_scan_name(c)), + U::Join { left, .. } | U::SetOp { left, .. } | U::BinaryOp { lhs: left, .. } => { + leftmost_scan_name(left) + } + } +} + +/// Every distinct column name referenced anywhere in `tree` that resolves +/// positionally — every place a front end puts a name-based reference: +/// `Scan.predicates`, `Aggregate`'s `reduction`/`having`/per-measure `col`, +/// `Dedup.cols`, `PromqlSeriesSample.by`, `Filter.pred`, `Project.cols`, +/// `Sort`/`Limit`/`SQLWindowFunc` keys, `Join.pred`, `PromqlRelabel.value`, +/// `Concat.discriminator_unique_key`. Operators referenced from scalar +/// positions (`scalar(v)`, subqueries) are walked too, as the old +/// `PromqlScalarFromVector` operator child was. Sorted and deduplicated. +pub fn collect_referenced_columns(tree: &UnresolvedOp) -> Vec { + use UnresolvedOp as U; + fn named(expr: &UnresolvedScalar, out: &mut Vec) { + for c in expr.columns_referenced() { + push_ref_name(c, out); + } + for op in expr.operator_refs() { + walk(op, out); + } + } + fn group_keys(g: &GroupKeys, out: &mut Vec) { + g.keys().iter().for_each(|k| push_ref_name(k, out)); + } + fn measure_cols(measures: &[AggIntent], out: &mut Vec) { + for m in measures { + for c in m.input_cols() { + push_ref_name(&c, out); + } + } + } + fn walk(node: &UnresolvedOp, out: &mut Vec) { + match node { + U::Scan { predicates, .. } => { + for p in predicates { + named(&p.0, out); + } + } + U::Values { rows, .. } => { + for e in rows.iter().flatten() { + named(e, out); + } + } + U::Aggregate { + reduction, + measures, + having, + child, + .. + } => { + if let Reduction::Reduce(by) = reduction { + group_keys(by, out); + } + measure_cols(measures, out); + if let Some(h) = having { + named(&h.0, out); + } + walk(child, out); + } + U::Dedup { cols, child } => { + cols.iter().for_each(|c| push_ref_name(c, out)); + walk(child, out); + } + U::PromqlSeriesSample { by, child, .. } => { + group_keys(by, out); + walk(child, out); + } + U::Filter { pred, child } => { + named(&pred.0, out); + walk(child, out); + } + U::Project { cols, child, .. } => { + for item in cols { + named(&item.expr, out); + } + walk(child, out); + } + U::Sort { + keys, + partition_by, + child, + } => { + for k in keys { + named(&k.expr, out); + } + group_keys(partition_by, out); + walk(child, out); + } + U::Limit { + partition_by, + child, + .. + } => { + group_keys(partition_by, out); + walk(child, out); + } + U::SQLWindowFunc { + args, + partition_by, + order_by, + child, + .. + } => { + for a in args { + named(a, out); + } + group_keys(partition_by, out); + for k in order_by { + named(&k.expr, out); + } + walk(child, out); + } + U::PromqlRelabel { value, child, .. } => { + named(value, out); + walk(child, out); + } + U::Join { + pred, left, right, .. + } => { + named(&pred.0, out); + walk(left, out); + walk(right, out); + } + U::PromqlVectorFromScalar(inner) => named(inner, out), + U::PromqlMap { child, .. } + | U::PromqlScalarOp { child, .. } + | U::PromqlInfoEnrich { child, .. } + | U::PromqlSubquery { child, .. } + | U::TimeRange { child, .. } + | U::TimeShift { child, .. } => walk(child, out), + U::Concat { + children, + discriminator_unique_key, + } => { + // An own-field `ColumnRef` must be seeded like `Dedup.cols`, or + // a discriminator column referenced nowhere else in the tree is + // absent from the fallback schema and fails `NotFound` later. + if let Some(key) = discriminator_unique_key { + push_ref_name(key.discriminator(), out); + key.inner_key().iter().for_each(|c| push_ref_name(c, out)); + } + children.iter().for_each(|c| walk(c, out)); + } + U::SetOp { left, right, .. } => { + walk(left, out); + walk(right, out); + } + U::BinaryOp { lhs, rhs, .. } => { + walk(lhs, out); + walk(rhs, out); + } + } + } + let mut out: Vec = Vec::new(); + walk(tree, &mut out); + out.sort(); + out.dedup(); + out +} + +#[cfg(test)] +mod tests { + use std::rc::Rc; + + use asap_types::ir::operator::{AggIntent, Reduction, Source}; + + use super::*; + use crate::unresolved::UnresolvedSortKey; + + fn src(name: &str) -> UnresolvedOp { + UnresolvedOp::Scan { + source: Source::TimeSeries { + metric: name.into(), + }, + predicates: vec![], + schema: None, + } + } + + // Both correlation inputs seed the usage-derived schema. + #[test] + fn pearson_corr_inputs_seed_usage_derived_schema() { + let tree = UnresolvedOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![AggIntent::PearsonCorr { + left: ColumnRef::Named("x".into()), + right: ColumnRef::Named("y".into()), + }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(src("m")), + }; + assert_eq!(collect_referenced_columns(&tree), vec!["x", "y"]); + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("x").is_some()); + assert!(schema.column_id("y").is_some()); + } + + // A bare source gets exactly the (ts, value) floor. + #[test] + fn bare_source_yields_ts_value_floor() { + let schema = SchemaResolver::new().resolve_schema(&src("m")); + assert_eq!(schema.fields.len(), 2); + assert_eq!(schema.fields[0].name, "ts"); + assert_eq!(schema.fields[1].name, "value"); + assert_eq!(schema.time_index, Some(0)); + } + + // Per-group ranking keys (`topk by (host)` → `Sort.partition_by`) are + // seeded into the usage-derived leaf so they resolve positionally. + #[test] + fn sort_partition_keys_land_in_schema() { + let tree = UnresolvedOp::Sort { + keys: vec![UnresolvedSortKey { + expr: UnresolvedScalar::Column(ColumnRef::SampleValue), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), + child: Rc::new(src("hits")), + }; + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("host").is_some()); + } + + // `Limit.partition_by` (PromQL `topk by (..)`) is seeded like `Sort`'s. + #[test] + fn limit_partition_keys_land_in_schema() { + let tree = UnresolvedOp::Limit { + n: Some(3), + offset: 0, + partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), + child: Rc::new(src("hits")), + }; + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("host").is_some()); + } + + // A `Concat`'s discriminator key columns, even ones referenced nowhere + // else, are seeded like `Dedup.cols` (issue #228 review). + #[test] + fn concat_discriminator_key_is_seeded_into_the_resolver_schema() { + let tree = UnresolvedOp::concat_with_discriminator( + vec![src("m")], + ColumnRef::Named("phi".into()), + vec![ColumnRef::Named("host".into())], + ); + let schema = SchemaResolver::new().resolve_schema(&tree); + assert!(schema.column_id("phi").is_some(), "discriminator seeded"); + assert!(schema.column_id("host").is_some(), "inner_key seeded"); + } + + // Inherited names are seeded alongside the tree's own references; plain + // `resolve_schema` does not conjure them (issue #52). + #[test] + fn inherited_names_are_seeded_alongside_referenced() { + let schema = + SchemaResolver::new().resolve_schema_with_inherited(&src("m"), &["__name__".into()]); + assert!(schema.column_id("__name__").is_some()); + let plain = SchemaResolver::new().resolve_schema(&src("m")); + assert!(plain.column_id("__name__").is_none()); + } + + // A catalog-known source supplies its base columns, typed as the catalog says. + #[test] + fn custom_catalog_supplies_base_columns() { + struct FixedCatalog; + impl SchemaCatalog for FixedCatalog { + fn columns_for(&self, source: &str) -> Option> { + (source == "known").then(|| { + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("datacenter", DataType::Utf8, false), + ] + }) + } + } + let schema = SchemaResolver::with_catalog(FixedCatalog).resolve_schema(&src("known")); + let dc = schema + .column_id("datacenter") + .and_then(|id| schema.fields.get(id)); + assert!(matches!(dc, Some(c) if !c.nullable)); + } +} diff --git a/crates/frontend-common/src/unresolved.rs b/crates/frontend-common/src/unresolved.rs new file mode 100644 index 000000000..eb6ab3587 --- /dev/null +++ b/crates/frontend-common/src/unresolved.rs @@ -0,0 +1,375 @@ +//! The front-end-emitted, name-based operator tree: a mirror of the unified +//! IR ([`NonASAPOp`](asap_types::ir::NonASAPOp) / [`ScalarExpr`](asap_types::ir::ScalarExpr)) +//! before name resolution. +//! +//! Differences from the resolved IR, and nothing else: +//! - every `ColumnId` is a name-based [`ColumnRef`]; +//! - `Scan.schema` is `Option` — a front end knows the schema only for +//! a catalog-backed SQL leaf; `None` (PromQL) defers to the +//! [`SchemaResolver`](crate::schema_resolver::SchemaResolver); +//! - children are `Rc` rather than `Rc` — no +//! derived schema exists yet. + +use std::rc::Rc; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; + +use asap_types::ir::operator::operator_properties::ConcatDiscriminatorKey; +use asap_types::ir::operator::{ + AggIntent, GroupKeys, InfoMatcher, JoinKind, Reduction, RelationalSetOpKind, SampleKind, + Source, TimeShift, WindowFrame, WindowFuncKind, +}; +use asap_types::ir::scalar::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; +use asap_types::ir::schema::{DataType, Schema}; +use asap_types::ir::BinaryOperator; +use asap_types::ir::{ExprSemantics, TimeRangeKind}; + +/// A row-level filter predicate (WHERE clause / PromQL label matcher). +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct UnresolvedPredicate(pub UnresolvedScalar); + +/// One item in a SELECT projection list. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct UnresolvedProjectItem { + pub alias: Option, + pub expr: UnresolvedScalar, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct UnresolvedSortKey { + pub expr: UnresolvedScalar, + pub ascending: bool, + pub nulls_first: bool, +} + +/// A name-based scalar expression; see +/// [`ScalarExpr`](asap_types::ir::ScalarExpr) for the meaning of each variant. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum UnresolvedScalar { + Column(ColumnRef), + Literal(ScalarValue), + Negative { + expr: Box, + semantics: ExprSemantics, + }, + Compare { + left: Box, + op: CompareOpKind, + right: Box, + semantics: ExprSemantics, + }, + BoolAnd(Vec), + BoolOr(Vec), + Not(Box), + IsNull(Box), + IsNotNull(Box), + Cast { + expr: Box, + to: DataType, + try_cast: bool, + }, + InList { + expr: Box, + list: Vec, + negated: bool, + }, + FunctionCall { + name: String, + args: Vec, + }, + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + semantics: ExprSemantics, + }, + Case { + operand: Option>, + branches: Vec<(UnresolvedScalar, UnresolvedScalar)>, + else_expr: Option>, + }, + CurrentTimestamp, + EvalTimestamp, + /// PromQL `scalar(v)`. The operator is resolved as a root in its own scope. + PromqlScalarFromVector(Rc), + ScalarSubquery(Rc), + Exists { + subquery: Rc, + negated: bool, + }, + InSubquery { + expr: Box, + subquery: Rc, + negated: bool, + }, +} + +/// The name-based operator tree; see [`NonASAPOp`](asap_types::ir::NonASAPOp) +/// for the meaning of each variant. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum UnresolvedOp { + Scan { + source: Source, + predicates: Vec, + /// `Some` for a catalog-backed (SQL) leaf; `None` defers to the + /// usage-derived schema resolver. + schema: Option, + }, + Values { + rows: Vec>, + schema: Schema, + }, + Filter { + pred: UnresolvedPredicate, + child: Rc, + }, + Project { + cols: Vec, + qualifier: Option, + child: Rc, + }, + Aggregate { + reduction: Reduction, + measures: Vec>, + output_names: Vec, + filters: Vec>, + having: Option, + child: Rc, + }, + Join { + kind: JoinKind, + pred: UnresolvedPredicate, + left: Rc, + right: Rc, + }, + SetOp { + kind: RelationalSetOpKind, + all: bool, + left: Rc, + right: Rc, + }, + Concat { + children: Vec>, + discriminator_unique_key: Option>, + }, + Dedup { + cols: Vec, + child: Rc, + }, + Sort { + keys: Vec, + partition_by: GroupKeys, + child: Rc, + }, + Limit { + n: Option, + offset: usize, + partition_by: GroupKeys, + child: Rc, + }, + BinaryOp { + operator: BinaryOperator, + return_bool: bool, + lhs: Rc, + rhs: Rc, + }, + SQLWindowFunc { + func: WindowFuncKind, + args: Vec, + partition_by: GroupKeys, + order_by: Vec, + frame: Option, + output_name: String, + child: Rc, + }, + TimeRange { + range: Duration, + kind: TimeRangeKind, + child: Rc, + }, + TimeShift { + shift: TimeShift, + child: Rc, + }, + PromqlVectorFromScalar(UnresolvedScalar), + PromqlRelabel { + dst: String, + value: UnresolvedScalar, + child: Rc, + }, + PromqlInfoEnrich { + selector: Vec, + child: Rc, + }, + PromqlSeriesSample { + by: GroupKeys, + kind: SampleKind, + child: Rc, + }, + PromqlSubquery { + range: Duration, + resolution: Option, + child: Rc, + }, + /// Bind the complete vector schema before lowering to Project or Filter. + /// Frontend-only expansion to a projection preserving the complete series identity. + PromqlMap { + child: Rc, + sample: UnresolvedScalar, + drop_metric_name: bool, + }, + PromqlScalarOp { + child: Rc, + scalar: UnresolvedScalar, + op: asap_types::ir::operator::BinaryOpKind, + scalar_left: bool, + return_bool: bool, + }, +} + +impl UnresolvedScalar { + /// The direct scalar sub-expressions (not the operators this expression + /// reads — see [`operator_refs`](Self::operator_refs)). + pub fn children(&self) -> Vec<&UnresolvedScalar> { + use UnresolvedScalar::*; + match self { + Column(_) + | Literal(_) + | CurrentTimestamp + | EvalTimestamp + | PromqlScalarFromVector(_) + | ScalarSubquery(_) + | Exists { .. } => vec![], + Negative { expr, .. } + | Not(expr) + | IsNull(expr) + | IsNotNull(expr) + | Cast { expr, .. } + | InSubquery { expr, .. } => vec![expr], + Compare { left, right, .. } | Arithmetic { left, right, .. } => vec![left, right], + BoolAnd(parts) | BoolOr(parts) => parts.iter().collect(), + InList { expr, list, .. } => { + let mut v = vec![expr.as_ref()]; + v.extend(list.iter()); + v + } + FunctionCall { args, .. } => args.iter().collect(), + Case { + operand, + branches, + else_expr, + } => { + let mut v = Vec::new(); + if let Some(op) = operand { + v.push(op.as_ref()); + } + for (when, then) in branches { + v.push(when); + v.push(then); + } + if let Some(e) = else_expr { + v.push(e.as_ref()); + } + v + } + } + } + + /// Every column referenced in this expression, not inside the operators + /// it reads (those have their own scope). + pub fn columns_referenced(&self) -> Vec<&ColumnRef> { + let mut out = Vec::new(); + self.collect_columns(&mut out); + out + } + + fn collect_columns<'a>(&'a self, out: &mut Vec<&'a ColumnRef>) { + if let UnresolvedScalar::Column(c) = self { + out.push(c); + } + for child in self.children() { + child.collect_columns(out); + } + } + + /// The operators this expression (transitively) reads. + pub fn operator_refs(&self) -> Vec<&Rc> { + let mut out = Vec::new(); + self.collect_operator_refs(&mut out); + out + } + + fn collect_operator_refs<'a>(&'a self, out: &mut Vec<&'a Rc>) { + use UnresolvedScalar::*; + match self { + PromqlScalarFromVector(op) | ScalarSubquery(op) => out.push(op), + Exists { subquery, .. } | InSubquery { subquery, .. } => out.push(subquery), + _ => {} + } + for child in self.children() { + child.collect_operator_refs(out); + } + } +} + +impl UnresolvedOp { + /// An ordinary `Concat` (no unique-key claim). + pub fn concat(children: Vec) -> Self { + UnresolvedOp::Concat { + children: children.into_iter().map(Rc::new).collect(), + discriminator_unique_key: None, + } + } + + /// A `Concat` whose output carries the caller-proven compound unique key + /// `(discriminator, inner_key)`. Nothing verifies the claim. + pub fn concat_with_discriminator( + children: Vec, + discriminator: ColumnRef, + inner_key: Vec, + ) -> Self { + UnresolvedOp::Concat { + children: children.into_iter().map(Rc::new).collect(), + discriminator_unique_key: Some(ConcatDiscriminatorKey::new(discriminator, inner_key)), + } + } + + /// Every scalar expression this operator owns. + pub fn scalar_exprs(&self) -> Vec<&UnresolvedScalar> { + use UnresolvedOp::*; + match self { + Scan { predicates, .. } => predicates.iter().map(|p| &p.0).collect(), + Values { rows, .. } => rows.iter().flatten().collect(), + Filter { pred, .. } | Join { pred, .. } => vec![&pred.0], + Project { cols, .. } => cols.iter().map(|c| &c.expr).collect(), + Aggregate { + filters, having, .. + } => filters + .iter() + .flatten() + .chain(having.iter()) + .map(|p| &p.0) + .collect(), + Sort { keys, .. } => keys.iter().map(|k| &k.expr).collect(), + SQLWindowFunc { args, order_by, .. } => args + .iter() + .chain(order_by.iter().map(|k| &k.expr)) + .collect(), + PromqlVectorFromScalar(e) => vec![e], + PromqlScalarOp { scalar, .. } => vec![scalar], + PromqlMap { sample, .. } => vec![sample], + PromqlRelabel { value, .. } => vec![value], + SetOp { .. } + | Concat { .. } + | Dedup { .. } + | Limit { .. } + | BinaryOp { .. } + | TimeRange { .. } + | TimeShift { .. } + | PromqlInfoEnrich { .. } + | PromqlSeriesSample { .. } + | PromqlSubquery { .. } => vec![], + } + } +} diff --git a/crates/frontend-metricsql/Cargo.toml b/crates/frontend-metricsql/Cargo.toml index 8fa341b80..cba166f7c 100644 --- a/crates/frontend-metricsql/Cargo.toml +++ b/crates/frontend-metricsql/Cargo.toml @@ -5,5 +5,6 @@ edition = "2021" [dependencies] asap-types = { path = "../types" } +asap-frontend-common = { path = "../frontend-common" } metricsql_parser = { path = "../metricsql-parser-vendored" } thiserror = "2" diff --git a/crates/frontend-metricsql/src/lib.rs b/crates/frontend-metricsql/src/lib.rs index 26a034c43..2524cd150 100644 --- a/crates/frontend-metricsql/src/lib.rs +++ b/crates/frontend-metricsql/src/lib.rs @@ -1,12 +1,15 @@ -//! MetricsQL AST to canonical `QueryExpr` frontend. +//! MetricsQL AST → the name-based `UnresolvedOp` tree → the unified operator DAG. use std::{rc::Rc, time::Duration}; -use asap_types::pre_asap::{ - resolve_root, AggIntent, ArithmeticOpKind, BinaryOpKind, ColumnRef, CompareOpKind, GroupKeys, - Predicate, PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, - UnresolvedQueryExpr as U, +use asap_frontend_common::{ + resolve_root, UnresolvedOp as U, UnresolvedPredicate, UnresolvedScalar, }; +use asap_types::ir::operator::{ + AggIntent, BinaryOpKind, GroupKeys, PromQLVectorSetOpKind, Reduction, Source, +}; +use asap_types::ir::scalar::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; +use asap_types::ir::{BinaryOperator, ExprSemantics, OperatorNode, TimeRangeKind}; use asap_types::types::AccuracyTarget; use metricsql_parser::ast::{AggregateModifier, DurationExpr, Expr, MetricExpr, RollupExpr}; use metricsql_parser::functions::{AggregateFunction, BuiltinFunction, RollupFunction}; @@ -33,10 +36,31 @@ pub fn canonical_metricsql(query: &str) -> Result { Ok(parse_metricsql(query)?.to_string()) } -pub fn lower_metricsql(query: &str, accuracy: AccuracyTarget) -> Result { +pub fn lower_metricsql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, MetricsqlError> { + match lower_metricsql_query(query, accuracy)? { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + _ => Err(unsupported("scalar root: use lower_metricsql_query")), + } +} + +/// Lower scalar constants without fabricating a relational operator. +pub fn lower_metricsql_query( + query: &str, + accuracy: AccuracyTarget, +) -> Result { let ast = parse_metricsql(query)?; + if let Expr::NumberLiteral(number) = &ast { + return Ok(asap_types::ir::QueryRoot::Scalar( + asap_types::ir::ScalarExpr::literal_f64(number.value), + )); + } let unresolved = Lowerer { accuracy }.lower(&ast)?; - resolve_root(&unresolved).map_err(|e| MetricsqlError::Resolve(e.to_string())) + resolve_root(&unresolved) + .map(asap_types::ir::QueryRoot::Operator) + .map_err(|e| MetricsqlError::Resolve(e.to_string())) } struct Lowerer { @@ -50,12 +74,16 @@ impl Lowerer { Expr::Rollup(e) => self.rollup(e), Expr::Function(e) => self.function(e), Expr::Aggregation(e) => self.aggregate(e), - Expr::NumberLiteral(e) => Ok(U::promql_scalar(e.value)), - Expr::UnaryOperator(e) => Ok(U::BinaryOp { + Expr::NumberLiteral(_) => { + Err(unsupported("scalar root requires lower_metricsql_query")) + } + // Vector negation is `x * -1` (as in the PromQL front end). + Expr::UnaryOperator(e) => Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(&e.expr)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(-1.0)), op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(self.lower(&e.expr)?), - rhs: Rc::new(U::promql_scalar(-1.0)), - vector_match: None, + scalar_left: false, + return_bool: false, }), Expr::BinaryOperator(e) => self.binary(e), Expr::Parens(e) if e.expressions.len() == 1 => self.lower(&e.expressions[0]), @@ -83,7 +111,7 @@ impl Lowerer { }, predicates: filters .into_iter() - .map(|f| Predicate(Rc::new(matcher(f)))) + .map(|f| UnresolvedPredicate(matcher(f))) .collect(), schema: None, }) @@ -101,6 +129,7 @@ impl Lowerer { None => Ok(child), Some(window) => Ok(U::TimeRange { range: duration(window)?, + kind: TimeRangeKind::Range, child: Rc::new(child), }), } @@ -258,12 +287,41 @@ impl Lowerer { return Err(unsupported(format!("MetricsQL operator `{}`", expr.op))) } }; - Ok(U::BinaryOp { + for (scalar, vector, scalar_left) in [ + (&expr.left, &expr.right, true), + (&expr.right, &expr.left, false), + ] { + if let Expr::NumberLiteral(n) = scalar.as_ref() { + return Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(vector)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(n.value)), + op, + scalar_left, + return_bool: false, + }); + } + } + Ok(binary_op( op, - lhs: Rc::new(self.lower(&expr.left)?), - rhs: Rc::new(self.lower(&expr.right)?), + self.lower(&expr.left)?, + self.lower(&expr.right)?, + )) + } +} + +/// A `BinaryOp` with default matching; MetricsQL modifiers (including `bool`) +/// are rejected before reaching here. +fn binary_op(kind: BinaryOpKind, lhs: U, rhs: U) -> U { + U::BinaryOp { + operator: BinaryOperator { + kind, vector_match: None, - }) + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), } } @@ -282,17 +340,22 @@ fn aggregate(reduction: Reduction, intent: AggIntent, chil } } -fn matcher(filter: &LabelFilter) -> U { +fn matcher(filter: &LabelFilter) -> UnresolvedScalar { let op = match filter.op { LabelFilterOp::Equal => CompareOpKind::Eq, LabelFilterOp::NotEqual => CompareOpKind::Ne, LabelFilterOp::RegexEqual => CompareOpKind::Regex, LabelFilterOp::RegexNotEqual => CompareOpKind::NotRegex, }; - U::Compare { - left: Rc::new(U::Column(ColumnRef::Named(filter.label.clone()))), + UnresolvedScalar::Compare { + left: Box::new(UnresolvedScalar::Column(ColumnRef::Named( + filter.label.clone(), + ))), op, - right: Rc::new(U::Literal(ScalarValue::Utf8(filter.value.clone()))), + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Utf8( + filter.value.clone(), + ))), + semantics: ExprSemantics::Promql, } } diff --git a/crates/frontend-metricsql/tests/lowering.rs b/crates/frontend-metricsql/tests/lowering.rs index 3add8ee2a..1c8b59d50 100644 --- a/crates/frontend-metricsql/tests/lowering.rs +++ b/crates/frontend-metricsql/tests/lowering.rs @@ -1,62 +1,65 @@ +use std::rc::Rc; use std::time::Duration; use asap_frontend_metricsql::{ canonical_metricsql, lower_metricsql, parse_metricsql, MetricsqlError, }; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::operator::{AggIntent, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; use asap_types::types::AccuracyTarget; -fn lower(query: &str) -> QueryExpr { +fn lower(query: &str) -> Rc { lower_metricsql(query, AccuracyTarget::Epsilon(0.01)).unwrap() } #[test] fn selector_range_aggregate_and_call_share_the_canonical_shape() { let query = r#"sum by (job) (rate(http_requests_total{status=~"5.."}[5m]))"#; - let dag = lower(query); - let QueryExpr::Aggregate { + let tree = lower(query); + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = dag + } = tree.expect_non_asap() else { panic!("expected outer aggregate"); }; - assert_eq!(reduction, Reduction::by(vec![2])); + assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected rate aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected range"); }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(child.as_ref(), QueryExpr::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) + matches!(child.expect_non_asap(), NonASAPOp::Scan { source: Source::TimeSeries { metric }, predicates, .. } if metric == "http_requests_total" && predicates.len() == 1) ); } #[test] fn default_rollup_with_explicit_range_is_last_over_time() { - let dag = lower("default_rollup(cpu_usage[5m])"); - let QueryExpr::Aggregate { + let tree = lower("default_rollup(cpu_usage[5m])"); + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = dag + } = tree.expect_non_asap() else { panic!("expected aggregate"); }; - assert_eq!(reduction, Reduction::PerEntity); + assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::LastOverTime])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if *range == Duration::from_secs(300)) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, .. } + if *range == Duration::from_secs(300) && *kind == TimeRangeKind::Range) ); } @@ -147,7 +150,13 @@ fn metricsql_multi_argument_aggregates_fail_closed() { #[test] fn supported_parameterized_functions_require_their_exact_arity() { let quantile = lower("quantile(0.9, requests_total)"); - assert!(matches!(quantile, QueryExpr::Aggregate { .. })); + assert!(matches!( + quantile.expect_non_asap(), + NonASAPOp::Aggregate { .. } + )); let rollup = lower("quantile_over_time(0.9, requests_total[5m])"); - assert!(matches!(rollup, QueryExpr::Aggregate { .. })); + assert!(matches!( + rollup.expect_non_asap(), + NonASAPOp::Aggregate { .. } + )); } diff --git a/crates/frontend-promql/Cargo.toml b/crates/frontend-promql/Cargo.toml index b108e32c7..0ac525de5 100644 --- a/crates/frontend-promql/Cargo.toml +++ b/crates/frontend-promql/Cargo.toml @@ -3,10 +3,12 @@ name = "asap-frontend-promql" version = "0.1.0" edition = "2021" -# PromQL front end: L1 (parse) → L2 relational, then the shared L2→L3 converter -# — both in asap-types. Pulls the PromQL parser only — never DataFusion. +# PromQL front end: parse → the shared name-based `UnresolvedOp` tree +# (asap-frontend-common) → the unified IR. Pulls the PromQL parser only — +# never DataFusion. [dependencies] asap-types = { path = "../types" } +asap-frontend-common = { path = "../frontend-common" } # Shared ProjectASAP parser contract. Keep this immutable revision aligned # with backend parsing and treat newly parsed functions as unsupported until @@ -14,7 +16,8 @@ asap-types = { path = "../types" } promql-parser = { git = "https://github.com/ProjectASAP/promql-parser", rev = "9fede7eecca923c9882fe256484d00d37f8706cb" } [dev-dependencies] -asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-plan-selection = { path = "../plan-selection" } +asap-logical-optimizer = { path = "../logical-optimizer" } # Corpus tests live in domain folders (observability) but must still be declared # as test targets — Cargo only auto-discovers `.rs` files directly in `tests/`. diff --git a/crates/frontend-promql/src/error.rs b/crates/frontend-promql/src/error.rs index 6f996b11d..ebbcf49c2 100644 --- a/crates/frontend-promql/src/error.rs +++ b/crates/frontend-promql/src/error.rs @@ -1,12 +1,12 @@ use std::fmt; -use asap_types::pre_asap::ResolveDAGError; +use asap_frontend_common::ResolveDAGError; use asap_types::workload::WorkloadError; -/// Errors from lowering a PromQL query (parse → the canonical, unresolved -/// DAG, built directly → -/// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved DAG, issue #179). +/// Errors from lowering a PromQL query (parse → the name-based unresolved +/// tree, built directly → +/// [`resolve_root`](asap_frontend_common::resolve_root) binds it to the +/// unified operator DAG, issue #179). /// /// Carries no DataFusion type — the PromQL front end never depends on the SQL /// stack. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` diff --git a/crates/frontend-promql/src/histogram.rs b/crates/frontend-promql/src/histogram.rs index f3f971a55..ecb8cd2c4 100644 --- a/crates/frontend-promql/src/histogram.rs +++ b/crates/frontend-promql/src/histogram.rs @@ -1,18 +1,10 @@ //! Sample-type metadata for the `histogram_quantile` discrimination (issue #79). //! -//! `histogram_quantile(φ, m)` has two lowerings: exact interpolation over -//! classic cumulative `le` buckets (`AggIntent::HistogramQuantile`, **not** -//! sketch-able) versus the generic sketch-able `Quantile` (native histograms / -//! raw samples, which post-ASAP binding can approximate to an accuracy -//! target). The true -//! signal is the argument's **sample type**, which query structure only -//! *proxies* — see the structural `is_classic_bucket_arg` heuristic, whose -//! false-positive (`…_bucket`-named non-histogram) and false-negative -//! (suffix-less classic histogram) cases this metadata fixes. -//! -//! A client that knows its sample types supplies a [`HistogramCatalog`]; it is -//! consulted first, and the structural heuristic remains the fallback when a -//! metric is undeclared. +//! Classic cumulative buckets use exact interpolation. The explicitly declared +//! `RawSamples` extension permits generic quantile sketches; it is not standard +//! PromQL histogram semantics. Native samples are rejected until the IR has a +//! native histogram sample type. Undeclared metrics require classic bucket +//! evidence (`by (le)`, a `_bucket` metric, or an `le` matcher). use std::cell::RefCell; use std::collections::HashMap; @@ -25,7 +17,7 @@ pub enum HistogramKind { /// distribution can't be reconstructed from them, so it is **not** /// sketch-able: `histogram_quantile` is exact bucket interpolation. ClassicBucket, - /// Native (exponential) histogram — sketch-able to an accuracy target. + /// Native histogram samples; currently rejected because the IR lacks their type. Native, /// Raw float samples the client retains — sketch-able. This is the case the /// generic `Quantile` lowering exists for (a client holding raw samples can @@ -37,7 +29,7 @@ impl HistogramKind { /// Whether `histogram_quantile` over this kind lowers to the sketch-able /// generic `Quantile` (`true`) rather than exact bucket interpolation. pub fn is_sketchable(self) -> bool { - !matches!(self, HistogramKind::ClassicBucket) + matches!(self, HistogramKind::RawSamples) } } @@ -106,9 +98,9 @@ mod tests { use super::*; #[test] - fn only_classic_buckets_are_not_sketchable() { + fn only_explicit_raw_samples_are_sketchable() { assert!(!HistogramKind::ClassicBucket.is_sketchable()); - assert!(HistogramKind::Native.is_sketchable()); + assert!(!HistogramKind::Native.is_sketchable()); assert!(HistogramKind::RawSamples.is_sketchable()); } diff --git a/crates/frontend-promql/src/lib.rs b/crates/frontend-promql/src/lib.rs index 31b169359..e7d99fe4c 100644 --- a/crates/frontend-promql/src/lib.rs +++ b/crates/frontend-promql/src/lib.rs @@ -1,25 +1,26 @@ -//! PromQL front end: parse (via `promql-parser`) → the canonical, unresolved -//! shape, built directly (issue #179) → [`resolve_root`]. +//! PromQL front end: parse (via `promql-parser`) → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree, built directly +//! in canonical shape (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved -//! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational DAG; `resolve_root` runs the -//! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. -//! Depends on the PromQL parser only — never on the SQL / DataFusion stack. +//! `resolve_root` runs the +//! [`SchemaResolver`](asap_frontend_common::SchemaResolver) for positional +//! name resolution and returns the unified +//! [`OperatorNode`](asap_types::ir::OperatorNode) DAG. Depends on the PromQL +//! parser only — never on the SQL / DataFusion stack. pub mod error; pub mod histogram; pub mod promql; -use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use std::rc::Rc; + +use asap_types::ir::OperatorNode; use asap_types::workload::{DurationMs, PlanningWorkload, QueryLanguage, WorkloadError}; pub use error::PromqlError; pub use histogram::{HistogramCatalog, HistogramKind}; -/// Lower every normalized PromQL workload entry to a plan-ready `QueryExpr`. +/// Lower every normalized PromQL workload entry to a plan-ready operator DAG. /// /// PromQL workloads must declare a non-zero `data_ingestion_interval`; it is /// injected around each bare instant selector. Explicit range selectors keep @@ -29,7 +30,7 @@ pub use histogram::{HistogramCatalog, HistogramKind}; pub fn lower_promql_workload( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { lower_promql_workload_inner(workload, now_ms) } @@ -39,15 +40,47 @@ pub fn lower_promql_workload_with_histograms( workload: &PlanningWorkload, histograms: HistogramCatalog, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { let _guard = histogram::CatalogGuard::install(histograms); lower_promql_workload_inner(workload, now_ms) } +/// Lower scalar and vector query roots without introducing constant operators. +pub fn lower_promql_query_workload( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms) +} + +pub fn lower_promql_query_workload_with_histograms( + workload: &PlanningWorkload, + histograms: HistogramCatalog, + now_ms: u64, +) -> Result, PromqlError> { + let _guard = histogram::CatalogGuard::install(histograms); + lower_promql_query_workload_inner(workload, now_ms) +} + fn lower_promql_workload_inner( workload: &PlanningWorkload, now_ms: u64, -) -> Result, PromqlError> { +) -> Result>, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms)? + .into_iter() + .map(|root| match root { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + asap_types::ir::QueryRoot::Scalar(_) => Err(PromqlError::UnsupportedFeature( + "scalar root: use lower_promql_query_workload".into(), + )), + }) + .collect() +} + +fn lower_promql_query_workload_inner( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { if !matches!(workload.query_workload.language, QueryLanguage::PromQL) { return Err(PromqlError::WrongLanguage(format!( "{:?}", @@ -66,12 +99,12 @@ fn lower_promql_workload_inner( .query_workload .entries() .map(|entry| { - let unresolved = promql::PromqlLowerer::lower_with_ingestion_interval( + let root = promql::PromqlLowerer::lower_query_with_ingestion_interval( &entry.query.0, &entry.requirements.accuracy.target(), std::time::Duration::from_millis(interval_ms), )?; - Ok(resolve_root(&unresolved)?) + Ok(root) }) .collect() } @@ -123,7 +156,7 @@ mod tests { } use std::time::Duration; - use asap_types::pre_asap::QueryExpr; + use asap_types::ir::{NonASAPOp, TimeRangeKind}; use asap_types::workload::{ BatchEntry, DataWorkload, Evidence, PlanningWorkload, Query, QueryRequirements, QueryWorkload, TimeSelection, @@ -155,27 +188,34 @@ mod tests { } } + // A bare instant selector reads the latest sample within the declared + // ingestion interval: an `Instant` lookback of that length. #[test] fn instant_selector_uses_declared_ingestion_interval() { let query = lower_promql_workload(&workload("sum by (job) (data)"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(1) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(1) + && *kind == TimeRangeKind::Instant + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } + // An explicit `m[5m]` keeps its own window as a `Range` selection. #[test] fn explicit_range_selector_keeps_its_query_range() { let query = lower_promql_workload(&workload("sum_over_time(data[5m])"), 0).unwrap(); - let QueryExpr::Aggregate { child, .. } = &query[0] else { + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { panic!("expected aggregate") }; assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, child } - if *range == Duration::from_secs(300) && matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(300) + && *kind == TimeRangeKind::Range + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index 83251053e..53ec79c86 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -1,14 +1,13 @@ -//! PromQL string → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! PromQL string → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree. //! //! - **Parsing** is delegated to `promql-parser` 0.8. //! - **Lowering** builds *directly in canonical shape* here (issue #179): the //! walk interprets PromQL semantics (range vectors, aggregate operators, -//! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved -//! `ColumnRef`s — the same DAG shape -//! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to -//! canonical, positional `QueryExpr`. The structural decisions a +//! label matchers) and emits `UnresolvedOp` / `UnresolvedScalar` nodes with +//! unresolved `ColumnRef`s — the same tree shape +//! [`resolve_root`](asap_frontend_common::resolve_root) later binds to the +//! positional [`OperatorNode`](asap_types::ir::OperatorNode) DAG. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter //! `topk` recognition, the `PerEntity`/`Reduce` reduction choice, //! `without(...)` grouping) are made right here, since a front end @@ -38,9 +37,11 @@ //! | `increase(m[w])` | `Aggregate{[Increase], TimeRange{w}}` | //! | `changes`/`delta`/`idelta`/`deriv`/`resets`/`predict_linear`/`double_exponential_smoothing`(`m[w]`, …) | `Aggregate{[Changes/Delta/…], TimeRange{w}}` — per-series counter-derivative intents (issue #44) | //! | `absent(v)` / `absent_over_time(m[w])` / `present_over_time(m[w])` | `Aggregate{[Absent/AbsentOverTime/PresentOverTime]}` — presence intents; the empty→synthesized-sample logic is a post-ASAP concern (issue #47) | -//! | `abs`/`ceil`/`sqrt`/`ln`/`clamp*`/`round`/trig(`v`), `pi()` | `Aggregate{[Math(f)]}` element-wise transform (issue #45); `pi()` → a `PromqlScalarBridge` leaf | -//! | `time()` / `timestamp`/`hour`/`day_of_week`/… (`v`) | `EvalTimestamp` leaf / `Aggregate{[TimeFn(f)]}` (issue #46) | -//! | `vector(s)` / `scalar(v)` | `PromqlVectorFromScalar` / `PromqlScalarFromVector` — the scalar⇄vector bridges (issue #48) | +//! | `abs`/`ceil`/`sqrt`/`ln`/`clamp*`/`round`/trig(`v`), `pi()` | typed scalar `Project` (issue #45); `pi()` → a `ScalarExpr::Literal` root | +//! | `time()` / `timestamp`/`hour`/`day_of_week`/… (`v`) | `ScalarExpr::EvalTimestamp` root / `Aggregate{[TimeFn(f)]}` (issue #46) | +//! | `vector(s)` / `scalar(v)` | `PromqlVectorFromScalar(s)` / `ScalarExpr::PromqlScalarFromVector(v)` — the scalar⇄vector bridges (issue #48) | +//! | ` op ` (`time() - 1`, `1 < bool 2`, `-time()`) | `ScalarExpr::{Arithmetic, Case, Negative}` — a scalar expression, never an operator | +//! | `v op `, `a op bool b`, `v > bool 0` | `Project`/`Filter` with owned scalar expressions; vector/vector uses `BinaryOp{return_bool}` | //! | `label_replace(v,…)` / `label_join(v,…)` | `PromqlRelabel{dst, value}` — per-series label rewrite; value unchanged (issue #50) | //! | `info(v, [selector])` | `PromqlInfoEnrich{selector}` — label-enrichment join against the info metric(s); join keys resolved during post-ASAP binding (issue #84) | //! | `group` / `offset` / `@` / `info` | **rejected** — distinct semantics with no intent-algebra representation yet (`info` label-join → #84) | @@ -50,7 +51,7 @@ //! | `limitk(k, v)` / `limit_ratio(r, v)` | `PromqlSeriesSample{LimitK(k) \| LimitRatio(r)}` — series-sampling selection, whole series kept unchanged (issue #86) | //! | `topk(k, count_over_time(…))` / `topk(k, sum_over_time(…))` | `Aggregate{[TopK{k}]}` (heavy-hitter intent) over the explicit inner `Aggregate{[Count/Sum]}` | //! | `topk(k, )` / `bottomk(k, …)` | `Sort{value} → Limit{k}` | -//! | `m{f}` | `Scan{predicates}` | +//! | `m{f}` / `m{f}[w]` | `TimeRange{ingestion, Instant, Scan{predicates}}` / `TimeRange{w, Range, Scan}` | //! | `a OP b` | `BinaryOp{vector_match}` | //! | `expr[r:res]` | `PromqlSubquery{r, res}` | //! | ` offset ` / ` @ `/`start()`/`end()` | `TimeShift{shift}` over the selector's `Scan` — pass-through schema; a ranged selector shifts under its `TimeRange` (issue #40) | @@ -59,22 +60,29 @@ use std::rc::Rc; use std::time::{Duration, SystemTime}; use promql_parser::label::{MatchOp, Matcher}; +use promql_parser::parser::value::ValueType; use promql_parser::parser::{ self, token, AggregateExpr, AtModifier as ParserAtModifier, BinaryExpr, Call, Expr, LabelModifier, Offset, VectorMatchCardinality, VectorSelector, }; -use asap_types::pre_asap::agg_intent::{topk, AggIntent, MathFunc, TimeFunc}; -use asap_types::pre_asap::query_expr::{ - AtModifier, BinaryOpKind, GroupKeys, GroupSide, Predicate, PromQLVectorSetOpKind, Reduction, - SortKey, Source, TimeShift, UnresolvedQueryExpr as Unresolved, VectorGrouping, VectorMatch, - VectorMatchKind, +use asap_frontend_common::{ + UnresolvedOp as Unresolved, UnresolvedPredicate, UnresolvedScalar as Scalar, UnresolvedSortKey, }; -use asap_types::pre_asap::{ - ArithmeticOpKind, ColumnRef, CompareOpKind, InfoMatcher, SampleKind, ScalarValue, +use asap_types::ir::operator::agg_intent::{topk, AggIntent, TimeFunc}; +use asap_types::ir::operator::operator_properties::{ + AtModifier, BinaryOpKind, GroupKeys, GroupSide, PromQLVectorSetOpKind, Reduction, Source, + TimeShift, VectorGrouping, VectorMatch, VectorMatchKind, }; +use asap_types::ir::{BinaryOperator, ExprSemantics, TimeRangeKind}; + +use asap_types::ir::operator::{InfoMatcher, SampleKind}; +use asap_types::ir::scalar::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; use asap_types::types::AccuracyTarget; +/// Every scalar expression this front end builds follows PromQL's numeric rules. +const PROMQL: ExprSemantics = ExprSemantics::Promql; + use crate::error::PromqlError as LoweringError; type Result = std::result::Result; @@ -158,7 +166,7 @@ enum InnerFunc { struct Inner { metric: String, - matchers: Vec, + matchers: Vec, window: Option, func: Option, /// `offset` / `@` on the selector, carried to the `Source` (issue #40). @@ -172,16 +180,35 @@ struct Inner { const MAX_DEPTH: usize = 256; impl PromqlLowerer { - pub(crate) fn lower_with_ingestion_interval( + pub(crate) fn lower_query_with_ingestion_interval( query: &str, accuracy: &AccuracyTarget, interval: Duration, - ) -> Result { + ) -> Result { let _guard = AccuracyGuard::install(accuracy.clone()); let _interval = IngestionIntervalGuard::install(interval); let ast = parser::parse(query).map_err(LoweringError::Parse)?; check_depth(&ast, MAX_DEPTH)?; - walk(&ast) + let mut metrics = Vec::new(); + collect_metric_names(&ast, &mut metrics); + if metrics.iter().any(|metric| { + crate::histogram::current_kind_of(metric) + == Some(crate::histogram::HistogramKind::Native) + }) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + + if ast.value_type() == ValueType::Scalar { + Ok(asap_types::ir::QueryRoot::Scalar( + asap_frontend_common::resolve_scalar_root(&lower_scalar(&ast)?)?, + )) + } else { + Ok(asap_types::ir::QueryRoot::Operator( + asap_frontend_common::resolve_root(&walk(&ast)?)?, + )) + } } } @@ -273,6 +300,13 @@ fn check_depth(expr: &Expr, budget: usize) -> Result<()> { } fn walk(expr: &Expr) -> Result { + // A scalar-typed expression (`5`, `time() - 1`, `scalar(v)`, `1 < bool 2`) + // is a scalar expression at an operator position, never an operator tree. + if expr.value_type() == ValueType::Scalar { + return Err(LoweringError::UnsupportedFeature( + "scalar root requires query-root lowering".into(), + )); + } match expr { Expr::Aggregate(agg) => walk_aggregate(agg), Expr::Call(call) if call.func.name.starts_with("histogram_") => walk_histogram(call), @@ -282,31 +316,22 @@ fn walk(expr: &Expr) -> Result { Expr::Call(call) if is_typeconv_fn(call.func.name) => walk_typeconv(call), Expr::Call(call) if is_label_fn(call.func.name) => walk_label(call), Expr::Call(call) if is_sort_fn(call.func.name) => walk_sort(call), - // A bare `min_of`/`max_of(consts…)` scalar query folds to a `PromqlScalarBridge` - // leaf; a non-constant argument makes `num_expr` fail → rejected (#89). - Expr::Call(call) if is_scalar_reducer_fn(call.func.name) => { - Ok(Unresolved::promql_scalar(num_expr(expr)?)) - } Expr::Call(call) if call.func.name == "info" => walk_info(call), Expr::Call(call) => walk_call(call), Expr::Binary(bin) => walk_binary(bin), Expr::Paren(p) => walk(&p.expr), // `UnaryExpr` is built only by negation (`Neg`); unary `+` is folded to - // identity and `-` to a negated `NumberLiteral`, so this wraps a - // sub-expression whose samples must be sign-flipped. Now that a scalar - // operand exists (#35), express it as `x * -1` — a constant-foldable - // operand (`-(10*1024)`) collapses to a negated `PromqlScalarBridge` leaf; anything - // else is a vector, sign-flipped by a `Mul` against `PromqlScalarBridge(-1)`. `Mul` - // is commutative, so operand order carries no hazard (#36). - Expr::Unary(u) => match num_expr(&u.expr) { - Ok(v) => Ok(Unresolved::promql_scalar(-v)), - Err(_) => Ok(Unresolved::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(walk(&u.expr)?), - rhs: Rc::new(Unresolved::promql_scalar(-1.0)), - vector_match: None, - }), - }, + // identity and `-` to a negated `NumberLiteral`. A scalar + // operand was dispatched to `lower_scalar` above (→ `Negative`), so this + // is a vector projection. Unary negation retains the metric name. + Expr::Unary(u) => Ok(Unresolved::PromqlMap { + child: Rc::new(walk(&u.expr)?), + sample: Scalar::Negative { + expr: Box::new(Scalar::Column(ColumnRef::SampleValue)), + semantics: ExprSemantics::Promql, + }, + drop_metric_name: false, + }), Expr::Subquery(sq) => { let subquery = Unresolved::PromqlSubquery { range: sq.range, @@ -332,13 +357,14 @@ fn walk(expr: &Expr) -> Result { let (metric, matchers, shift) = vs_parts(&ms.vs)?; Ok(Unresolved::TimeRange { range: ms.range, + kind: TimeRangeKind::Range, child: Rc::new(filtered_source(metric, matchers, shift)), }) } - // A number literal is a scalar leaf (`v > 5`, or a bare scalar query - // `5`). String literals only appear as function args (`label_replace`, - // …), which are not supported, so reject them (issue #35). - Expr::NumberLiteral(n) => Ok(Unresolved::promql_scalar(n.val)), + // Scalar-typed, dispatched above; kept for exhaustiveness. String + // literals only appear as function args (`label_replace`, …), so a + // bare one is rejected (issue #35). + Expr::NumberLiteral(_) => unreachable!("scalar handled above"), Expr::StringLiteral(_) => Err(LoweringError::UnsupportedFeature( "bare string literal".into(), )), @@ -348,6 +374,98 @@ fn walk(expr: &Expr) -> Result { } } +/// Lower a scalar-typed PromQL expression to a scalar expression. A constant +/// sub-expression folds to one `Literal` (as `num_expr` always did); anything +/// else keeps its structure: `-time()` → `Negative`, `time() - 1` → +/// `Arithmetic`, `scalar(v)` → `PromqlScalarFromVector`, and a `bool` +/// comparison → `Case(Compare → 1, else 0)` (PromQL yields `0`/`1`). +fn lower_scalar(expr: &Expr) -> Result { + if let Ok(v) = num_expr(expr) { + return Ok(Scalar::Literal(ScalarValue::Float64(v))); + } + match expr { + Expr::Paren(p) => lower_scalar(&p.expr), + Expr::Unary(u) => Ok(Scalar::Negative { + expr: Box::new(lower_scalar(&u.expr)?), + semantics: PROMQL, + }), + Expr::Binary(bin) => lower_scalar_binary(bin), + Expr::Call(call) => match call.func.name { + "time" => Ok(Scalar::EvalTimestamp), + "pi" => Ok(Scalar::Literal(ScalarValue::Float64(std::f64::consts::PI))), + "scalar" => Ok(Scalar::PromqlScalarFromVector(Rc::new(walk(arg( + call, 0, + )?)?))), + // `min_of`/`max_of` fold only over constants (#89); the fold above + // failed, so surface its error for the non-constant argument. + name if is_scalar_reducer_fn(name) => Err(num_expr(expr).unwrap_err()), + other => Err(LoweringError::UnsupportedFunction(other.to_string())), + }, + other => Err(LoweringError::UnsupportedFeature(format!( + "scalar expression `{other}`" + ))), + } +} + +/// ` op `: arithmetic is an `Arithmetic` expression; a +/// comparison needs the `bool` modifier (PromQL has no scalar filter) and +/// becomes `Case(Compare → 1.0, else 0.0)`. The parser already rejects both a +/// bool-less scalar comparison and a scalar set op; both are re-checked here. +fn lower_scalar_binary(bin: &BinaryExpr) -> Result { + let left = Box::new(lower_scalar(&bin.lhs)?); + let right = Box::new(lower_scalar(&bin.rhs)?); + match binop(bin.op.id())? { + BinaryOpKind::Arithmetic(op) => Ok(Scalar::Arithmetic { + op, + left, + right, + semantics: PROMQL, + }), + BinaryOpKind::Compare(op) | BinaryOpKind::CompareBool(op) => { + if !bin.return_bool() { + return Err(LoweringError::InvalidParameter( + "a comparison between two scalars requires the `bool` modifier".into(), + )); + } + let compare = Scalar::Compare { + left, + op, + right, + semantics: PROMQL, + }; + Ok(Scalar::Case { + operand: None, + branches: vec![(compare, Scalar::Literal(ScalarValue::Float64(1.0)))], + else_expr: Some(Box::new(Scalar::Literal(ScalarValue::Float64(0.0)))), + }) + } + BinaryOpKind::Set(_) => Err(LoweringError::UnsupportedFeature( + "set operator between two scalars".into(), + )), + } +} + +/// A binary operation over two vectors. +fn vector_binary( + kind: BinaryOpKind, + vector_match: Option, + return_bool: bool, + lhs: Unresolved, + rhs: Unresolved, +) -> Unresolved { + Unresolved::BinaryOp { + operator: BinaryOperator { + kind, + vector_match, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), + } +} + /// Lower a bare function call (`rate(m[5m])`, `max_over_time(m[5m])`, …). /// /// The common case routes through the flat `lower_inner_call` template. The one @@ -576,7 +694,7 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// build this node) decides `PerEntity` vs `Reduce(by)` *without* knowing /// about `without` yet — it only ever sees `by`-mode keys, since `without`'s /// excluded-labels list is applied here, after the fact, exactly like the -/// pre-#179 legacy `relational::QueryExpr` DAG's own `mark_without` did (its +/// pre-#179 legacy relational tree's own `mark_without` did (its /// converter read `without` only after this front-end step had already set /// it). Whether /// `reduction_for` picked `PerEntity` (only possible when `keys` was empty) @@ -652,24 +770,36 @@ fn build_over_sub_dag(outer: Outer, keys: Vec, child: Unresolved) -> child, )); } - let sorted = Unresolved::Sort { - keys: vec![SortKey { - expr: Unresolved::Column(ColumnRef::SampleValue), - ascending: !descending, - nulls_first: false, - }], - partition_by: keys.into(), - child: Rc::new(child), - }; - Unresolved::Limit { - n: k as usize, - offset: 0, - child: Rc::new(sorted), - } + ranked_by_value(keys, k, descending, child) } }) } +/// Generic `topk`/`bottomk`: `Limit{k} → Sort{value, partition_by: keys}` over +/// `child` — an order-by-value ranking, not a heavy-hitter intent. +fn ranked_by_value( + keys: Vec, + k: u64, + descending: bool, + child: Unresolved, +) -> Unresolved { + let sorted = Unresolved::Sort { + keys: vec![UnresolvedSortKey { + expr: Scalar::Column(ColumnRef::SampleValue), + ascending: !descending, + nulls_first: false, + }], + partition_by: keys.into(), + child: Rc::new(child), + }; + Unresolved::Limit { + n: Some(k as usize), + offset: 0, + partition_by: GroupKeys::none(), + child: Rc::new(sorted), + } +} + /// The `histogram_*` function family (issues #43, histogram_quantile). /// /// `histogram_quantile(φ, )` lowers `` in full — preserving any @@ -696,7 +826,7 @@ fn walk_histogram(call: &Call) -> Result { // The true signal is the argument's sample type: a declared // `HistogramKind` (issue #79) drives the choice when available, else we // fall back to the structural `by (le)`/`_bucket` heuristic (issue #43). - if !histogram_arg_is_sketchable(arg_expr) { + if !histogram_arg_is_sketchable(arg_expr)? { return Ok(classic_histogram_quantile(phi, "", walk(arg_expr)?)); } let func = AggIntent::Quantile { @@ -706,23 +836,9 @@ fn walk_histogram(call: &Call) -> Result { }; return Ok(outer_aggregate(vec![], func, walk(arg_expr)?)); } - // (histogram_quantile handled above; accessors below) - let (func, vec_idx) = match call.func.name { - "histogram_count" => (AggIntent::HistogramCount, 0), - "histogram_sum" => (AggIntent::HistogramSum, 0), - "histogram_avg" => (AggIntent::HistogramAvg, 0), - "histogram_stddev" => (AggIntent::HistogramStdDev, 0), - "histogram_stdvar" => (AggIntent::HistogramStdVar, 0), - "histogram_fraction" => ( - AggIntent::HistogramFraction { - lower: num_arg(call, 0)?, - upper: num_arg(call, 1)?, - }, - 2, - ), - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - Ok(outer_aggregate(vec![], func, walk(arg(call, vec_idx)?)?)) + Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )) } /// Classic-bucket `histogram_quantile(φ, child)`. One histogram is the set of @@ -769,7 +885,7 @@ fn walk_histogram_quantiles(call: &Call) -> Result { )); } // The bucket-vs-native choice is a property of the argument, not of φ. - let sketchable = histogram_arg_is_sketchable(vec_expr); + let sketchable = histogram_arg_is_sketchable(vec_expr)?; let branches = (2..call.args.args.len()) .map(|i| { let phi = bounded_quantile_param(num_arg(call, i)?)?; @@ -796,9 +912,7 @@ fn walk_histogram_quantiles(call: &Call) -> Result { }; Ok(Unresolved::PromqlRelabel { dst: label.clone(), - value: Rc::new(Unresolved::Literal(ScalarValue::Utf8(open_metrics_float( - phi, - )))), + value: Scalar::Literal(ScalarValue::Utf8(open_metrics_float(phi))), child: Rc::new(quantile), }) }) @@ -852,12 +966,12 @@ fn open_metrics_float(v: f64) -> String { } } -/// The time / calendar functions (issue #46). +/// The calendar functions (issue #46); `time()` is scalar-typed and lowers in +/// `lower_scalar`. fn is_time_fn(name: &str) -> bool { matches!( name, - "time" - | "timestamp" + "timestamp" | "minute" | "hour" | "day_of_week" @@ -869,33 +983,31 @@ fn is_time_fn(name: &str) -> bool { ) } -/// `time()` → the `EvalTimestamp` leaf. `timestamp(v)` and the calendar accessors → -/// `Aggregate{[TimeFn(f)]}` over the argument vector, or over `EvalTimestamp` for the +/// `timestamp(v)` and the calendar accessors → `Aggregate{[TimeFn(f)]}` over +/// the argument vector, or over `PromqlVectorFromScalar(EvalTimestamp)` for the /// no-argument calendar forms (`hour()`, `day_of_week()`, …). Issue #46. fn walk_time(call: &Call) -> Result { - if call.func.name == "time" { - return Ok(Unresolved::EvalTimestamp); + // timestamp() reads the selected sample's timestamp, not its value. + if call.func.name == "timestamp" { + return Ok(outer_aggregate( + vec![], + AggIntent::TimeFn(TimeFunc::Timestamp), + walk(arg(call, 0)?)?, + )); } - let func = match call.func.name { - "timestamp" => TimeFunc::Timestamp, - "minute" => TimeFunc::Minute, - "hour" => TimeFunc::Hour, - "day_of_week" => TimeFunc::DayOfWeek, - "day_of_month" => TimeFunc::DayOfMonth, - "day_of_year" => TimeFunc::DayOfYear, - "month" => TimeFunc::Month, - "year" => TimeFunc::Year, - "days_in_month" => TimeFunc::DaysInMonth, - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - // A calendar function with no argument reads the evaluation time; otherwise - // it maps over each sample's timestamp in the argument vector. - let inner = if call.args.args.is_empty() { - Unresolved::EvalTimestamp + let child = if call.args.args.is_empty() { + Unresolved::PromqlVectorFromScalar(Scalar::EvalTimestamp) } else { walk(arg(call, 0)?)? }; - Ok(outer_aggregate(vec![], AggIntent::TimeFn(func), inner)) + Ok(Unresolved::PromqlMap { + child: Rc::new(child), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args: vec![Scalar::Column(ColumnRef::SampleValue)], + }, + drop_metric_name: true, + }) } /// The presence functions (issue #47). @@ -919,24 +1031,19 @@ fn walk_presence(call: &Call) -> Result { Ok(outer_aggregate(vec![], func, walk(arg(call, 0)?)?)) } -/// The scalar⇄vector type-conversion functions (issue #48). `info` is *not* -/// here: it is a label-enrichment join against info metrics, not a type -/// conversion, so it falls through to the `UnsupportedFunction` path (#84). +/// The scalar→vector conversion (issue #48); `scalar(v)` is scalar-typed and +/// lowers in `lower_scalar`. `info` is *not* here: it is a label-enrichment +/// join, not a type conversion (#84). fn is_typeconv_fn(name: &str) -> bool { - matches!(name, "vector" | "scalar") + name == "vector" } -/// `vector(s)` — promote a scalar to a label-less instant vector. `scalar(v)` -/// — collapse a single-element vector to its value. Both are honest bridge -/// nodes in the IR; the "exactly one element → NaN otherwise" runtime rule of -/// `scalar` is a post-ASAP/runtime concern (issue #48). +/// `vector(s)` — promote a scalar to a label-less instant vector carrying the +/// scalar expression `s` (issue #48). fn walk_typeconv(call: &Call) -> Result { - let inner = walk(arg(call, 0)?)?; - Ok(match call.func.name { - "vector" => Unresolved::PromqlVectorFromScalar(Rc::new(inner)), - "scalar" => Unresolved::PromqlScalarFromVector(Rc::new(inner)), - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }) + Ok(Unresolved::PromqlVectorFromScalar(lower_scalar(arg( + call, 0, + )?)?)) } /// The instant-vector reordering functions (issue #51). @@ -960,13 +1067,13 @@ fn walk_sort(call: &Call) -> Result { "sort_by_label_desc" => (false, false), other => return Err(LoweringError::UnsupportedFunction(other.to_string())), }; - let sort_key = |expr| SortKey { + let sort_key = |expr| UnresolvedSortKey { expr, ascending, nulls_first: false, }; let keys = if by_value { - vec![sort_key(Unresolved::Column(ColumnRef::SampleValue))] + vec![sort_key(Scalar::Column(ColumnRef::SampleValue))] } else { // `sort_by_label(v, "l1", "l2", …)` — one key per label arg, in order. if call.args.args.len() < 2 { @@ -976,7 +1083,7 @@ fn walk_sort(call: &Call) -> Result { } (1..call.args.args.len()) .map(|i| { - Ok(sort_key(Unresolved::Column(ColumnRef::Named(str_arg( + Ok(sort_key(Scalar::Column(ColumnRef::Named(str_arg( call, i, )?)))) }) @@ -1050,19 +1157,15 @@ fn walk_label(call: &Call) -> Result { let replacement = str_arg(call, 2)?; let src = str_arg(call, 3)?; let regex = str_arg(call, 4)?; - let value = Unresolved::FunctionCall { + let value = Scalar::FunctionCall { name: "label_replace".into(), args: vec![ - Unresolved::Column(ColumnRef::Named(src)), - Unresolved::Literal(ScalarValue::Utf8(regex)), - Unresolved::Literal(ScalarValue::Utf8(replacement)), + Scalar::Column(ColumnRef::Named(src)), + Scalar::Literal(ScalarValue::Utf8(regex)), + Scalar::Literal(ScalarValue::Utf8(replacement)), ], }; - Ok(Unresolved::PromqlRelabel { - dst, - value: Rc::new(value), - child, - }) + Ok(Unresolved::PromqlRelabel { dst, value, child }) } "label_join" => { // label_join(v, dst, sep, src_1, …, src_n) — needs ≥1 source label. @@ -1073,19 +1176,15 @@ fn walk_label(call: &Call) -> Result { } let dst = str_arg(call, 1)?; let sep = str_arg(call, 2)?; - let mut args = vec![Unresolved::Literal(ScalarValue::Utf8(sep))]; + let mut args = vec![Scalar::Literal(ScalarValue::Utf8(sep))]; for i in 3..call.args.args.len() { - args.push(Unresolved::Column(ColumnRef::Named(str_arg(call, i)?))); + args.push(Scalar::Column(ColumnRef::Named(str_arg(call, i)?))); } - let value = Unresolved::FunctionCall { + let value = Scalar::FunctionCall { name: "label_join".into(), args, }; - Ok(Unresolved::PromqlRelabel { - dst, - value: Rc::new(value), - child, - }) + Ok(Unresolved::PromqlRelabel { dst, value, child }) } other => Err(LoweringError::UnsupportedFunction(other.to_string())), } @@ -1118,7 +1217,6 @@ fn is_math_fn(name: &str) -> bool { | "atanh" | "deg" | "rad" - | "pi" | "round" | "clamp" | "clamp_min" @@ -1127,59 +1225,24 @@ fn is_math_fn(name: &str) -> bool { } /// A math / trig function — a per-series element-wise value transform, lowered -/// to a per-series `Aggregate{[Math(f)]}` over the (instant) argument vector. -/// `pi()` is the constant π, lowered to a `PromqlScalarBridge` leaf (issue #45). +/// to a typed scalar projection over the instant-vector argument. +/// `pi()` is scalar-typed and lowers in `lower_scalar` (issue #45). fn walk_math(call: &Call) -> Result { - if call.func.name == "pi" { - return Ok(Unresolved::promql_scalar(std::f64::consts::PI)); + let mut args = vec![Scalar::Column(ColumnRef::SampleValue)]; + for index in 1..call.args.args.len() { + args.push(lower_scalar(arg(call, index)?)?); } - let func = match call.func.name { - "abs" => MathFunc::Abs, - "ceil" => MathFunc::Ceil, - "floor" => MathFunc::Floor, - "exp" => MathFunc::Exp, - "ln" => MathFunc::Ln, - "log2" => MathFunc::Log2, - "log10" => MathFunc::Log10, - "sqrt" => MathFunc::Sqrt, - "sgn" => MathFunc::Sgn, - "sin" => MathFunc::Sin, - "cos" => MathFunc::Cos, - "tan" => MathFunc::Tan, - "asin" => MathFunc::Asin, - "acos" => MathFunc::Acos, - "atan" => MathFunc::Atan, - "sinh" => MathFunc::Sinh, - "cosh" => MathFunc::Cosh, - "tanh" => MathFunc::Tanh, - "asinh" => MathFunc::Asinh, - "acosh" => MathFunc::Acosh, - "atanh" => MathFunc::Atanh, - "deg" => MathFunc::Deg, - "rad" => MathFunc::Rad, - // `round(v)` defaults the step to 1; `round(v, to)` reads arg 1. - "round" => MathFunc::Round { - to_nearest: if call.args.args.len() >= 2 { - num_arg(call, 1)? - } else { - 1.0 - }, - }, - "clamp" => MathFunc::Clamp { - min: num_arg(call, 1)?, - max: num_arg(call, 2)?, - }, - "clamp_min" => MathFunc::ClampMin { - min: num_arg(call, 1)?, - }, - "clamp_max" => MathFunc::ClampMax { - max: num_arg(call, 1)?, + if call.func.name == "round" && args.len() == 1 { + args.push(Scalar::Literal(ScalarValue::Float64(1.0))); + } + Ok(Unresolved::PromqlMap { + child: Rc::new(walk(arg(call, 0)?)?), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args, }, - other => return Err(LoweringError::UnsupportedFunction(other.to_string())), - }; - // The value being transformed is always arg 0 (a vector). - let inner = walk(arg(call, 0)?)?; - Ok(outer_aggregate(vec![], AggIntent::Math(func), inner)) + drop_metric_name: true, + }) } /// Whether `expr` is a **classic cumulative-bucket** `histogram_quantile` @@ -1202,15 +1265,31 @@ fn walk_math(call: &Call) -> Result { /// declared `RawSamples`) and the false-negative (a suffix-less classic /// histogram declared `ClassicBucket`) of the structural heuristic. With no /// declaration, fall back to the structural `by (le)`/`_bucket` heuristic. -fn histogram_arg_is_sketchable(arg: &Expr) -> bool { +fn histogram_arg_is_sketchable(arg: &Expr) -> Result { let mut metrics = Vec::new(); collect_metric_names(arg, &mut metrics); - for metric in &metrics { - if let Some(kind) = crate::histogram::current_kind_of(metric) { - return kind.is_sketchable(); + let kinds = metrics + .iter() + .filter_map(|metric| crate::histogram::current_kind_of(metric)) + .collect::>(); + if kinds.contains(&crate::histogram::HistogramKind::Native) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + if let Some(kind) = kinds.first() { + if kinds.iter().any(|other| other != kind) { + return Err(LoweringError::UnsupportedFeature( + "mixed histogram sample contracts".into(), + )); } + return Ok(kind.is_sketchable()); + } + if is_classic_bucket_arg(arg) { + Ok(false) + } else { + Err(LoweringError::UnsupportedFeature("histogram_quantile requires classic buckets; use quantile for float samples or explicitly declare the RawSamples extension".into())) } - !is_classic_bucket_arg(arg) } /// Collect the metric names of every vector/matrix selector reachable in `expr` @@ -1282,9 +1361,28 @@ fn selector_is_bucket(vs: &VectorSelector) -> bool { || vs.matchers.matchers.iter().any(|m| m.name == "le") } +/// A binary op with at least one vector operand (a scalar/scalar op is +/// scalar-typed and never reaches here). A scalar side lowers to a +/// scalar expression; mixed operations resolve to Project or Filter. fn walk_binary(bin: &BinaryExpr) -> Result { - let lhs = scalar_or_vector(&bin.lhs)?; - let rhs = scalar_or_vector(&bin.rhs)?; + let op = binop(bin.op.id())?; + let scalar_left = bin.lhs.value_type() == ValueType::Scalar; + if scalar_left || bin.rhs.value_type() == ValueType::Scalar { + let (scalar, vector) = if scalar_left { + (&bin.lhs, &bin.rhs) + } else { + (&bin.rhs, &bin.lhs) + }; + return Ok(Unresolved::PromqlScalarOp { + child: Rc::new(walk(vector)?), + scalar: lower_scalar(scalar)?, + op, + scalar_left, + return_bool: bin.return_bool(), + }); + } + let lhs = walk(&bin.lhs)?; + let rhs = walk(&bin.rhs)?; // `VectorMatch` has no fill field; dropping fill would change which series // are emitted and their values, so the query must fall back to exact // execution instead. @@ -1295,10 +1393,6 @@ fn walk_binary(bin: &BinaryExpr) -> Result { ))); } } - let op = match (binop(bin.op.id())?, bin.return_bool()) { - (BinaryOpKind::Compare(op), true) => BinaryOpKind::CompareBool(op), - (op, _) => op, - }; let vector_match = bin.modifier.as_ref().map(|m| { let (kind, labels) = match &m.matching { Some(LabelModifier::Include(ls)) => (VectorMatchKind::On, ls.labels.clone()), @@ -1329,12 +1423,7 @@ fn walk_binary(bin: &BinaryExpr) -> Result { grouping, } }); - Ok(Unresolved::BinaryOp { - op, - lhs: Rc::new(lhs), - rhs: Rc::new(rhs), - vector_match, - }) + Ok(vector_binary(op, vector_match, bin.return_bool(), lhs, rhs)) } fn lower_inner(expr: &Expr) -> Result { @@ -1601,20 +1690,7 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result Some(intent) => windowed_aggregate(inner, vec![], intent), None => instant_source(inner.metric, inner.matchers, inner.shift), }; - let sorted = Unresolved::Sort { - keys: vec![SortKey { - expr: Unresolved::Column(ColumnRef::SampleValue), - ascending: !descending, - nulls_first: false, - }], - partition_by: keys.into(), - child: Rc::new(base), - }; - Ok(Unresolved::Limit { - n: k as usize, - offset: 0, - child: Rc::new(sorted), - }) + Ok(ranked_by_value(keys, k, descending, base)) } } } @@ -1650,19 +1726,12 @@ fn windowed_aggregate( let child = match inner.window { Some(w) => Unresolved::TimeRange { range: w, + kind: TimeRangeKind::Range, child: Rc::new(base), }, - None => base, + None => ingestion_lookback(base), }; let reduction = reduction_for(&keys, inner.window.is_some() || intent.is_per_series()); - let child = if inner.window.is_none() { - Unresolved::TimeRange { - range: current_ingestion_interval(), - child: Rc::new(child), - } - } else { - child - }; Unresolved::Aggregate { reduction, measures: vec![intent], @@ -1715,13 +1784,10 @@ fn per_series_aggregate( } } -fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { +fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { let scan = Unresolved::Scan { source: Source::TimeSeries { metric }, - predicates: matchers - .into_iter() - .map(|m| Predicate(Rc::new(m))) - .collect(), + predicates: matchers.into_iter().map(UnresolvedPredicate).collect(), // Usage-derived (PromQL is schemaless) — the SchemaResolver fills this in. schema: None, }; @@ -1735,10 +1801,17 @@ fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) } } -fn instant_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { +/// An instant selector: the latest sample per series within the workload's +/// ingestion interval, so the lookback is an `Instant` `TimeRange`. +fn instant_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { + ingestion_lookback(filtered_source(metric, matchers, shift)) +} + +fn ingestion_lookback(child: Unresolved) -> Unresolved { Unresolved::TimeRange { range: current_ingestion_interval(), - child: Rc::new(filtered_source(metric, matchers, shift)), + kind: TimeRangeKind::Instant, + child: Rc::new(child), } } @@ -1881,7 +1954,7 @@ fn resolve_group(agg: &AggregateExpr) -> Result<(Vec, bool)> { // ── Free helpers ────────────────────────────────────────────────────────────── -fn vs_parts(vs: &VectorSelector) -> Result<(String, Vec, TimeShift)> { +fn vs_parts(vs: &VectorSelector) -> Result<(String, Vec, TimeShift)> { // A non-equality `__name__` matcher (`=~` / `!~` / `!=`) selects *across* // metric names. `Source::TimeSeries { metric }` carries a single concrete // metric name, so there is no representation for a regex/negated name @@ -1961,21 +2034,22 @@ fn system_time_ms(t: SystemTime) -> Result { }) } -fn matcher_to_compare(m: &Matcher) -> Unresolved { +fn matcher_to_compare(m: &Matcher) -> Scalar { let op = match &m.op { MatchOp::Equal => CompareOpKind::Eq, MatchOp::NotEqual => CompareOpKind::Ne, MatchOp::Re(_) => CompareOpKind::Regex, MatchOp::NotRe(_) => CompareOpKind::NotRegex, }; - Unresolved::Compare { - left: Rc::new(Unresolved::Column(ColumnRef::Named(m.name.clone()))), + Scalar::Compare { + left: Box::new(Scalar::Column(ColumnRef::Named(m.name.clone()))), op, - right: Rc::new(Unresolved::Literal(ScalarValue::Utf8(m.value.clone()))), + right: Box::new(Scalar::Literal(ScalarValue::Utf8(m.value.clone()))), + semantics: PROMQL, } } -fn extract_matrix(expr: &Expr) -> Result<(String, Vec, Duration, TimeShift)> { +fn extract_matrix(expr: &Expr) -> Result<(String, Vec, Duration, TimeShift)> { match expr { Expr::MatrixSelector(ms) => { let (metric, matchers, shift) = vs_parts(&ms.vs)?; @@ -2017,6 +2091,7 @@ fn num_expr(expr: &Expr) -> Result { match expr { Expr::NumberLiteral(n) => Ok(n.val), Expr::Paren(p) => num_expr(&p.expr), + Expr::Unary(u) => Ok(-num_expr(&u.expr)?), // Constant-fold a pure scalar arithmetic expression — the parser does // not fold `10*1024*1024` / `24 * 3600`. A `modifier` (vector matching) // or a non-arithmetic operator means it is not a pure scalar. @@ -2074,15 +2149,6 @@ fn is_scalar_reducer_fn(name: &str) -> bool { matches!(name, "min_of" | "max_of") } -/// A `BinaryOp` operand: fold a pure-scalar expression (`5`, `10*1024*1024`) to -/// a `PromqlScalarBridge` leaf, otherwise walk it as a vector (issue #35). -fn scalar_or_vector(expr: &Expr) -> Result { - match num_expr(expr) { - Ok(v) => Ok(Unresolved::promql_scalar(v)), - Err(_) => walk(expr), - } -} - /// `topk`/`bottomk` count parameter — a non-negative integer. Rejects /// fractional / negative / non-finite values rather than silently truncating /// or saturating them via `as u64` (`topk(2.7, …)` ≠ `topk(2, …)`). diff --git a/crates/frontend-promql/tests/count_planning.rs b/crates/frontend-promql/tests/count_planning.rs index 3d3b65651..01e712dc7 100644 --- a/crates/frontend-promql/tests/count_planning.rs +++ b/crates/frontend-promql/tests/count_planning.rs @@ -1,19 +1,20 @@ //! Query text through summary selection: counts use observations, never value weights. -use std::rc::Rc; - -use asap_aware_mapping::accuracy::DefaultAccuracyModel; -use asap_aware_mapping::cost_model::DefaultCostModel; -use asap_aware_mapping::{ - default_strategies, search_workload_with_targets, Replacement, ReplacementStrategy, - SketchAlgorithmStrategy, TargetSubDAG, +use asap_logical_optimizer::accuracy::DefaultAccuracyModel; +use asap_logical_optimizer::{ + default_strategies, search_workload_with_targets, ASAPStrategies, Replacement, + ReplacementStrategy, TargetSubDAG, }; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::cost::cost_model::DefaultCostModel; mod support; -use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, FieldDataType, NonNegativeWeightProof, - PostAsapOperatorPayload, SketchAlgorithm, SummaryExpr, SummaryInputExpr, WeightDomain, +use asap_types::ir::export::PhysicalASAPOperatorPayload; +use asap_types::ir::schema::{ + ExactKind, FieldDataType, NonNegativeWeightProof, SketchAlgorithm, SummaryInputExpr, + WeightDomain, }; +use asap_types::ir::{ASAPOp, Operator}; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, post_asap_dag}; #[test] fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { @@ -21,7 +22,7 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { epsilon: 0.01, delta: 0.01, }; - let root = Rc::new(lower_promql("count by(job)(up)", target.clone()).unwrap()); + let root = lower_promql("count by(job)(up)", target.clone()).unwrap(); let space = search_workload_with_targets( vec![("count", root, Some(target))], &default_strategies(), @@ -39,8 +40,7 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { assert!(hydra .iter() .all(|candidate| candidate.has_missing_accuracy_evidence())); - assert!(!space - .global_selection(&DefaultCostModel) + assert!(!global_selection(&space, &DefaultCostModel) .for_target(planned) .unwrap() .chosen @@ -51,14 +51,13 @@ fn grouped_count_keeps_uncertified_hydra_candidates_for_backend_review() { #[test] fn exact_counts_select_count_accumulators() { for query in ["count(up)", "count by(job)(up)", "count_over_time(up[5m])"] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact).unwrap()); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql(query, AccuracyTarget::Exact).unwrap(); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); assert!( candidates.iter().any(|candidate| { - matches!(&candidate.replacement, Replacement::Summary(node) - if matches!(&node.expr, SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Count, _), .. })) + matches!(&candidate.replacement, Replacement::SubDAG(node) + if matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Count, _), .. }))) }), "{query}: {candidates:?}" ); @@ -69,22 +68,22 @@ fn exact_counts_select_count_accumulators() { #[test] fn frequency_count_candidates_use_unit_weights() { for query in ["count_over_time(up[5m])", "count(up)"] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Epsilon(0.02)).unwrap()); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql(query, AccuracyTarget::Epsilon(0.02)).unwrap(); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); let mut algorithms = Vec::new(); for candidate in &candidates { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { continue; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator + else { continue; }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), input, .. - } = &summary_input.expr + }) = &summary_input.operator else { continue; }; @@ -98,11 +97,11 @@ fn frequency_count_candidates_use_unit_weights() { ) { continue; } - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); assert!( dag.nodes.iter().any(|node| matches!( &node.payload, - PostAsapOperatorPayload::SummaryAgg { input: actual, .. } if actual == input + PhysicalASAPOperatorPayload::SummaryAgg { input: actual, .. } if actual == input )), "post-ASAP DAG must preserve the count update contract" ); @@ -135,25 +134,26 @@ fn frequency_count_candidates_use_unit_weights() { // This narrow test oracle interprets the emitted aggregate, not Prometheus ingestion, // staleness, or scrape scheduling. Unsupported plan shapes fail explicitly. fn aggregate_fixture(query: &str, series: &[Vec]) -> Vec { - use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction}; + use asap_types::ir::operator::{AggIntent, Reduction}; + use asap_types::ir::NonASAPOp; let root = lower_promql(query, AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &root + } = root.expect_non_asap() else { panic!("expected aggregate: {root:?}"); }; - match child.as_ref() { - QueryExpr::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), - QueryExpr::TimeRange { range, child } => { + match child.expect_non_asap() { + NonASAPOp::Scan { .. } => assert!(series.iter().all(|samples| samples.len() == 1)), + NonASAPOp::TimeRange { range, child, .. } => { assert!(matches!(range.as_secs(), 1 | 300)); if range.as_secs() == 1 { assert!(series.iter().all(|samples| samples.len() == 1)); } - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } other => panic!("unsupported fixture input: {other:?}"), } @@ -224,23 +224,21 @@ fn count_over_time_counts_scrapes_not_sample_values() { // This checks the planner's numerical update contract, not a sketch-library runtime. #[test] fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { - use asap_types::pre_asap::ColumnRef; - let root = - Rc::new(lower_promql("count_over_time(up[5m])", AccuracyTarget::Epsilon(0.02)).unwrap()); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + use asap_types::ir::scalar::ColumnRef; + let root = lower_promql("count_over_time(up[5m])", AccuracyTarget::Epsilon(0.02)).unwrap(); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); let dag = candidates .iter() .find_map(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return None; }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); dag.nodes .iter() .any(|node| { matches!(&node.payload, - PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } if kind.algorithm() == &SketchAlgorithm::Cms) }) .then_some(dag) @@ -250,7 +248,7 @@ fn cms_count_updates_total_ten_for_zero_positive_and_negative_samples() { .nodes .iter() .find_map(|node| match &node.payload { - PostAsapOperatorPayload::SummaryAgg { input, .. } => Some(input), + PhysicalASAPOperatorPayload::SummaryAgg { input, .. } => Some(input), _ => None, }) .unwrap(); diff --git a/crates/frontend-promql/tests/histogram_metadata.rs b/crates/frontend-promql/tests/histogram_metadata.rs index 55f35ddeb..37f0d5e77 100644 --- a/crates/frontend-promql/tests/histogram_metadata.rs +++ b/crates/frontend-promql/tests/histogram_metadata.rs @@ -7,16 +7,17 @@ use asap_frontend_promql::{HistogramCatalog, HistogramKind}; mod support; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::types::AccuracyTarget; use support::{lower_promql, lower_promql_with_histograms}; /// The histogram/quantile intent kind in the lowered DAG: `"HQ"` for the /// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. -fn quantile_kind(qe: &QueryExpr) -> &'static str { - fn walk(e: &QueryExpr) -> Option<&'static str> { - match e { - QueryExpr::Aggregate { +fn quantile_kind(qe: &OperatorNode) -> &'static str { + fn walk(e: &OperatorNode) -> Option<&'static str> { + match e.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => measures .iter() @@ -26,12 +27,12 @@ fn quantile_kind(qe: &QueryExpr) -> &'static str { _ => None, }) .or_else(|| walk(child)), - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Project { child, .. } => walk(child), + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } + | NonASAPOp::Project { child, .. } => walk(child), _ => None, } } @@ -48,27 +49,26 @@ fn with_meta(q: &str, catalog: HistogramCatalog) -> &'static str { #[test] fn heuristic_baseline_is_unchanged_without_a_catalog() { - // Classic `by (le)`-bucket form → HistogramQuantile; anything else → Quantile. + // Classic buckets are represented; undeclared native samples are rejected. assert_eq!( heuristic( "histogram_quantile(0.9, sum by (le) (rate(http_request_duration_seconds_bucket[5m])))" ), "HQ" ); - assert_eq!(heuristic("histogram_quantile(0.9, native_latency)"), "Q"); + assert!(lower_promql( + "histogram_quantile(0.9, native_latency)", + AccuracyTarget::Exact + ) + .is_err()); } #[test] fn declared_classic_bucket_fixes_the_false_negative() { // A classic histogram exposed WITHOUT the `_bucket` suffix and queried with - // no `le` grouping/matcher: the heuristic wrongly routes it to the - // sketch-able Quantile. Declaring it `ClassicBucket` corrects it. + // no `le` grouping/matcher requires an explicit sample-type declaration. let q = "histogram_quantile(0.9, latency_seconds)"; - assert_eq!( - heuristic(q), - "Q", - "heuristic mis-routes the suffix-less classic histogram" - ); + assert!(lower_promql(q, AccuracyTarget::Exact).is_err()); assert_eq!( with_meta( q, @@ -80,7 +80,7 @@ fn declared_classic_bucket_fixes_the_false_negative() { } #[test] -fn declared_raw_or_native_fixes_the_false_positive() { +fn declared_raw_extension_and_native_gap_override_the_heuristic() { // A metric merely NAMED `…_bucket` that actually holds raw samples / a native // histogram: the heuristic wrongly routes it to bucket interpolation. let q = "histogram_quantile(0.9, foo_bucket)"; @@ -97,14 +97,9 @@ fn declared_raw_or_native_fixes_the_false_positive() { "Q", "raw samples are sketch-able" ); - assert_eq!( - with_meta( - q, - HistogramCatalog::new().with("foo_bucket", HistogramKind::Native) - ), - "Q", - "native histograms are sketch-able" - ); + let catalog = HistogramCatalog::new().with("foo_bucket", HistogramKind::Native); + assert!(lower_promql_with_histograms(q, AccuracyTarget::Exact, catalog.clone()).is_err()); + assert!(lower_promql_with_histograms("foo_bucket", AccuracyTarget::Exact, catalog).is_err()); } #[test] @@ -119,10 +114,12 @@ fn undeclared_metric_falls_back_to_the_heuristic() { ), "HQ" ); - assert_eq!( - with_meta("histogram_quantile(0.9, native_thing)", catalog), - "Q" - ); + assert!(lower_promql_with_histograms( + "histogram_quantile(0.9, native_thing)", + AccuracyTarget::Exact, + catalog + ) + .is_err()); } #[test] diff --git a/crates/frontend-promql/tests/maintained_population_horizon.rs b/crates/frontend-promql/tests/maintained_population_horizon.rs index 88b1c88fe..6611ec781 100644 --- a/crates/frontend-promql/tests/maintained_population_horizon.rs +++ b/crates/frontend-promql/tests/maintained_population_horizon.rs @@ -1,37 +1,33 @@ mod support; -use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; -use asap_types::post_asap::maintained_population::PopulationInput; -use asap_types::post_asap::{SummaryExpr, ValueOperation}; +use asap_logical_optimizer::pass1::maintained_population::MaintainedPopulationStrategy; +use asap_types::ir::operator::maintained_population::PopulationInput; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator}; use asap_types::types::AccuracyTarget; -use std::rc::Rc; // A population for a one-second selector must expire members after one second. #[test] fn population_preserves_selector_horizon() { - let root = Rc::new(support::lower_promql("sum(a)", AccuracyTarget::Exact).unwrap()); + let root = support::lower_promql("sum(a)", AccuracyTarget::Exact).unwrap(); let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &candidate.expr else { + // The evaluation sits over the maintained population. + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &candidate.operator else { panic!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!() }; let PopulationInput::CurrentSeries(spec) = &population.input else { panic!() }; assert_eq!(spec.lookback_ms, 1_000); - asap_types::post_asap::compile_post_asap_dag(&candidate).unwrap(); - let asap_types::pre_asap::QueryExpr::Aggregate { child: source, .. } = root.as_ref() else { + support::post_asap_dag(&candidate); + let NonASAPOp::Aggregate { child: source, .. } = root.expect_non_asap() else { panic!() }; - assert!(spec.matches_input(source)); + assert!(spec.matches_node(source)); let mut wrong = spec.clone(); wrong.lookback_ms = 300_000; - assert!(!wrong.matches_input(source)); + assert!(!wrong.matches_node(source)); } diff --git a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs index 1b37cdee4..695be71f3 100644 --- a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs +++ b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs @@ -29,7 +29,11 @@ use asap_frontend_promql::PromqlError as LoweringError; #[path = "../support.rs"] mod support; -use asap_types::pre_asap::{AggIntent, BinaryOpKind, CompareOpKind, QueryExpr, Reduction}; +use std::rc::Rc; + +use asap_types::ir::operator::{AggIntent, BinaryOpKind, Reduction}; +use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ScalarExpr}; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -44,78 +48,29 @@ fn queries() -> impl Iterator { } /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } -/// Every `AggIntent` in the DAG. -fn intents(e: &QueryExpr) -> Vec { +/// Every `AggIntent` in the tree. `AggIntent` only ever lives in +/// `Aggregate.measures`, never in a scalar position (issue #205); +/// `children()` also descends into the operators a scalar position reads. +fn intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } - // `AggIntent` only ever lives in `Aggregate.measures`, never in a - // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} + fn go(e: &OperatorNode, out: &mut Vec) { + if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { + out.extend(measures.iter().cloned()); + } + for child in e.children() { + go(child, out); } } go(e, &mut out); out } -fn has bool>(e: &QueryExpr, p: F) -> bool { +fn has bool>(e: &OperatorNode, p: F) -> bool { intents(e).iter().any(p) } @@ -180,15 +135,21 @@ fn vector_vs_vector_comparison_lowers_to_binaryop() { // Both operands are instant vectors → a `BinaryOp{Compare}` of two // ingestion-interval-bounded scans. let qe = ok("node_hwmon_temp_celsius > node_hwmon_temp_max_celsius"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); assert!( - matches!(rhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } @@ -197,8 +158,8 @@ fn kube_replica_mismatch_comparison_lowers() { // Kubernetes: `kube_replicaset_spec_replicas != kube_replicaset_status_ready_replicas`. let qe = ok("kube_replicaset_spec_replicas != kube_replicaset_status_ready_replicas"); assert!(matches!( - &qe, - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) + qe.expect_non_asap(), + NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Ne) )); } @@ -225,7 +186,11 @@ fn error_ratio_core_lowers() { // threshold: `sum(rate(failed[5m])) / sum(rate(total[5m]))` → a `BinaryOp(Div)` // of two cross-series sums over per-series rates. let qe = ok("sum(rate(litellm_proxy_failed_requests_metric_total[5m])) / sum(rate(litellm_proxy_total_requests_metric_total[5m]))"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert!(matches!(op, BinaryOpKind::Arithmetic(_))); @@ -250,11 +215,11 @@ fn all_targets_missing_core_lowers() { // Prometheus self-monitoring `sum by (job) (up)` (the corpus query is // `… == 0`). Cross-series sum grouped positionally on `job`. let qe = ok("sum by (job) (up)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; @@ -273,19 +238,17 @@ fn all_targets_missing_core_lowers() { #[test] fn scalar_threshold_comparisons_lower_to_binaryop_scalar() { // ~822/949 corpus queries are ` `. The numeric - // threshold is now a `PromqlScalarBridge` operand of the `BinaryOp` (issue + // threshold is now a `ScalarExpr` operand of the `BinaryOp` (issue // #35) — the single biggest unblock for real alerts. for q in [ "prometheus_config_last_reload_successful != 1", "increase(prometheus_tsdb_compactions_failed_total[1m]) > 0", "rate(alertmanager_notifications_failed_total[3m]) > 0.05", ] { - let QueryExpr::BinaryOp { rhs, .. } = ok(q) else { - panic!("expected a BinaryOp for {q:?}"); - }; + let qe = ok(q); assert!( - matches!(rhs.as_ref(), QueryExpr::PromqlScalarBridge(_)), - "scalar threshold operand for {q:?}, got {rhs:?}" + matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. }), + "{q}" ); } } @@ -336,12 +299,12 @@ fn vector_literal_lowers_to_a_labelless_vector() { // `vector(1)` — used in dead-man's-switch ("always firing") alerts. Now // lowers to a `PromqlVectorFromScalar` over the scalar `1` (issue #48). let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; - assert_eq!(inner.as_promql_scalar(), Some(1.0)); + assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); // The result is a vector: it carries a time index (unlike a bare scalar). - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(qe.schema.time_index.is_some()); } #[test] @@ -352,14 +315,14 @@ fn without_grouping_lowers_to_the_exclusion_form() { // labels are stored and the kept set is runtime-resolved (issue #39). let qe = ok(r#"(min without (cpu) (rate(node_cpu_seconds_total{mode="idle"}[1h]))) > 0.8"#); // Top level is the `> 0.8` comparison; the `min without (cpu)` is its LHS. - let QueryExpr::BinaryOp { lhs, .. } = &qe else { + let NonASAPOp::Filter { child: lhs, .. } = qe.expect_non_asap() else { panic!("expected a comparison BinaryOp, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = lhs.as_ref() + } = lhs.expect_non_asap() else { panic!("expected a `min without` Aggregate on the LHS, got {lhs:?}"); }; diff --git a/crates/frontend-promql/tests/observability/metrics_observability.rs b/crates/frontend-promql/tests/observability/metrics_observability.rs index 87d653ef2..24178c20f 100644 --- a/crates/frontend-promql/tests/observability/metrics_observability.rs +++ b/crates/frontend-promql/tests/observability/metrics_observability.rs @@ -6,15 +6,14 @@ use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; -use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, -}; use asap_frontend_promql::PromqlError; +use asap_logical_optimizer::pass1::replacement::{retain_exact, RealizationError}; +use asap_logical_optimizer::{ + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +}; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; @@ -64,19 +63,18 @@ fn queries(corpus: &str) -> impl Iterator { .filter(|line| !line.is_empty() && !line.starts_with('#')) } -fn post_asap_candidate(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn post_asap_candidate(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default() .replacements(&target) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. }) => Ok(node), - _ => keep_pre_asap(&root), + _ => retain_exact(root), } } @@ -99,9 +97,8 @@ fn benchmark_corpora_are_total_and_report_coverage() { Ok(expr) => { lowered += 1; match post_asap_candidate(&expr) { - Ok(node) if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) => { - post_asap_candidates += 1 - } + // An ASAP operator bound somewhere below the root. + Ok(node) if node.contains_asap() => post_asap_candidates += 1, Ok(_) => { post_asap_unchanged += 1; if std::env::var_os("METRICS_OBSERVABILITY_REPORT").is_some() { diff --git a/crates/frontend-promql/tests/observability/promql_corpus.rs b/crates/frontend-promql/tests/observability/promql_corpus.rs index 1bc7e2166..606b2dbab 100644 --- a/crates/frontend-promql/tests/observability/promql_corpus.rs +++ b/crates/frontend-promql/tests/observability/promql_corpus.rs @@ -15,37 +15,35 @@ use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; -use asap_aware_mapping::{ - Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, -}; use asap_frontend_promql::PromqlError as LoweringError; +use asap_logical_optimizer::pass1::replacement::{retain_exact, RealizationError}; +use asap_logical_optimizer::{ + ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +}; #[path = "../support.rs"] mod support; -use asap_types::post_asap::{SummaryExpr, SummaryNode}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; -/// This crate has no "bind me one DAG" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// This crate has no "bind me one dag" public API any more — +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so [`bind_tally`] /// gets one representative `Result` per query, matching what a totality /// check over the whole corpus wants. -fn bind(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn bind(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default() .replacements(&target) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. }) => Ok(node), - _ => keep_pre_asap(&root), + _ => retain_exact(root), } } @@ -77,7 +75,10 @@ impl Tally { fn tally(corpus: &str) -> Tally { let mut t = Tally::default(); for q in queries(corpus) { - match lower_promql(q, AccuracyTarget::Exact) { + match asap_frontend_promql::lower_promql_query_workload( + &support::workload(q, AccuracyTarget::Exact), + 0, + ) { Ok(_) => t.lowered += 1, Err(LoweringError::Parse(_)) => t.unparseable += 1, Err(_) => t.rejected += 1, @@ -87,15 +88,16 @@ fn tally(corpus: &str) -> Tally { } /// Every query that lowers, additionally run through the pre-ASAP → -/// post-ASAP `asap-aware-mapping` binding pass (issue #98), at an +/// post-ASAP `asap-logical-optimizer` binding pass (issue #98), at an /// approximate accuracy target so the sketch-selection boundary actually /// fires (an `Exact` target would only ever exercise the exact-accumulator /// arm). #[derive(Default, Debug)] struct BindTally { - /// Root bound to `SummaryAgg`/`SummaryEstimate` — the pass did something. + /// An ASAP operator was bound somewhere below the root — the pass did + /// something. transformed: usize, - /// Root stayed `KeepPreAsap` — the pass left the query untouched. + /// The kept pre-ASAP dag — the pass left the query untouched. unchanged: usize, /// [`bind`] returned `Err` (schema derivation failed). errored: usize, @@ -108,7 +110,7 @@ fn bind_tally(corpus: &str, accuracy: AccuracyTarget) -> BindTally { continue; }; match bind(&dag) { - Ok(bound) if matches!(bound.expr, SummaryExpr::KeepPreAsap(_)) => t.unchanged += 1, + Ok(bound) if !bound.contains_asap() => t.unchanged += 1, Ok(_) => t.transformed += 1, Err(_) => t.errored += 1, } @@ -148,23 +150,17 @@ fn lowering_is_total_over_the_entire_corpus() { "testdata corpus unexpectedly small: {td:?}" ); - // Coverage tripwire: a code change that breaks lowering for a large slice of - // real PromQL trips this. Current numbers on the private promql-parser `asap` - // branch: docs 48 lowered / 1 rejected, testdata 1512 lowered / 76 rejected / - // 235 unparseable. The floors sit ~1% under those, so they guard regressions - // rather than pin an exact count — ratchet them up as coverage lands. - // - // The 235 unparseable are parser-fork gaps (issue #108); the rejections are - // lowering gaps (#109). Both shrink over time, so these floors normally only - // rise. Exception: the testdata floor was lowered to the measured 1485 when - // the 44 `fill` vector-matching queries became rejected rather than - // silently lowered without their fill semantics. + // Coverage tripwire after rejecting unrepresented native histogram samples: + // docs 48 lowered / 1 rejected; testdata 1121 lowered / 469 rejected / + // 233 parser gaps. Earlier coverage counted native histogram operations + // incorrectly treated as float quantiles. Keep the rejection cases in the + // corpus: accepting them requires a native histogram sample representation. assert!( docs.lowered >= 47, "docs lowering coverage regressed: {docs:?}" ); assert!( - td.lowered >= 1485, + td.lowered >= 1121, "testdata lowering coverage regressed: {td:?}" ); } diff --git a/crates/frontend-promql/tests/promql_binding_regressions.rs b/crates/frontend-promql/tests/promql_binding_regressions.rs index 26d1be06d..daa88a967 100644 --- a/crates/frontend-promql/tests/promql_binding_regressions.rs +++ b/crates/frontend-promql/tests/promql_binding_regressions.rs @@ -33,9 +33,10 @@ fn irate_and_rate_have_distinct_canonical_intents() { /// PromQL count counts series even when two sample values are equal. #[test] fn count_is_row_count_not_distinct_sample_value_count() { - use asap_types::pre_asap::{AggIntent, QueryExpr}; - let dag = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { measures, .. } = dag else { + use asap_types::ir::operator::AggIntent; + use asap_types::ir::NonASAPOp; + let tree = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); + let NonASAPOp::Aggregate { measures, .. } = tree.expect_non_asap() else { panic!("expected aggregate") }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); diff --git a/crates/frontend-promql/tests/promql_conformance.rs b/crates/frontend-promql/tests/promql_conformance.rs index 6f932879e..eae9b710a 100644 --- a/crates/frontend-promql/tests/promql_conformance.rs +++ b/crates/frontend-promql/tests/promql_conformance.rs @@ -31,22 +31,27 @@ // `__GAP`-suffixed test names intentionally SHOUT the documented divergences. #![allow(non_snake_case)] +use std::rc::Rc; use std::time::Duration; use asap_frontend_promql::PromqlError as LoweringError; mod support; -use asap_types::pre_asap::schema::DataType; -use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, MathFunc, - PromQLVectorSetOpKind, QueryExpr, Reduction, SampleKind, Source, TimeFunc, +use asap_types::ir::operator::{ + AggIntent, AtModifier, BinaryOpKind, PromQLVectorSetOpKind, Reduction, SampleKind, Source, + TimeFunc, +}; +use asap_types::ir::scalar::{ArithmeticOpKind, CompareOpKind, ScalarValue}; +use asap_types::ir::schema::DataType; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, }; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, promql_scalar}; // ── harness helpers ───────────────────────────────────────────────────────────── /// Lower, expecting success. -fn ok(q: &str) -> QueryExpr { +fn ok(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact) .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } @@ -59,72 +64,30 @@ fn rejected(q: &str) -> LoweringError { } } -/// Every `AggIntent` anywhere in the DAG, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { +/// Every `AggIntent` anywhere in the tree, root-to-leaf. +fn intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); collect(e, &mut out); out } -fn collect(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - collect(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => collect(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { - collect(lhs, out); - collect(rhs, out); - } - QueryExpr::Join { left, right, .. } | QueryExpr::SetOp { left, right, .. } => { - collect(left, out); - collect(right, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| collect(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - collect(inner, out) - } - // `AggIntent` only ever lives in `Aggregate.measures`, never in a - // scalar position (issue #205) — nothing to collect there. - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} +/// `AggIntent` only ever lives in `Aggregate.measures`, never in a scalar +/// position (issue #205); `children()` also descends into the operators a +/// scalar position reads (`scalar(v)`). +fn collect(e: &OperatorNode, out: &mut Vec) { + if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { + out.extend(measures.iter().cloned()); + } + for child in e.children() { + collect(child, out); } } /// The first `Scan` reached by descending single-child nodes, with its metric /// name and predicate count. -fn first_scan(e: &QueryExpr) -> (String, usize) { - match e { - QueryExpr::Scan { +fn first_scan(e: &OperatorNode) -> (String, usize) { + match e.expect_non_asap() { + NonASAPOp::Scan { source, predicates, .. } => { let name = match source { @@ -133,45 +96,32 @@ fn first_scan(e: &QueryExpr) -> (String, usize) { }; (name, predicates.len()) } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_scan(child), + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_scan(child), other => panic!("no Scan reachable from {other:?}"), } } -fn has bool>(e: &QueryExpr, pred: F) -> bool { +fn has bool>(e: &OperatorNode, pred: F) -> bool { intents(e).iter().any(pred) } -/// Whether the DAG contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary +/// Whether the tree contains a `Mul`-by-`ScalarExpr(-1)` anywhere — the shape unary /// negation lowers to (issue #36). -fn negates_via_scalar(e: &QueryExpr) -> bool { - let is_neg_one = |q: &QueryExpr| { - q.as_promql_scalar() - .is_some_and(|v| (v + 1.0).abs() < 1e-12) - }; - match e { - QueryExpr::BinaryOp { op, lhs, rhs, .. } => { - (*op == BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul) - && (is_neg_one(lhs) || is_neg_one(rhs))) - || negates_via_scalar(lhs) - || negates_via_scalar(rhs) - } - QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Project { child, .. } => negates_via_scalar(child), - _ => false, +fn negates_via_scalar(e: &OperatorNode) -> bool { + fn negative(expr: &ScalarExpr) -> bool { + matches!(expr, ScalarExpr::Negative { .. }) || expr.children().iter().any(|e| negative(e)) } + e.expect_non_asap() + .scalar_exprs() + .iter() + .any(|e| negative(e)) + || e.children().iter().any(|e| negates_via_scalar(e)) } // ───────────────────────────────────────────────────────────────────────────── @@ -193,10 +143,10 @@ fn promql_scan_schema_is_open() { // runtime-only, so the binding schema lists only the (ts, value) floor + // referenced labels and may be a subset of the runtime row. let qe = ok("node_cpu_seconds_total"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected a TimeRange for a bare selector, got {qe:?}"); }; - let QueryExpr::Scan { schema, .. } = child.as_ref() else { + let NonASAPOp::Scan { schema, .. } = child.expect_non_asap() else { panic!("expected a Scan inside the TimeRange, got {qe:?}"); }; assert!( @@ -241,7 +191,7 @@ fn range_vector_selector_is_time_range() { // SEMANTICS: `[5m]` turns an instant vector into a range vector, // represented in the canonical DAG as a dedicated `TimeRange` node. let qe = ok("node_cpu_seconds_total[5m]"); - let QueryExpr::TimeRange { range, .. } = &qe else { + let NonASAPOp::TimeRange { range, .. } = qe.expect_non_asap() else { panic!("expected TimeRange for a range-vector selector, got {qe:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -252,19 +202,44 @@ fn range_vector_selector_is_time_range() { // functions.test) // ───────────────────────────────────────────────────────────────────────────── +#[test] +fn selector_time_ranges_carry_their_kind() { + // SEMANTICS: an instant selector reads the latest sample within the + // ingestion interval (`Instant`); `m[5m]` is a range selection (`Range`). + // Same length is not the same shape: `m` and `m[1s]` stay distinct. + assert!(matches!( + ok("node_cpu_seconds_total").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Instant, + .. + } + )); + assert!(matches!( + ok("node_cpu_seconds_total[5m]").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); + assert_ne!( + ok("node_cpu_seconds_total"), + ok("node_cpu_seconds_total[1s]") + ); +} + #[test] fn rate_range_lives_in_time_range_node() { // SEMANTICS: per-second average rate; the temporal range lives on the // enclosing `TimeRange` node, not inside the intent. let qe = ok("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -281,14 +256,14 @@ fn irate_maps_to_its_own_intent() { #[test] fn increase_range_lives_in_time_range_node() { let qe = ok("increase(http_requests_total[1h])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -303,7 +278,7 @@ fn increase_range_lives_in_time_range_node() { fn sum_collapses_all_series() { // SEMANTICS: `sum(v)` → one output series. No grouping → no Partition. let qe = ok("sum(node_filesystem_size_bytes)"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -314,12 +289,12 @@ fn sum_by_groups_via_positional_aggregate() { // name-based Partition). SchemaResolver leaf = [ts, value, instance, job] (referenced // keys appended sorted), so the keys resolve to columns [2, 3]. let qe = ok("sum by(job, instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected positional Aggregate for `by(...)`, got {qe:?}"); }; @@ -330,7 +305,7 @@ fn sum_by_groups_via_positional_aggregate() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); } @@ -369,11 +344,11 @@ fn sum_without_groups_by_the_complement() { // the runtime: the grouping is the exclusion form and the output schema // stays OPEN (unlike `by`, which freezes to closed). let qe = ok("sum without(instance) (node_filesystem_size_bytes)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -385,7 +360,7 @@ fn sum_without_groups_by_the_complement() { assert_eq!(by.keys().len(), 1, "the one excluded label (instance)"); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - !qe.output_schema().unwrap().closed, + !qe.schema.clone().closed, "a `without` result keeps an open schema (kept label set is runtime-only)" ); } @@ -418,16 +393,16 @@ fn group_aggregator_lowers_to_a_distinct_intent() { fn sum_of_rate_is_two_levels() { // SEMANTICS: per-series rate, THEN cross-series sum. Both must survive. let qe = ok("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -436,12 +411,12 @@ fn sum_by_of_rate_groups_outer_level() { // Outer cross-series Sum grouped on positional `Aggregate.by` over the // label-preserving inner Rate. Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (rate(node_network_receive_bytes_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by instance, got {qe:?}"); }; @@ -449,8 +424,8 @@ fn sum_by_of_rate_groups_outer_level() { assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // child is the inner per-series Rate aggregate. assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -461,26 +436,29 @@ fn sum_by_of_over_time_groups_outer_level() { // preserving, so the key resolves positionally just like the rate case (no // name-based Partition). Leaf = [ts, value, instance] → by = [2]. let qe = ok("sum by(instance) (avg_over_time(node_cpu_seconds_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by instance, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // child is the inner per-series reduction: Aggregate{Avg} over TimeRange. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (per-series avg_over_time) under the Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ───────────────────────────────────────────────────────────────────────────── @@ -501,7 +479,7 @@ fn over_time_functions_reduce_over_time_range() { ] { let qe = ok(q); assert!( - matches!(&qe, QueryExpr::Aggregate { .. }), + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. }), "{q}: expected Aggregate" ); let matched = intents(&qe).iter().any(|i| match want { @@ -519,7 +497,7 @@ fn over_time_functions_reduce_over_time_range() { #[test] fn quantile_over_time_is_aggregate_over_time_range() { let qe = ok("quantile_over_time(0.9, request_latency_seconds[5m])"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has( &qe, |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) @@ -536,7 +514,7 @@ fn histogram_quantile_over_rate() { // φ-quantile from bucket rates. The `_bucket` metric marks the classic // cumulative-bucket form → `HistogramQuantile` (even without `sum by (le)`). let qe = ok("histogram_quantile(0.9, rate(demo_api_request_duration_seconds_bucket[5m]))"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected Aggregate{{HistogramQuantile}}, got {qe:?}"); }; assert!( @@ -553,9 +531,9 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { let qe = ok( "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", ); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; @@ -566,11 +544,11 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { )); // `sum by(le)` now survives as a positional Aggregate (by = [2], `le`), over // the inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected `sum by(le)` as a positional Aggregate, got {child:?}"); }; @@ -586,7 +564,11 @@ fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { #[test] fn vector_arithmetic() { let qe = ok("node_memory_MemFree_bytes + node_memory_Cached_bytes"); - let QueryExpr::BinaryOp { op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Add)); @@ -597,14 +579,17 @@ fn on_matching_with_group_left() { // SEMANTICS: many-to-one matching on a label subset. let qe = ok("rate(demo_cpu_usage_seconds_total[1m]) / on(instance, job) group_left demo_num_cpus"); - let QueryExpr::BinaryOp { - op, vector_match, .. - } = &qe - else { + let NonASAPOp::BinaryOp { operator, .. } = qe.expect_non_asap() else { panic!("expected BinaryOp, got {qe:?}"); }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); - let vm = vector_match.as_ref().expect("on(...) group_left present"); + assert_eq!( + operator.kind, + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + ); + let vm = operator + .vector_match + .as_ref() + .expect("on(...) group_left present"); assert_eq!(vm.labels, vec!["instance".to_string(), "job".to_string()]); assert!( vm.grouping.is_some(), @@ -617,15 +602,51 @@ fn vector_comparison_filters() { // SEMANTICS: `>` between two vectors keeps the LHS series where it holds. let qe = ok("go_goroutines > go_threads"); assert!( - matches!(&qe, QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) + matches!(qe.expect_non_asap(), NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) ); } +#[test] +fn comparison_bool_modifier_returns_zero_or_one() { + // SEMANTICS (operators.test): `bool` turns a filtering comparison into a + // 0/1-valued one. On a vector operand it is `return_bool` on the + // `BinaryOp`; between two scalars it is a `Case(Compare → 1, else 0)` + // scalar expression under PromQL numeric rules — and a scalar comparison + // without `bool` is not a PromQL expression at all. + let bool_flag = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { return_bool, .. } => *return_bool, + NonASAPOp::Project { .. } => true, + NonASAPOp::Filter { .. } => false, + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert!(bool_flag("go_goroutines > bool go_threads")); + assert!(bool_flag("go_goroutines > bool 0")); + assert!(!bool_flag("go_goroutines > go_threads")); + assert!(!bool_flag("go_goroutines > 0")); + + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { branches, .. } = &qe else { + panic!("expected a scalar Case, got {qe:?}"); + }; + assert!(matches!( + branches.as_slice(), + [( + ScalarExpr::Compare { + op: CompareOpKind::Lt, + semantics: ExprSemantics::Promql, + .. + }, + _ + )] + )); + rejected("1 < 2"); +} + #[test] fn unary_negation_lowers_as_multiply_by_minus_one() { // SEMANTICS (PromQL, issue #36): `-expr` flips the sign of every sample. // Now that a scalar operand exists (#35), it lowers as `expr * -1` — a `Mul` - // BinaryOp of the (label-preserving) vector against `PromqlScalarBridge(-1)`. These are + // BinaryOp of the (label-preserving) vector against `ScalarExpr(-1)`. These are // the five cases the old `__GAP` test pinned as rejected. for q in [ "-rate(http_errors_total[5m])", @@ -635,93 +656,38 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { "sum(-node_cpu_seconds_total)", ] { let qe = ok(q); - // A `Mul`-by-`-1` against a `PromqlScalarBridge(-1)` appears somewhere in every DAG. + // A `Mul`-by-`-1` against a `ScalarExpr(-1)` appears somewhere in every tree. assert!( negates_via_scalar(&qe), "no `* -1` negation found in {q}: {qe:?}" ); } - // `-some_metric` at the root: `Scan * PromqlScalarBridge(-1)`, schema follows the vector. - let QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, - } = &ok("-some_metric") - else { - panic!("expected a BinaryOp for `-some_metric`"); - }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), - "vector on the left" - ); - assert!( - rhs.as_promql_scalar() - .is_some_and(|v| (v + 1.0).abs() < 1e-12), - "negation multiplies by PromqlScalarBridge(-1), got {rhs:?}" - ); - assert!( - vector_match.is_none(), - "scalar negation carries no vector match" - ); - // Label-preserving: the schema is the vector operand's, unchanged. - let schema = ok("-some_metric").output_schema().unwrap(); - assert_eq!( - schema - .fields - .iter() - .map(|c| c.name.as_str()) - .collect::>(), - vec!["ts", "value"], - ); - - // `sum(-m)` — the negation lowers inside the aggregate argument (issue #27 - // nesting), so the outer node is the `Sum` aggregate over the `Mul`. - let QueryExpr::Aggregate { - measures, child, .. - } = &ok("sum(-node_cpu_seconds_total)") - else { - panic!("expected an outer Aggregate for `sum(-m)`"); - }; - assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!( - child.as_ref(), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - .. - } - )); + let negated = ok("-some_metric"); + assert!(negates_via_scalar(&negated)); + assert!(negated.schema.has_promql_series_identity()); + assert!(negated.schema.time_index.is_some()); + let summed = ok("sum(-node_cpu_seconds_total)"); + assert!(has(&summed, |i| matches!(i, AggIntent::Sum { .. }))); + assert!(negates_via_scalar(&summed)); } #[test] fn unary_negation_of_constant_folds_to_scalar() { // `-(10*1024*1024)` — the operand is constant-foldable, so negation collapses - // to a single negated `PromqlScalarBridge` leaf (no `BinaryOp`), just like a bare literal. - assert!(ok("-(10*1024*1024)") - .as_promql_scalar() + // to a single negated `ScalarExpr` leaf (no `BinaryOp`), just like a bare literal. + assert!(promql_scalar(&support::scalar_root("-(10*1024*1024)")) .is_some_and(|v| (v + 10_485_760.0).abs() < 1e-6)); } #[test] fn double_unary_negation_nests() { - // `- -some_metric` — negation of a negation: `(m * -1) * -1`. Both levels - // lower; the value is unchanged but the structure is faithfully nested. - let QueryExpr::BinaryOp { op, lhs, .. } = &ok("- -some_metric") else { - panic!("expected outer BinaryOp for `- -some_metric`"); + let qe = ok("- -some_metric"); + let NonASAPOp::Project { child, .. } = qe.expect_non_asap() else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!( - matches!( - lhs.as_ref(), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - .. - } - ), - "inner negation nests under the outer one" - ); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); + assert!(negates_via_scalar(child)); } #[test] @@ -754,39 +720,22 @@ fn count_maps_to_count_and_inherits_accuracy() { #[test] fn scalar_literal_operand_lowers_as_binaryop_scalar() { - // Issue #35: ` op ` — the numeric threshold is a - // `PromqlScalarBridge` operand of the `BinaryOp`, and constant arithmetic - // (`10*1024*1024`) is folded. The output schema is the vector side's. let qe = ok("node_filesystem_avail_bytes > 10*1024*1024"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Compare { op, right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Compare(CompareOpKind::Gt)); - assert!( - matches!(lhs.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), - "vector on the left" - ); - assert!( - rhs.as_promql_scalar() - .is_some_and(|v| (v - 10_485_760.0).abs() < 1e-6), - "folded scalar threshold on the right, got {rhs:?}" - ); - // Schema derivation follows the vector side (a scalar contributes no labels). - assert!(qe.output_schema().is_ok()); + assert_eq!(*op, CompareOpKind::Gt); + assert_eq!(promql_scalar(right), Some(10_485_760.0)); } #[test] fn scalar_arithmetic_scales_the_vector() { - // `rate(m[5m]) * 100` — a unit conversion. Arithmetic BinaryOp of the vector - // with a `PromqlScalarBridge(100)`. let qe = ok("rate(m[5m]) * 100"); - let QueryExpr::BinaryOp { op, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Arithmetic { op, right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul)); - assert!(rhs - .as_promql_scalar() - .is_some_and(|v| (v - 100.0).abs() < 1e-9)); + assert_eq!(*op, ArithmeticOpKind::Mul); + assert_eq!(promql_scalar(right), Some(100.0)); } // ───────────────────────────────────────────────────────────────────────────── @@ -797,12 +746,22 @@ fn scalar_arithmetic_scales_the_vector() { #[test] fn set_ops_lower_to_binaryop() { // SEMANTICS: or = union of label sets; and = intersection; unless = difference. - assert!(matches!(&ok("up{job=\"a\"} or up{job=\"b\"}"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Or))); - assert!(matches!(&ok("node_network_mtu_bytes and node_up"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::And))); - assert!(matches!(&ok("node_network_mtu_bytes unless node_down"), - QueryExpr::BinaryOp { op, .. } if *op == BinaryOpKind::Set(PromQLVectorSetOpKind::Unless))); + let set_op = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { operator, .. } => operator.kind.clone(), + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert_eq!( + set_op("up{job=\"a\"} or up{job=\"b\"}"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Or) + ); + assert_eq!( + set_op("node_network_mtu_bytes and node_up"), + BinaryOpKind::Set(PromQLVectorSetOpKind::And) + ); + assert_eq!( + set_op("node_network_mtu_bytes unless node_down"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) + ); } // ───────────────────────────────────────────────────────────────────────────── @@ -824,7 +783,7 @@ fn topk_over_count_is_heavy_hitter() { fn bottomk_is_generic_sort_limit() { // SEMANTICS: bottom-k → generic ascending order + limit (no sketch). let qe = ok("bottomk(3, count_over_time(http_requests_total[5m]))"); - assert!(matches!(&qe, QueryExpr::Limit { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); } #[test] @@ -833,9 +792,9 @@ fn topk_over_nested_sum_preserves_weighted_topk_accuracy() { // The final rates are query-time values. Their ordering does not establish // frequency-sketch membership semantics. let qe = ok("topk(3, sum by(instance) (rate(node_cpu_seconds_total[5m])))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected weighted TopK aggregate, got {qe:?}"); }; @@ -861,18 +820,18 @@ fn outer_aggregate_over_nested_aggregate_nests() { // flat two-level template rejected. Each level survives into the // canonical DAG (issue #27). let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner `sum by (job)` Aggregate, got {child:?}"); }; @@ -895,12 +854,12 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { // the query lowers with the provably-absent key dropped, exactly // `sum(sum by (group)(…))`. let qe = ok(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -910,7 +869,7 @@ fn outer_group_key_absent_from_nested_aggregate_is_dropped() { "absent `job` key dropped → global aggregate" ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { reduction, .. } = child.expect_non_asap() else { panic!("expected inner `sum by (group)` Aggregate, got {child:?}"); }; assert_eq!( @@ -927,16 +886,16 @@ fn outer_group_key_present_after_inner_aggregate_still_resolves() { // resolving positionally — the absent-key drop only fires on provable // absence, never on a resolvable key. let qe = ok("sum(sum by (job, group)(http_requests)) by (job)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate, got {child:?}"); }; @@ -957,9 +916,9 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // still resolve. Each `or` side is bound independently against its own // sub-DAG, so the key is seeded as an inherited column on both sides. let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -969,12 +928,12 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { 1, "grouped by the one `__name__` key" ); - let QueryExpr::BinaryOp { lhs, rhs, .. } = child.as_ref() else { + let NonASAPOp::BinaryOp { lhs, rhs, .. } = child.expect_non_asap() else { panic!("expected a BinaryOp child, got {child:?}"); }; // Both independently-bound sides carry `__name__` at the same position, so // the outer group key is consistent across the union. - let (ls, rs) = (lhs.output_schema().unwrap(), rhs.output_schema().unwrap()); + let (ls, rs) = (lhs.schema.clone(), rhs.schema.clone()); assert_eq!(ls.column_id("__name__"), rs.column_id("__name__")); assert_eq!( ls.column_id("__name__"), @@ -983,8 +942,8 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // The general case (a plain label, not just `__name__`) also lowers. assert!(matches!( - ok("sum by (job)(metric_a or metric_b)"), - QueryExpr::Aggregate { .. } + ok("sum by (job)(metric_a or metric_b)").expect_non_asap(), + NonASAPOp::Aggregate { .. } )); } @@ -994,15 +953,15 @@ fn aggregate_over_binary_op_nests() { // op over two range vectors. The old template only accepted a single inner // selector/call; now the binary op lowers and the outer sum wraps it. let qe = ok("sum(rate(a[5m]) + rate(b[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!( - matches!(child.as_ref(), QueryExpr::BinaryOp { .. }), + matches!(child.expect_non_asap(), NonASAPOp::BinaryOp { .. }), "argument lowers as a BinaryOp, got {child:?}" ); } @@ -1015,7 +974,10 @@ fn aggregate_over_binary_op_nests() { fn subquery_wraps_inner_query() { // SEMANTICS: `[range:res]` evaluates the inner query across a range. let qe = ok("rate(demo_api_request_duration_seconds_count[5m])[1h:]"); - assert!(matches!(&qe, QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); } @@ -1026,12 +988,12 @@ fn over_time_of_subquery_reduces_per_series() { // then `max_over_time` takes the max of those samples *per series*. It lowers // to a per-series `Max` reduction over a `PromqlSubquery` (issue #27). let qe = ok("max_over_time(rate(demo_api_request_duration_seconds_count[5m])[1h:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate at the root, got {qe:?}"); }; @@ -1044,7 +1006,7 @@ fn over_time_of_subquery_reduces_per_series() { // The reduction rides directly on the sub-query (the structural range marker // that keeps it label-preserving), which wraps the inner `rate`. assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), "the `Max` reduces over a PromqlSubquery, got {child:?}" ); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Rate))); @@ -1055,16 +1017,19 @@ fn quantile_over_time_of_subquery_carries_phi() { // The `quantile_over_time` φ parameter is read from arg 0; the sub-query is // arg 1. It lowers to a per-series `Quantile(φ)` over the `PromqlSubquery`. let qe = ok("quantile_over_time(0.9, rate(demo[5m])[1h:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.9).abs() < 1e-9) ); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); } #[test] @@ -1074,12 +1039,12 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { // survives for the OUTER cross-series `sum by (job)` to group on. If the // inner `Max` collapsed labels, `job` would not resolve here. let qe = ok("sum by (job) (max_over_time(rate(demo{job=\"api\"}[5m])[1h:]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1089,20 +1054,20 @@ fn aggregation_over_over_time_of_subquery_keeps_labels() { ); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner node is the per-series `max_over_time` reduction over the subquery. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, measures: inner_measures, child: inner_child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate, got {child:?}"); }; assert_eq!(inner_reduction, &Reduction::PerEntity); assert!(matches!(inner_measures.as_slice(), [AggIntent::Max { .. }])); assert!(matches!( - inner_child.as_ref(), - QueryExpr::PromqlSubquery { .. } + inner_child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } )); } @@ -1124,66 +1089,66 @@ fn nested_subquery_from_prometheus_docs() { // the label-preserving `[ts, value]`. let qe = ok("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected `max_over_time` Aggregate at the root, got {qe:?}"); }; assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range, resolution, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the outer `[10m:]` PromqlSubquery, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(600)); assert_eq!(*resolution, None, "`[10m:]` keeps the default resolution"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the `deriv` Aggregate, got {child:?}"); }; assert_eq!(reduction, &Reduction::PerEntity); assert!(matches!(measures.as_slice(), [AggIntent::Deriv])); - let QueryExpr::PromqlSubquery { + let NonASAPOp::PromqlSubquery { range, resolution, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the inner `[30s:5s]` PromqlSubquery, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(30)); assert_eq!(*resolution, Some(Duration::from_secs(5))); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected the `rate` Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected the `[5s]` TimeRange under rate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(5)); // Per-series end to end: the schema keeps the (ts, value) floor and stays open. - let schema = qe.output_schema().expect("schema derivation"); + let schema = qe.schema.clone(); assert_eq!( schema .fields @@ -1205,21 +1170,22 @@ fn offset_modifier_lowers_to_a_time_shift() { // past — a `TimeShift` wrapper over the selector (signed ms; a negative // offset shifts forward). Schema is unchanged (the shift only moves *when*). let qe = ok("http_requests_total offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.offset_ms, 300_000); assert!(shift.at.is_none()); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); // `offset -5m` shifts forward → negative ms. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total offset -5m") else { + let qe = ok("http_requests_total offset -5m"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift"); }; assert_eq!(shift.offset_ms, -300_000); @@ -1230,29 +1196,31 @@ fn at_modifier_lowers_to_a_time_shift() { // SEMANTICS (PromQL, issue #40): `@ ` pins the evaluation to an absolute // instant (PromQL seconds → IR milliseconds); `@ start()` / `@ end()` anchor // to the query range bounds. - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ 1609746000") else { + let qe = ok("http_requests_total @ 1609746000"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift for `@ `"); }; assert_eq!(shift.at, Some(AtModifier::Timestamp(1_609_746_000_000))); assert_eq!(shift.offset_ms, 0); - let QueryExpr::TimeRange { child, .. } = &ok("http_requests_total @ start()") else { + let qe = ok("http_requests_total @ start()"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift for `@ start()`"); }; assert_eq!(shift.at, Some(AtModifier::Start)); // Offset and `@` compose: `@ end() offset 5m` carries both. let qe = ok("http_requests_total @ end() offset 5m"); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected an ingestion TimeRange, got {qe:?}"); }; - let QueryExpr::TimeShift { shift, .. } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { panic!("expected a TimeShift, got {qe:?}"); }; assert_eq!(shift.at, Some(AtModifier::End)); @@ -1265,21 +1233,21 @@ fn offset_on_a_ranged_selector_wraps_inside_the_time_range() { // `TimeShift` sits *under* the `TimeRange` (the 5m window is taken at the // shifted time), and the whole thing under the per-series `Rate` (#40). let qe = ok("rate(http_requests_total[5m] offset 1h)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected the rate Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { child, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { child, .. } = child.expect_non_asap() else { panic!("expected a TimeRange under rate, got {child:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("expected a TimeShift under the TimeRange, got {child:?}"); }; assert_eq!(shift.offset_ms, 3_600_000); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } // ───────────────────────────────────────────────────────────────────────────── @@ -1313,7 +1281,7 @@ fn count_over_time_value_column_is_float64() { // #69: a per-series range reduction produces a PromQL sample value, which is // always float64. `count_over_time`'s `Count` intent types `Int64`, but the // derived `value` column must be `Float64` like every other range reducer. - let schema = ok("count_over_time(m[5m])").output_schema().unwrap(); + let schema = ok("count_over_time(m[5m])").schema.clone(); let value = schema .fields .iter() @@ -1335,12 +1303,12 @@ fn counter_derivative_functions_lower_to_distinct_intents() { ("resets(m[1h])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate for {q:?}, got {qe:?}"); }; @@ -1355,7 +1323,7 @@ fn counter_derivative_functions_lower_to_distinct_intents() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { .. }), + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. }), "{q}: reduction rides on a TimeRange, got {child:?}" ); } @@ -1366,9 +1334,9 @@ fn predict_linear_carries_horizon_seconds() { // `predict_linear(v[w], t)` — the 2nd (scalar) arg is the prediction horizon // in seconds; it must be carried in the intent (it changes the result). let qe = ok("predict_linear(node_filesystem_avail_bytes[3h], 86400)"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -1376,7 +1344,10 @@ fn predict_linear_carries_horizon_seconds() { measures.as_slice(), &[AggIntent::PredictLinear { seconds: 86400.0 }] ); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -1394,12 +1365,12 @@ fn aggregation_over_counter_derivative_keeps_labels() { // A counter-derivative is per-series (label-preserving), so an outer // `sum by (job)` can group on a label the inner `changes` preserved. let qe = ok(r#"sum by (job) (changes(m{job="api"}[15m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1420,12 +1391,12 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { // grouped outer (`avg by (dc)`) must resolve its key against the labels the // inner reduction preserved, threading any scalar param (predict horizon). let qe = ok("avg by (dc) (predict_linear(m[3h], 3600))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate, got {qe:?}"); }; @@ -1434,11 +1405,11 @@ fn outer_stat_over_counter_derivative_nests_two_levels() { "outer `avg by (dc)` groups on a label" ); assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: inner_reduction, measures: inner_measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner per-series Aggregate, got {child:?}"); }; @@ -1458,11 +1429,14 @@ fn topk_over_counter_derivative_is_generic_sort_limit() { // `topk(k, deriv(...))` ranks the per-series derivative values — a generic // `Sort + Limit`, NOT a heavy-hitter `TopK` (that's only `count_over_time`). let qe = ok("topk(3, deriv(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - assert!(matches!(child.as_ref(), QueryExpr::Sort { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Sort { .. })); assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Deriv))); assert!( !intents(&qe) @@ -1477,28 +1451,37 @@ fn counter_derivative_composes_in_binary_ops() { // As a vector operand: `delta(a[5m]) / delta(b[5m])` is a BinaryOp of two // per-series Delta reductions. let ratio = ok("delta(a[5m]) / delta(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &ratio else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = ratio.expect_non_asap() + else { panic!("expected BinaryOp, got {ratio:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) ); // Under an aggregate over a binary op mixing a counter-derivative with // another per-series function: `sum(rate(m[5m]) + changes(m[5m]))`. let mixed = ok("sum(rate(m[5m]) + changes(m[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &mixed + } = mixed.expect_non_asap() else { panic!("expected Aggregate, got {mixed:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::BinaryOp { .. } + )); assert!(intents(&mixed).iter().any(|i| matches!(i, AggIntent::Rate))); assert!(intents(&mixed) .iter() @@ -1522,12 +1505,12 @@ fn range_functions_over_a_subquery_reduce_per_series() { ("resets(sum(m)[5m:])", AggIntent::Resets), ] { let qe = ok(q); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("{q}: expected an Aggregate, got {qe:?}"); }; @@ -1542,7 +1525,7 @@ fn range_functions_over_a_subquery_reduce_per_series() { "{q}: wrong intent" ); assert!( - matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. }), + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), "{q}: reduces directly over the PromqlSubquery (no TimeRange), got {child:?}" ); } @@ -1570,8 +1553,8 @@ fn predict_linear_and_double_exp_over_a_subquery_carry_params() { #[test] fn histogram_quantile_classic_bucket_vs_native() { // Two lowerings of `histogram_quantile(φ, …)`: the classic cumulative-bucket - // form → exact `HistogramQuantile`; a native-histogram / raw-samples argument - // → the generic (sketch-able) `Quantile`. The classic form is recognised by + // form → exact `HistogramQuantile`; native samples require a new type. + // The classic form is recognised by // `by (le)`, a `_bucket` metric, or an `le` matcher (issue #43). for classic in [ "histogram_quantile(0.9, sum by (le) (rate(x_bucket[5m])))", @@ -1595,65 +1578,27 @@ fn histogram_quantile_classic_bucket_vs_native() { "histogram_quantile(0.9, my_native_histogram)", "histogram_quantile(0.9, request_duration_seconds)", // raw samples (your extension) ] { - let qe = ok(native); - assert!( - has( - &qe, - |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) - ), - "native/raw form → generic Quantile: {native}" - ); - assert!( - !has(&qe, |i| matches!(i, AggIntent::HistogramQuantile { .. })), - "{native}" - ); + rejected(native); } } #[test] -fn histogram_accessors_lower_to_per_series_intents() { - // `histogram_(v)` extracts a float per series from a native - // histogram — a per-series `Aggregate{[accessor]}` directly over the - // (instant) argument, no grouping. (`histogram_quantile` has its own two - // lowerings — see `histogram_quantile_classic_bucket_vs_native`.) - for (q, want) in [ - ("histogram_count(v)", AggIntent::HistogramCount), - ("histogram_sum(v)", AggIntent::HistogramSum), - ("histogram_avg(v)", AggIntent::HistogramAvg), - ("histogram_stddev(v)", AggIntent::HistogramStdDev), - ("histogram_stdvar(v)", AggIntent::HistogramStdVar), +fn native_histogram_accessors_are_explicit_gaps() { + // Native histogram samples have no typed representation yet. + for q in [ + "histogram_count(v)", + "histogram_sum(v)", + "histogram_avg(v)", + "histogram_stddev(v)", + "histogram_stdvar(v)", ] { - let qe = ok(q); - let QueryExpr::Aggregate { - reduction, - measures, - .. - } = &qe - else { - panic!("{q}: expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); - assert_eq!( - measures.as_slice(), - std::slice::from_ref(&want), - "{q}: wrong intent" - ); + rejected(q); } } #[test] -fn histogram_fraction_carries_its_bounds() { - // `histogram_fraction(lower, upper, v)` — bounds from args 0/1, vector arg 2. - let qe = ok("histogram_fraction(0, 0.2, v)"); - assert!(intents(&qe).iter().any(|i| matches!( - i, - AggIntent::HistogramFraction { lower, upper } - if *lower == 0.0 && (*upper - 0.2).abs() < 1e-9 - ))); +fn histogram_fraction_is_an_explicit_gap() { + rejected("histogram_fraction(0, 0.2, v)"); } // ───────────────────────────────────────────────────────────────────────────── @@ -1661,68 +1606,42 @@ fn histogram_fraction_carries_its_bounds() { // ───────────────────────────────────────────────────────────────────────────── #[test] -fn math_functions_lower_to_per_series_math_intents() { - // Each `f(v)` is a per-series element-wise value transform — a per-series - // `Aggregate{[Math(f)]}` over the (instant) argument, no grouping. - for (q, want) in [ - ("abs(v)", MathFunc::Abs), - ("ceil(v)", MathFunc::Ceil), - ("floor(v)", MathFunc::Floor), - ("sqrt(v)", MathFunc::Sqrt), - ("ln(v)", MathFunc::Ln), - ("log2(v)", MathFunc::Log2), - ("sgn(v)", MathFunc::Sgn), - ("sin(v)", MathFunc::Sin), - ("atanh(v)", MathFunc::Atanh), - ("deg(v)", MathFunc::Deg), - ("rad(v)", MathFunc::Rad), +fn math_functions_lower_to_typed_scalar_projections() { + for name in [ + "abs", "ceil", "floor", "sqrt", "ln", "log2", "sgn", "sin", "atanh", "deg", "rad", ] { - let qe = ok(q); - let QueryExpr::Aggregate { - reduction, - measures, - .. - } = &qe - else { - panic!("{q}: expected an Aggregate, got {qe:?}"); - }; - assert_eq!( - reduction, - &Reduction::PerEntity, - "{q}: per-series, no grouping" - ); + let query = ok(&format!("{name}(v)")); assert!( - matches!(measures.as_slice(), [AggIntent::Math(m)] if *m == want), - "{q}: wrong intent, got {measures:?}" + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) ); + query.validate_structure().unwrap(); } } #[test] fn clamp_and_round_carry_their_params() { - assert!(intents(&ok("clamp(v, 0, 100)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Clamp { min, max }) if *min == 0.0 && *max == 100.0) - )); - assert!(intents(&ok("clamp_min(v, 1)")) - .iter() - .any(|i| matches!(i, AggIntent::Math(MathFunc::ClampMin { min }) if *min == 1.0))); - assert!(intents(&ok("clamp_max(v, 5)")) - .iter() - .any(|i| matches!(i, AggIntent::Math(MathFunc::ClampMax { max }) if *max == 5.0))); - // `round(v)` defaults the step to 1; `round(v, 5)` reads it. - assert!(intents(&ok("round(v)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Round { to_nearest }) if *to_nearest == 1.0) - )); - assert!(intents(&ok("round(v, 5)")).iter().any( - |i| matches!(i, AggIntent::Math(MathFunc::Round { to_nearest }) if *to_nearest == 5.0) - )); + for (query, params) in [ + ("clamp(v,0,100)", vec![0.0, 100.0]), + ("clamp_min(v,1)", vec![1.0]), + ("clamp_max(v,5)", vec![5.0]), + ("round(v)", vec![1.0]), + ("round(v,5)", vec![5.0]), + ] { + let node = ok(query); + let ScalarExpr::FunctionCall { args, .. } = support::sample_expression(&node) else { + panic!() + }; + assert_eq!( + args.iter().skip(1).map(promql_scalar).collect::>(), + params.into_iter().map(Some).collect::>() + ); + } } #[test] fn pi_lowers_to_a_scalar_constant() { - // `pi()` is the constant π — a `PromqlScalarBridge` leaf, not a `Math` intent. - assert!(ok("pi()") - .as_promql_scalar() + // `pi()` is the constant π — a `ScalarExpr` leaf, not a `Math` intent. + assert!(promql_scalar(&support::scalar_root("pi()")) .is_some_and(|v| (v - std::f64::consts::PI).abs() < 1e-12)); } @@ -1747,7 +1666,7 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { // `absent(v)` synthesizes its output labels from `v`'s equality matchers, so // those labels must survive into the schema — here `job` from `{job="x"}`. let qe = ok(r#"absent(up{job="x"})"#); - let cols = qe.output_schema().unwrap(); + let cols = qe.schema.clone(); assert!( cols.fields.iter().any(|c| c.name == "job"), "matcher label `job` kept, got {:?}", @@ -1761,73 +1680,55 @@ fn absent_keeps_matcher_labels_for_the_synthesized_output() { #[test] fn time_lowers_to_the_eval_time_scalar() { - // SEMANTICS: `time()` is the query evaluation timestamp as a scalar — a leaf, - // not an aggregate over any series. - assert!(matches!(ok("time()"), QueryExpr::EvalTimestamp)); - // …and it is scalar-shaped: a single float `value`, no time index. - let sch = ok("time()").output_schema().unwrap(); - assert_eq!(sch.fields.len(), 1); - assert_eq!(sch.fields[0].name, "value"); - assert!(sch.time_index.is_none()); + assert!(matches!( + support::scalar_root("time()"), + ScalarExpr::EvalTimestamp + )); } #[test] fn time_minus_vector_is_the_uptime_pattern() { - // `time() - process_start_time_seconds` — the canonical uptime expression. - // The scalar `time()` broadcasts against the vector; the result takes the - // vector's schema. let qe = ok("time() - process_start_time_seconds"); - let QueryExpr::BinaryOp { lhs, op, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); - }; - assert!(matches!(lhs.as_ref(), QueryExpr::EvalTimestamp)); - assert!(matches!( - op, - BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub) - )); - assert!(qe.output_schema().is_ok()); + assert!( + matches!(support::sample_expression(&qe), ScalarExpr::Arithmetic { op: ArithmeticOpKind::Sub, left, .. } if matches!(left.as_ref(), ScalarExpr::EvalTimestamp)) + ); + assert!(qe.schema.time_index.is_some()); } #[test] fn calendar_functions_lower_to_time_fn_intents() { - // SEMANTICS: each of these is a per-series float transform of its argument's - // timestamp (or, for `timestamp`, the sample's own time). functions.test. - for (q, want) in [ - ("timestamp(up)", TimeFunc::Timestamp), - ("minute(v)", TimeFunc::Minute), - ("hour(v)", TimeFunc::Hour), - ("day_of_week(v)", TimeFunc::DayOfWeek), - ("day_of_month(v)", TimeFunc::DayOfMonth), - ("day_of_year(v)", TimeFunc::DayOfYear), - ("month(v)", TimeFunc::Month), - ("year(v)", TimeFunc::Year), - ("days_in_month(v)", TimeFunc::DaysInMonth), + assert!(has(&ok("timestamp(up)"), |i| *i + == AggIntent::TimeFn(TimeFunc::Timestamp))); + for name in [ + "minute", + "hour", + "day_of_week", + "day_of_month", + "day_of_year", + "month", + "year", + "days_in_month", ] { - let qe = ok(q); + let query = ok(&format!("{name}(v)")); assert!( - has(&qe, |i| *i == AggIntent::TimeFn(want)), - "{q} → TimeFn({want:?}), got {:?}", - intents(&qe) + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) ); } } #[test] fn no_arg_calendar_function_reads_the_eval_time() { - // `day_of_week()` with no argument computes over the evaluation time itself, - // so it is a `TimeFn` aggregate whose child is the `EvalTimestamp` scalar. - let qe = ok("day_of_week()"); - let QueryExpr::Aggregate { - measures, child, .. - } = &qe - else { - panic!("expected an Aggregate, got {qe:?}"); + let query = ok("day_of_week()"); + let NonASAPOp::Project { child, .. } = query.expect_non_asap() else { + panic!() }; assert!(matches!( - measures.as_slice(), - [AggIntent::TimeFn(TimeFunc::DayOfWeek)] + child.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(ScalarExpr::EvalTimestamp) )); - assert!(matches!(child.as_ref(), QueryExpr::EvalTimestamp)); + assert!( + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name,.. } if name=="promql_day_of_week") + ); } #[test] @@ -1848,30 +1749,23 @@ fn vector_promotes_a_scalar_to_a_vector() { // SEMANTICS: `vector(s)` is the scalar→instant-vector bridge — a label-less // single series carrying the scalar's value. let qe = ok("vector(1)"); - let QueryExpr::PromqlVectorFromScalar(inner) = &qe else { + let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { panic!("expected PromqlVectorFromScalar, got {qe:?}"); }; - assert_eq!(inner.as_promql_scalar(), Some(1.0)); + assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); // Vector-typed: schema has a time index (a scalar leaf has none). - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.time_index.is_some()); assert!(sch.fields.iter().any(|c| c.name == "value")); } #[test] fn scalar_collapses_a_vector_to_a_scalar() { - // SEMANTICS: `scalar(v)` is the instant-vector→scalar bridge. - let qe = ok("scalar(node_load1)"); - let QueryExpr::PromqlScalarFromVector(inner) = &qe else { - panic!("expected PromqlScalarFromVector, got {qe:?}"); + let qe = support::scalar_root("scalar(node_load1)"); + let ScalarExpr::PromqlScalarFromVector(inner) = &qe else { + panic!() }; - let (metric, _) = first_scan(inner); - assert_eq!(metric, "node_load1"); - // PromqlScalarBridge-typed: single `value` column, no time index. - let sch = qe.output_schema().unwrap(); - assert!(sch.time_index.is_none()); - assert_eq!(sch.fields.len(), 1); - assert_eq!(sch.fields[0].name, "value"); + assert_eq!(first_scan(inner).0, "node_load1"); } #[test] @@ -1880,26 +1774,32 @@ fn vector_zero_is_a_vector_operand_of_a_set_op() { // vectors, so `vector(0)` must be a vector (a `PromqlVectorFromScalar`), never a // folded scalar operand. let qe = ok("up or vector(0)"); - let QueryExpr::BinaryOp { rhs, op, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected a BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)); - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlVectorFromScalar(_))); + assert!(matches!( + rhs.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(_) + )); } #[test] fn scalar_of_a_vector_feeds_a_threshold_comparison() { - // `node_load1 > scalar(node_cpu_count)` — `scalar(...)` is a scalar operand, - // so the BinaryOp output takes the vector (lhs) side's schema. let qe = ok("node_load1 > scalar(node_cpu_count)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { - panic!("expected a BinaryOp, got {qe:?}"); + let ScalarExpr::Compare { right, .. } = support::sample_expression(&qe) else { + panic!() }; - assert!(matches!(rhs.as_ref(), QueryExpr::PromqlScalarFromVector(_))); - // The BinaryOp output schema follows the vector (lhs) side, not the scalar. - let (metric, _) = first_scan(lhs); - assert_eq!(metric, "node_load1"); - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(matches!( + right.as_ref(), + ScalarExpr::PromqlScalarFromVector(_) + )); + assert!(qe.schema.time_index.is_some()); } #[test] @@ -1909,13 +1809,13 @@ fn info_lowers_to_a_label_enrichment_join() { // (issue #84). The value/time axis pass through; the enriched labels are // runtime, so the schema stays the child's. let qe = ok("info(rate(http_requests_total[5m]))"); - let QueryExpr::PromqlInfoEnrich { selector, child } = &qe else { + let NonASAPOp::PromqlInfoEnrich { selector, child } = qe.expect_non_asap() else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert!(selector.is_empty(), "no selector → default target_info"); // The child is the untouched input (a per-series rate reduction here). assert!(has(child, |i| *i == AggIntent::Rate)); - assert!(qe.output_schema().unwrap().time_index.is_some()); + assert!(qe.schema.clone().time_index.is_some()); } #[test] @@ -1925,7 +1825,7 @@ fn info_selector_carries_the_info_side_matchers() { // matchers are kept symbolically (not run through the single-metric selector // path). let qe = ok(r#"info(build_info, {__name__=~".+_info", another_data=~".+"})"#); - let QueryExpr::PromqlInfoEnrich { selector, .. } = &qe else { + let NonASAPOp::PromqlInfoEnrich { selector, .. } = qe.expect_non_asap() else { panic!("expected an PromqlInfoEnrich, got {qe:?}"); }; assert_eq!( @@ -1949,12 +1849,12 @@ fn info_composes_under_an_aggregation_and_over_a_time_shift() { // `offset` / `@` on the input now lower to a `TimeShift` under the info-join // (issue #40) — the enrichment composes over the shifted selector. assert!(matches!( - ok("info(metric @ 60)"), - QueryExpr::PromqlInfoEnrich { .. } + ok("info(metric @ 60)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } )); assert!(matches!( - ok("info(metric offset 1m)"), - QueryExpr::PromqlInfoEnrich { .. } + ok("info(metric offset 1m)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } )); } @@ -1967,12 +1867,12 @@ fn group_lowers_to_a_constant_group_intent() { // SEMANTICS: `group(v)` yields a constant 1 per group — a distinct intent, // NOT folded onto `sum` (which would return the value sum instead of 1). let qe = ok("group(up)"); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Group])); // Output column is the constant-1 `group` value. - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "group")); } @@ -1980,7 +1880,7 @@ fn group_lowers_to_a_constant_group_intent() { fn group_by_keeps_the_grouping_keys() { // `group by (job) (up)` — the grouping keys ride on `Aggregate.by`. let qe = ok("group by (job) (up)"); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "job")); assert!(has(&qe, |i| *i == AggIntent::Group)); } @@ -1991,13 +1891,13 @@ fn count_values_groups_by_value_and_synthesizes_a_label() { // value, counts each distinct value, and emits that value as a new label // `l`. The intent carries the label; schema gains a `Utf8` `l` column. let qe = ok(r#"count_values("version", build_version)"#); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::CountValues { label }] if label == "version") ); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); let version = sch .fields .iter() @@ -2023,7 +1923,7 @@ fn count_values_accepts_a_parenthesised_label_and_by_grouping() { &qe, |i| matches!(i, AggIntent::CountValues { label } if label == "v") )); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "job")); assert!(sch.fields.iter().any(|c| c.name == "v")); } @@ -2034,7 +1934,7 @@ fn count_values_label_colliding_with_a_group_key_is_not_duplicated() { // with a group-by key. PromQL's synthesized label takes precedence; the // output must carry a single `job` column, never two. let qe = ok(r#"count_values by (job) ("job", version)"#); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); let jobs = sch.fields.iter().filter(|c| c.name == "job").count(); assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.fields); assert!(sch.fields.iter().any(|c| c.name == "count")); @@ -2046,18 +1946,18 @@ fn limitk_and_limit_ratio_lower_to_series_sampling() { // series kept unchanged (NOT a ranking), so they lower to the dedicated // `PromqlSeriesSample` node, never `topk`'s `Sort → Limit` (issue #86). assert!(matches!( - ok("limitk(2, http_requests)"), - QueryExpr::PromqlSeriesSample { + ok("limitk(2, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitK(2), .. } )); assert!(matches!( - ok("limit_ratio(0.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 + ok("limit_ratio(0.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 )); // Series-preserving: the output schema equals the input's (ts, value). - let sch = ok("limitk(2, http_requests)").output_schema().unwrap(); + let sch = ok("limitk(2, http_requests)").schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some()); } @@ -2067,12 +1967,12 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { // A negative ratio selects the complementary fraction — it must survive, not // be normalised away. Out-of-range magnitudes clamp to [-1, 1] (Prometheus). assert!(matches!( - ok("limit_ratio(-0.5, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 + ok("limit_ratio(-0.5, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 )); assert!(matches!( - ok("limit_ratio(1.1, http_requests)"), - QueryExpr::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 + ok("limit_ratio(1.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 )); } @@ -2080,7 +1980,7 @@ fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { fn limitk_by_carries_the_grouping_and_composes_in_a_set_op() { // `limitk by (group)` samples per group; the grouping label is seeded. let qe = ok("limitk by (group) (2, http_requests)"); - let QueryExpr::PromqlSeriesSample { by, .. } = &qe else { + let NonASAPOp::PromqlSeriesSample { by, .. } = qe.expect_non_asap() else { panic!("expected a PromqlSeriesSample, got {qe:?}"); }; assert!(!by.is_empty(), "grouped sampling keeps its `by` keys"); @@ -2106,20 +2006,20 @@ fn dynamic_and_non_finite_sample_params_are_rejected() { // ───────────────────────────────────────────────────────────────────────────── /// Descend single-child nodes to the first `PromqlRelabel`. -fn first_relabel(e: &QueryExpr) -> &QueryExpr { - match e { - QueryExpr::PromqlRelabel { .. } => e, - QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } => first_relabel(child), +fn first_relabel(e: &OperatorNode) -> &OperatorNode { + match e.expect_non_asap() { + NonASAPOp::PromqlRelabel { .. } => e, + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } => first_relabel(child), other => panic!("no PromqlRelabel reachable from {other:?}"), } } /// True when `value` is a `FunctionCall` with the given name. -fn is_fn_named(value: &QueryExpr, name: &str) -> bool { - matches!(value, QueryExpr::FunctionCall { name: n, .. } if n == name) +fn is_fn_named(value: &ScalarExpr, name: &str) -> bool { + matches!(value, ScalarExpr::FunctionCall { name: n, .. } if n == name) } #[test] @@ -2127,7 +2027,7 @@ fn label_replace_is_a_relabel_over_the_vector() { // SEMANTICS: `label_replace(v, dst, repl, src, regex)` rewrites the `dst` // label per series from a regex over `src`; the sample value is untouched. let qe = ok(r#"label_replace(up, "host", "$1", "instance", "(.+):.*")"#); - let QueryExpr::PromqlRelabel { dst, value, child } = &qe else { + let NonASAPOp::PromqlRelabel { dst, value, child } = qe.expect_non_asap() else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "host"); @@ -2137,7 +2037,7 @@ fn label_replace_is_a_relabel_over_the_vector() { // The value expression is a `label_replace` fn reading the `src` label. assert!(is_fn_named(value, "label_replace")); // Output: the child's columns + the synthesized `host` label; value & ts kept. - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "host")); assert!(sch.fields.iter().any(|c| c.name == "value")); assert!(sch.time_index.is_some(), "the vector's time axis survives"); @@ -2148,12 +2048,12 @@ fn label_join_concatenates_source_labels() { // SEMANTICS: `label_join(v, dst, sep, src…)` joins the source labels with // `sep` into `dst`. let qe = ok(r#"label_join(up, "combined", "-", "job", "instance")"#); - let QueryExpr::PromqlRelabel { dst, value, .. } = &qe else { + let NonASAPOp::PromqlRelabel { dst, value, .. } = qe.expect_non_asap() else { panic!("expected a PromqlRelabel, got {qe:?}"); }; assert_eq!(dst, "combined"); assert!(is_fn_named(value, "label_join")); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "combined")); } @@ -2164,9 +2064,11 @@ fn label_replace_composes_under_an_aggregation() { let qe = ok(r#"sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))"#); // A PromqlRelabel sits below the outer Sum. let relabel = first_relabel(&qe); - assert!(matches!(relabel, QueryExpr::PromqlRelabel { dst, .. } if dst == "host")); + assert!( + matches!(relabel.expect_non_asap(), NonASAPOp::PromqlRelabel { dst, .. } if dst == "host") + ); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); assert!(sch.fields.iter().any(|c| c.name == "host")); } @@ -2191,7 +2093,7 @@ fn extra_over_time_reducers_lower_to_per_series_intents() { assert!(has(&qe, |i| *i == want), "{q}: {:?}", intents(&qe)); // Per-series: the range window survives as a `TimeRange`. assert!( - matches!(&qe, QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeRange { .. })), + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. })), "{q} keeps its range as a TimeRange" ); } @@ -2215,13 +2117,13 @@ fn sort_and_sort_desc_reorder_by_value_without_a_limit() { ("sort_desc(http_requests)", false), ] { let qe = ok(q); - let QueryExpr::Sort { keys, child, .. } = &qe else { + let NonASAPOp::Sort { keys, child, .. } = qe.expect_non_asap() else { panic!("{q}: expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 1); assert_eq!(keys[0].ascending, ascending, "{q}"); // No Limit above the Sort — every series is preserved. - assert!(!matches!(&qe, QueryExpr::Limit { .. })); + assert!(!matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); // The value column is what it ranks on: descend to the scan. let (metric, _) = first_scan(child); assert_eq!(metric, "http_requests"); @@ -2233,12 +2135,12 @@ fn sort_by_label_orders_on_each_label_in_turn() { // `sort_by_label(v, "group", "instance", "job")` — one ascending sort key per // label, in argument order; the labels are seeded into the schema. let qe = ok(r#"sort_by_label(http_requests, "group", "instance", "job")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { panic!("expected a Sort, got {qe:?}"); }; assert_eq!(keys.len(), 3, "one key per label"); assert!(keys.iter().all(|k| k.ascending)); - let sch = qe.output_schema().unwrap(); + let sch = qe.schema.clone(); for label in ["group", "instance", "job"] { assert!(sch.fields.iter().any(|c| c.name == label), "{label} seeded"); } @@ -2247,7 +2149,7 @@ fn sort_by_label_orders_on_each_label_in_turn() { #[test] fn sort_by_label_desc_is_descending() { let qe = ok(r#"sort_by_label_desc(http_requests, "instance")"#); - let QueryExpr::Sort { keys, .. } = &qe else { + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { panic!("expected a Sort, got {qe:?}"); }; assert!(keys.iter().all(|k| !k.ascending)); @@ -2256,28 +2158,43 @@ fn sort_by_label_desc_is_descending() { #[test] fn min_of_max_of_fold_constant_scalars() { // `min_of`/`max_of` are n-ary scalar reducers. When every argument is a - // constant they constant-fold to a `PromqlScalarBridge` leaf, just like scalar + // constant they constant-fold to a `ScalarExpr` leaf, just like scalar // arithmetic (#35) — the only form the intent algebra can hold (#89). - assert_eq!(ok("min_of(3, 5)").as_promql_scalar(), Some(3.0)); - assert_eq!(ok("max_of(3, 5)").as_promql_scalar(), Some(5.0)); - assert_eq!(ok("min_of(-2, -5)").as_promql_scalar(), Some(-5.0)); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(3, 5)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, 5)")), + Some(5.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(-2, -5)")), + Some(-5.0) + ); // Nested folds and use as a threshold operand. assert_eq!( - ok("max_of(min_of(2, 3), 10)").as_promql_scalar(), + promql_scalar(&support::scalar_root("max_of(min_of(2, 3), 10)")), Some(10.0) ); let qe = ok("up > max_of(1, 2)"); - let QueryExpr::BinaryOp { rhs, .. } = &qe else { + let ScalarExpr::Compare { right: rhs, .. } = support::sample_expression(&qe) else { panic!("{qe:?}") }; - assert_eq!(rhs.as_promql_scalar(), Some(2.0)); + assert_eq!(promql_scalar(rhs), Some(2.0)); } #[test] fn min_of_max_of_ignore_nan_like_the_min_max_aggregators() { // A NaN argument is skipped (Prometheus `min`/`max` NaN semantics). - assert_eq!(ok("max_of(3, NaN)").as_promql_scalar(), Some(3.0)); - assert_eq!(ok("min_of(NaN, 3)").as_promql_scalar(), Some(3.0)); + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, NaN)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(NaN, 3)")), + Some(3.0) + ); } #[test] diff --git a/crates/frontend-promql/tests/promql_equivalence.rs b/crates/frontend-promql/tests/promql_equivalence.rs index 9d0cab2ad..ac3c5a06f 100644 --- a/crates/frontend-promql/tests/promql_equivalence.rs +++ b/crates/frontend-promql/tests/promql_equivalence.rs @@ -17,12 +17,14 @@ #![allow(non_snake_case)] +use std::rc::Rc; + mod support; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use support::lower_promql; -fn lo(q: &str) -> QueryExpr { +fn lo(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("{q:?} should lower: {e}")) } diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 243bf6d18..81f471660 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -1,10 +1,12 @@ //! End-to-end tests for PromQL → unresolved → canonical DAG lowering. +use std::rc::Rc; use std::time::Duration; -use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, QueryExpr, Reduction, ScalarValue, - Source, +use asap_types::ir::operator::{AggIntent, BinaryOpKind, Reduction, Source}; +use asap_types::ir::scalar::{ArithmeticOpKind, CompareOpKind, ScalarValue}; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, }; use asap_types::types::AccuracyTarget; use asap_types::workload::{ @@ -16,7 +18,7 @@ use asap_frontend_promql::{lower_promql_workload, PromqlError as LoweringError}; mod support; use support::lower_promql; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } @@ -77,12 +79,12 @@ fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { #[test] fn bare_selector_is_scan_with_predicates() { let qe = lower(r#"http_requests_total{env="prod",status!="500"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let NonASAPOp::Scan { source, predicates, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Scan, got {qe:?}"); }; @@ -92,29 +94,32 @@ fn bare_selector_is_scan_with_predicates() { assert_eq!(predicates.len(), 2); assert!(predicates .iter() - .all(|p| matches!(p.0.as_ref(), QueryExpr::Compare { .. }))); + .all(|p| matches!(&p.0, ScalarExpr::Compare { .. }))); } #[test] fn regex_matcher_lowers_to_regex_compareop() { let qe = lower(r#"http_requests_total{path=~"/api/.*"}"#); - let QueryExpr::TimeRange { child, .. } = &qe else { + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { panic!("expected TimeRange, got {qe:?}"); }; - let QueryExpr::Scan { + let NonASAPOp::Scan { predicates, schema, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Scan, got {qe:?}"); }; - let QueryExpr::Compare { left, op, right } = predicates[0].0.as_ref() else { + let ScalarExpr::Compare { + left, op, right, .. + } = &predicates[0].0 + else { panic!("expected Compare, got {:?}", predicates[0].0); }; assert_eq!(*op, CompareOpKind::Regex); // The label matcher's column is resolved positionally against the scan schema. let path_id = schema.column_id("path").expect("path in scan schema"); - assert!(matches!(left.as_ref(), QueryExpr::Column(id) if *id == path_id)); - assert!(matches!(right.as_ref(), QueryExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); + assert!(matches!(left.as_ref(), ScalarExpr::Column(id) if *id == path_id)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); } // ── *_over_time → Aggregate over TimeRange ────────────────────────────────────── @@ -122,12 +127,12 @@ fn regex_matcher_lowers_to_regex_compareop() { #[test] fn quantile_over_time_is_time_range_aggregate() { let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; @@ -135,12 +140,14 @@ fn quantile_over_time_is_time_range_aggregate() { assert!( matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.99).abs() < 1e-9) ); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); // The label matcher folded onto the Scan. - assert!(matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1)); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) + ); } #[test] @@ -151,39 +158,42 @@ fn outer_sum_by_over_quantile_over_time_groups_positionally() { // a name-based Partition. Leaf = [ts, value, host, service] (referenced // names appended sorted) → host = col 2. let qe = lower(r#"sum by (host) (quantile_over_time(0.99, latency{service="web"}[5m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by host, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); // Inner: Aggregate{Quantile} over TimeRange (per-series over_time reduction). - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (quantile_over_time) under the outer Sum, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Quantile { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] fn avg_over_time_maps_to_avg_intent() { let qe = lower("avg_over_time(cpu_seconds_total[10m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(600)); @@ -192,9 +202,9 @@ fn avg_over_time_maps_to_avg_intent() { #[test] fn stddev_and_stdvar_over_time() { let qe = lower("stddev_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; @@ -205,12 +215,15 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); let qe = lower("stdvar_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; @@ -221,7 +234,10 @@ fn stddev_and_stdvar_over_time() { .. }] )); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -230,32 +246,33 @@ fn histogram_quantile_wraps_inner_in_quantile() { // not squashed away. The `_bucket` metric + `le` matcher mark the classic // form → `HistogramQuantile` over `Aggregate{Rate}` over Scan. let qe = lower(r#"histogram_quantile(0.95, rate(http_duration_seconds_bucket{le="0.5"}[5m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.95).abs() < 1e-9) ); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { + let NonASAPOp::TimeRange { range, child: tr_child, - } = child.as_ref() + .. + } = child.expect_non_asap() else { panic!("expected TimeRange under Rate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); assert!( - matches!(tr_child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1) + matches!(tr_child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) ); } @@ -266,9 +283,9 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { // `sum by (le)` aggregate; now the `le` grouping survives into the // canonical DAG. let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); }; @@ -278,11 +295,11 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { ); // `sum by (le)` survives as a positional Aggregate (by = [2], `le`) over the // inner Rate — no name-based Partition. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected `sum by (le)` as a positional Aggregate, got {child:?}"); }; @@ -292,12 +309,12 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { /// The classic `histogram_quantile` aggregate: its `without` keys, `le` /// column, and output column names. -fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { - let QueryExpr::Aggregate { +fn classic_histogram(qe: &OperatorNode) -> (Vec, usize, Vec) { + let NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, .. - } = qe + } = qe.expect_non_asap() else { panic!("expected a reducing Aggregate, got {qe:?}"); }; @@ -305,13 +322,7 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { panic!("expected HistogramQuantile, got {measures:?}"); }; assert!(by.is_without(), "histogram_quantile groups without (le)"); - let names = qe - .output_schema() - .unwrap() - .fields - .iter() - .map(|c| c.name.clone()) - .collect(); + let names = qe.schema.fields.iter().map(|c| c.name.clone()).collect(); (by.keys().to_vec(), *le, names) } @@ -321,10 +332,10 @@ fn classic_histogram(qe: &QueryExpr) -> (Vec, usize, Vec) { fn classic_histogram_quantile_groups_without_le() { let qe = lower("histogram_quantile(0.9, rate(http_duration_seconds_bucket[5m]))"); let (keys, le, names) = classic_histogram(&qe); - let QueryExpr::Aggregate { child, .. } = &qe else { + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { unreachable!() }; - let child = child.output_schema().unwrap(); + let child = &child.schema; assert_eq!(child.fields[le].name, "le"); assert_eq!(keys, vec![le]); assert_eq!(names, vec!["histogram_quantile"]); @@ -349,14 +360,16 @@ fn classic_histogram_quantile_keeps_out_of_range_quantiles() { ("histogram_quantile(-1, x_bucket)", -1.), ("histogram_quantile(2, x_bucket)", 2.), ] { - let QueryExpr::Aggregate { measures, .. } = lower(query) else { + let root = lower(query); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { panic!("{query}"); }; assert!( matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if *q == expected) ); } - let QueryExpr::Aggregate { measures, .. } = lower("histogram_quantile(NaN, x_bucket)") else { + let root = lower("histogram_quantile(NaN, x_bucket)"); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { panic!("NaN"); }; assert!(matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if q.is_nan())); @@ -379,14 +392,14 @@ fn classic_histogram_quantile_rejects_an_argument_without_le() { #[test] fn rate_has_time_range_child_not_window() { let qe = lower("rate(http_requests_total[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate for rate, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child (not Window), got {child:?}"); }; assert_eq!(*range, Duration::from_secs(300)); @@ -395,14 +408,14 @@ fn rate_has_time_range_child_not_window() { #[test] fn increase_maps_to_increase_intent() { let qe = lower("increase(errors_total[1h])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate for increase, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Increase])); - let QueryExpr::TimeRange { range, .. } = child.as_ref() else { + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { panic!("expected TimeRange child, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(3600)); @@ -415,21 +428,24 @@ fn sum_over_rate_keeps_both_levels() { // Regression: `sum(rate(m[w]))` — the most common PromQL shape — must keep // the cross-series Sum, not collapse to a bare per-series Rate. let qe = lower("sum(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Rate}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Rate])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -438,20 +454,20 @@ fn sum_by_over_rate_groups_the_outer_sum() { // on a positional `Aggregate.by` (the same shape SQL produces) over the // label-preserving inner Rate. Leaf = [ts, value, job] → by = [2]. let qe = lower("sum by (job) (rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by job, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -459,16 +475,16 @@ fn sum_by_over_rate_groups_the_outer_sum() { fn count_over_rate_keeps_both_levels() { // The `Outer::Count` sibling of the `sum(rate(...))` bug. let qe = lower("count(rate(http_requests_total[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate{{Count}}, got {qe:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) )); } @@ -487,12 +503,12 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ), ] { let dag = lower(query); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, reduction: actual, child, .. - } = &dag + } = dag.expect_non_asap() else { panic!("expected outer Count: {dag:?}"); }; @@ -501,12 +517,12 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { "{query}: {dag:?}" ); assert_eq!(actual, &reduction, "{query}"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, reduction, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner per-series Cardinality: {dag:?}"); }; @@ -516,7 +532,7 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { ); assert_eq!(reduction, &Reduction::PerEntity, "{query}"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { range, .. } if range.as_secs() == 300) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, .. } if range.as_secs() == 300) ); } } @@ -552,14 +568,17 @@ fn count_never_lowers_to_distinct_sample_values() { #[test] fn count_over_time_is_count_intent() { let qe = lower("count_over_time(m[5m])"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] @@ -568,26 +587,29 @@ fn outer_count_counts_series() { // over the window (label-preserving), outer cross-series row count grouped // on a positional `Aggregate.by`. Leaf = [ts, value, symbol] → symbol = col 2. let qe = lower("count by (symbol) (count_over_time(financial_last_trade_price[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected outer Aggregate grouped by symbol, got {qe:?}"); }; assert_eq!(reduction, &Reduction::by(vec![2])); assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); // Inner: Aggregate{Count} over TimeRange (per-series count_over_time). - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (count_over_time) under the outer count, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ── topk / bottomk ──────────────────────────────────────────────────────────── @@ -596,12 +618,12 @@ fn outer_count_counts_series() { fn topk_over_count_is_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (10, count_over_time(requests{env="prod"}[1m]))"#); // Heavy-hitter: Aggregate{TopK} with grouping resolved to positional ids. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate with TopK, got {qe:?}"); }; @@ -612,29 +634,29 @@ fn topk_over_count_is_heavy_hitter_topk() { [AggIntent::TopK { k: 10, .. }] )); // The count_over_time under the TopK is a TimeRange-backed aggregate. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (count_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { panic!("expected TimeRange under Count aggregate, got {child:?}"); }; assert_eq!(*range, Duration::from_secs(60)); - assert!(matches!(child.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); } #[test] fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { let qe = lower(r#"topk by (service) (5, sum_over_time(requests{env="prod"}[1m]))"#); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate with TopK, got {qe:?}"); }; @@ -643,29 +665,38 @@ fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { measures.as_slice(), [AggIntent::TopK { k: 5, .. }] )); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Aggregate (sum_over_time) under TopK, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } #[test] fn topk_over_avg_is_generic_sort_limit() { let qe = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { n, offset, child } = &qe else { + let NonASAPOp::Limit { + n: Some(n), + offset, + child, + .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 5); assert_eq!(*offset, 0); - let QueryExpr::Sort { + let NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort under Limit, got {child:?}"); }; @@ -678,7 +709,7 @@ fn topk_over_avg_is_generic_sort_limit() { // Underneath: the label-preserving windowed avg aggregate (by: []), no // intervening Partition. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, measures, .. } + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, measures, .. } if reduction == &Reduction::PerEntity && matches!(measures.as_slice(), [AggIntent::Avg { .. }])), "expected bare per-series Avg aggregate under Sort, got {child:?}" ); @@ -687,7 +718,7 @@ fn topk_over_avg_is_generic_sort_limit() { #[test] fn ungrouped_topk_over_sum_is_heavy_hitter() { let qe = lower("topk(5, sum_over_time(m[5m]))"); - assert!(matches!(&qe, QueryExpr::Aggregate { .. })); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); assert!(has_intent(&qe, |i| matches!(i, AggIntent::Sum { .. }))); assert!(has_intent(&qe, |i| matches!( i, @@ -699,11 +730,14 @@ fn ungrouped_topk_over_sum_is_heavy_hitter() { fn bottomk_over_count_is_generic_sort_ascending() { // `bottomk` is never a heavy-hitter (descending=false), even over count. let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -715,11 +749,14 @@ fn bottomk_over_count_is_generic_sort_ascending() { #[test] fn bottomk_is_always_generic_sort_ascending() { let qe = lower("bottomk(3, count_over_time(m[5m]))"); - let QueryExpr::Limit { n, child, .. } = &qe else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { panic!("expected Limit, got {qe:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { keys, .. } = child.as_ref() else { + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { panic!("expected Sort"); }; assert!(keys[0].ascending, "bottomk ranks ascending"); @@ -731,12 +768,12 @@ fn topk_count_output_schema_carries_group_key() { // (`service`) flows through to the outer TopK's `by` column. Leaf schema = // [ts, value, service] → TopK groups on service (col 2). let qe = lower("topk by (service) (5, count_over_time(m[1m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected Aggregate{{TopK}}, got {qe:?}"); }; @@ -750,14 +787,17 @@ fn topk_count_output_schema_carries_group_key() { [AggIntent::TopK { k: 5, .. }] )); // Inner Count aggregate is visible with its TimeRange child. - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected inner Aggregate{{Count}}, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); - assert!(matches!(child.as_ref(), QueryExpr::TimeRange { .. })); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); } // ── binary ops ──────────────────────────────────────────────────────────────── @@ -765,26 +805,36 @@ fn topk_count_output_schema_carries_group_key() { #[test] fn binary_op_division() { let qe = lower("rate(a[5m]) / rate(b[5m])"); - let QueryExpr::BinaryOp { op, lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); assert!( - matches!(lhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); assert!( - matches!(rhs.as_ref(), QueryExpr::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); } #[test] fn binary_op_with_on_grouping() { let qe = lower("a / on(host) b"); - let QueryExpr::BinaryOp { vector_match, .. } = &qe else { + let NonASAPOp::BinaryOp { + operator: BinaryOperator { vector_match, .. }, + .. + } = qe.expect_non_asap() + else { panic!("expected BinaryOp, got {qe:?}"); }; let vm = vector_match.as_ref().expect("vector_match present"); - use asap_types::pre_asap::VectorMatchKind; + use asap_types::ir::operator::VectorMatchKind; assert_eq!(vm.kind, VectorMatchKind::On); assert_eq!(vm.labels, vec!["host".to_string()]); } @@ -793,18 +843,38 @@ fn binary_op_with_on_grouping() { // carry it. #[test] fn bool_comparisons_are_distinct() { - let op = |q: &str| match lower(q) { - QueryExpr::BinaryOp { op, .. } => op, + let op = |q: &str| match lower(q).expect_non_asap() { + NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } => (operator.kind.clone(), *return_bool), + NonASAPOp::Filter { + pred: asap_types::ir::Predicate(ScalarExpr::Compare { op, .. }), + .. + } => (BinaryOpKind::Compare(op.clone()), false), + NonASAPOp::Project { cols, .. } => { + let ScalarExpr::Case { branches, .. } = &cols[1].expr else { + panic!() + }; + let ScalarExpr::Compare { op, .. } = &branches[0].0 else { + panic!() + }; + (BinaryOpKind::Compare(op.clone()), true) + } other => panic!("expected BinaryOp, got {other:?}"), }; - assert_eq!(op("a > 1"), BinaryOpKind::Compare(CompareOpKind::Gt)); + assert_eq!( + op("a > 1"), + (BinaryOpKind::Compare(CompareOpKind::Gt), false) + ); assert_eq!( op("a > bool 1"), - BinaryOpKind::CompareBool(CompareOpKind::Gt) + (BinaryOpKind::Compare(CompareOpKind::Gt), true) ); assert_eq!( op("a == bool on(job) b"), - BinaryOpKind::CompareBool(CompareOpKind::Eq) + (BinaryOpKind::Compare(CompareOpKind::Eq), true) ); } @@ -814,7 +884,7 @@ fn binary_op_binds_each_branch_against_its_own_schema() { // single root schema threaded to both branches, the left scan would leak the // right's group key (and vice-versa). Per-branch binding keeps them separate. let qe = lower("count by (job) (a) / count by (region) (b)"); - let QueryExpr::BinaryOp { lhs, rhs, .. } = &qe else { + let NonASAPOp::BinaryOp { lhs, rhs, .. } = qe.expect_non_asap() else { panic!("expected BinaryOp, got {qe:?}"); }; let lcols = scan_columns(lhs); @@ -829,26 +899,26 @@ fn binary_op_binds_each_branch_against_its_own_schema() { ); } -/// Collect every `AggIntent` in the DAG, root-to-leaf. -fn all_intents(e: &QueryExpr) -> Vec { +/// Collect every `AggIntent` in the dag, root-to-leaf. +fn all_intents(e: &OperatorNode) -> Vec { let mut out = Vec::new(); collect_intents(e, &mut out); out } -fn collect_intents(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { +fn collect_intents(e: &OperatorNode, out: &mut Vec) { + match e.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => { out.extend(measures.iter().cloned()); collect_intents(child, out); } - QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => collect_intents(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } => { + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => collect_intents(child, out), + NonASAPOp::BinaryOp { lhs, rhs, .. } => { collect_intents(lhs, out); collect_intents(rhs, out); } @@ -856,20 +926,20 @@ fn collect_intents(e: &QueryExpr, out: &mut Vec) { } } -/// True if any `AggIntent` anywhere in the DAG satisfies `pred`. -fn has_intent bool>(e: &QueryExpr, pred: F) -> bool { +/// True if any `AggIntent` anywhere in the dag satisfies `pred`. +fn has_intent bool>(e: &OperatorNode, pred: F) -> bool { all_intents(e).iter().any(pred) } /// Field names on the first `Scan` reachable by descending single-child nodes. -fn scan_columns(e: &QueryExpr) -> Vec { - match e { - QueryExpr::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), - QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_columns(child), +fn scan_columns(e: &OperatorNode) -> Vec { + match e.expect_non_asap() { + NonASAPOp::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => scan_columns(child), _ => vec![], } } @@ -883,12 +953,12 @@ fn without_grouping_lowers_to_the_exclusion_form() { // label is stored positionally (the SchemaResolver seeds it), the grouping is the // `without` form, and the output schema stays open. let qe = lower("sum without (instance) (rate(m[5m]))"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = &qe + } = qe.expect_non_asap() else { panic!("expected an Aggregate, got {qe:?}"); }; @@ -899,10 +969,10 @@ fn without_grouping_lowers_to_the_exclusion_form() { // The inner per-series rate is preserved (label-preserving) under the outer // cross-series `without` reduction. assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) ); - assert!(!qe.output_schema().unwrap().closed); + assert!(!qe.schema.clone().closed); } // ── parameter validation (reject rather than silently truncate/garble) ────────── @@ -920,7 +990,7 @@ fn out_of_range_quantile_phi_is_accepted() { for query in [ "quantile(1.5, up)", "quantile_over_time(1.5, m[5m])", - "histogram_quantile(2.0, rate(b[5m]))", + "histogram_quantile(2.0, rate(b_bucket[5m]))", ] { assert!( lower_promql(query, AccuracyTarget::Exact).is_ok(), @@ -979,7 +1049,7 @@ fn accuracy_target_flows_into_quantile_intent() { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let QueryExpr::Aggregate { measures, .. } = &qe else { + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { panic!("expected Aggregate"); }; assert!(matches!( @@ -998,10 +1068,10 @@ fn aggregate_output_schema_preserves_time_axis_and_labels() { // predicate columns) to the scan schema, so `env` appears as a column // even though it is only used as a filter. // per_series_reduction_schema preserves the time axis and all label columns. - let QueryExpr::Aggregate { .. } = &qe else { + let NonASAPOp::Aggregate { .. } = qe.expect_non_asap() else { panic!("expected Aggregate, got {qe:?}"); }; - let schema = qe.output_schema().expect("aggregate schema"); + let schema = &qe.schema; let names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); assert_eq!(names, vec!["ts", "value", "env"]); assert_eq!( @@ -1016,16 +1086,16 @@ fn scan_schema_carries_ts_value_and_group_keys() { // `service` is a group key → the SchemaResolver lands it in the self-contained // Scan schema (positional). `env` is only a filter, so it is not a column. let qe = lower("count by (service) (count_over_time(requests[1m]))"); - fn find_scan(n: &QueryExpr) -> &QueryExpr { - match n { - QueryExpr::Scan { .. } => n, - QueryExpr::TimeRange { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Filter { child, .. } => find_scan(child), + fn find_scan(n: &OperatorNode) -> &OperatorNode { + match n.expect_non_asap() { + NonASAPOp::Scan { .. } => n, + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } => find_scan(child), other => panic!("unexpected node {other:?}"), } } - let QueryExpr::Scan { schema, .. } = find_scan(&qe) else { + let NonASAPOp::Scan { schema, .. } = find_scan(&qe).expect_non_asap() else { unreachable!() }; let mut names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); @@ -1113,19 +1183,23 @@ fn reducing_group_by_lowers_to_aggregate_by() { // Cross-series reduce, no keys → bare `Aggregate { reduction: Reduce([]) }`. let q = lower("sum(http_requests_total)"); assert!( - matches!(q, QueryExpr::Aggregate { ref reduction, .. } if reduction == &Reduction::by(vec![])) + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::by(vec![])) ); // Cross-series reduce grouped by a label → `Aggregate.reduction`. let q = lower("sum by (job) (http_requests_total)"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } - if reduction.expect_reduce().len() == 1)); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); // Reduce over a label-preserving `rate` grouped by a label → still // `Aggregate.reduction` (the keys resolve against rate's preserved schema). let q = lower("sum by (job) (rate(http_requests_total[5m]))"); - assert!(matches!(q, QueryExpr::Aggregate { ref reduction, .. } - if reduction.expect_reduce().len() == 1)); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); } #[test] @@ -1134,20 +1208,20 @@ fn generic_topk_grouping_lowers_to_sort_partition_by() { // reducing → the grouping rides on `Sort.partition_by`, and the windowed // reduction beneath stays label-preserving (`by: []`). No `Partition` node. let q = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); - let QueryExpr::Limit { child, .. } = &q else { + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let NonASAPOp::Sort { partition_by, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; assert_eq!(partition_by, &vec![2], "host is col 2 in [ts, value, host]"); assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) ); } @@ -1160,15 +1234,18 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // Partition→Sort.partition_by reframe in #12). Expected: // Limit{3} → Sort{value desc, partition_by:[job]} → Scan let q = lower("topk(3, http_requests_total) by (job)"); - let QueryExpr::Limit { n, child, .. } = &q else { + let NonASAPOp::Limit { + n: Some(n), child, .. + } = q.expect_non_asap() + else { panic!("expected Limit, got {q:?}"); }; assert_eq!(*n, 3); - let QueryExpr::Sort { + let NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; @@ -1177,7 +1254,7 @@ fn topk_over_bare_selector_by_label_ranks_per_group() { // No implicit reducing aggregate — the selector is label-preserving, so the // sort is directly over the selector horizon (the `job` label survives to partition by). assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })), + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })), "ranking is over the bare selector horizon, not a reducing Aggregate, got {child:?}" ); assert!( @@ -1191,20 +1268,20 @@ fn topk_over_bare_selector_ranks_raw_samples() { // Even without `by`, `topk(3, m)` ranks the raw instant-vector samples — it // does not sum them. The sort sits directly over the Scan, partition empty. let q = lower("topk(3, http_requests_total)"); - let QueryExpr::Limit { child, .. } = &q else { + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { panic!("expected Limit, got {q:?}"); }; - let QueryExpr::Sort { + let NonASAPOp::Sort { partition_by, child, .. - } = child.as_ref() + } = child.expect_non_asap() else { panic!("expected Sort, got {child:?}"); }; assert!(partition_by.is_empty(), "no `by` → global ranking"); assert!( - matches!(child.as_ref(), QueryExpr::TimeRange { child, .. } if matches!(child.as_ref(), QueryExpr::Scan { .. })) + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) ); assert!(!has_intent(&q, |i| matches!(i, AggIntent::Sum { .. }))); } @@ -1212,20 +1289,20 @@ fn topk_over_bare_selector_ranks_raw_samples() { // ── Issue #109: histogram_quantiles fans out into one branch per φ ────────── /// The `(label value, intent)` of each `histogram_quantiles` branch. -fn quantile_branches(q: &QueryExpr) -> Vec<(String, AggIntent)> { - let QueryExpr::Concat { children, .. } = q else { +fn quantile_branches(q: &OperatorNode) -> Vec<(String, AggIntent)> { + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected a Concat at the root, got {q:?}"); }; children .iter() .map(|c| { - let QueryExpr::PromqlRelabel { value, child, .. } = c else { + let NonASAPOp::PromqlRelabel { value, child, .. } = c.expect_non_asap() else { panic!("expected PromqlRelabel per branch, got {c:?}"); }; - let QueryExpr::Literal(ScalarValue::Utf8(v)) = value.as_ref() else { + let ScalarExpr::Literal(ScalarValue::Utf8(v)) = value else { panic!("expected a literal label value, got {value:?}"); }; - let QueryExpr::Aggregate { measures, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { measures, .. } = child.expect_non_asap() else { panic!("expected an Aggregate under the PromqlRelabel, got {child:?}"); }; (v.clone(), measures[0].clone()) @@ -1234,24 +1311,12 @@ fn quantile_branches(q: &QueryExpr) -> Vec<(String, AggIntent)> { } #[test] -fn histogram_quantiles_fans_out_over_native_histograms() { - // Raw / native-histogram argument → the sketch-able `Quantile` intent, - // exactly as the single-quantile `histogram_quantile` would choose. - let q = lower(r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#); - let branches = quantile_branches(&q); - assert_eq!(branches.len(), 3); - let labels: Vec<_> = branches.iter().map(|(l, _)| l.as_str()).collect(); - assert_eq!( - labels, - ["0.0", "0.25", "1.0"], - "OpenMetrics float formatting" - ); - for (_, intent) in &branches { - assert!( - matches!(intent, AggIntent::Quantile { .. }), - "native histogram → sketch-able Quantile, got {intent:?}" - ); - } +fn histogram_quantiles_rejects_unrepresented_native_histograms() { + assert!(lower_promql( + r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#, + AccuracyTarget::Exact + ) + .is_err()); } #[test] @@ -1270,15 +1335,15 @@ fn histogram_quantiles_over_classic_buckets_interpolates() { fn histogram_quantiles_branches_are_union_compatible() { // `Concat` derives its schema from the first child, so every branch must // agree on column names — the φ lives in the label, not the column name. - let q = lower(r#"histogram_quantiles(testhistogram3, "q", 0.5, 0.9)"#); - let QueryExpr::Concat { children, .. } = &q else { + let q = lower(r#"histogram_quantiles(testhistogram3_bucket, "q", 0.5, 0.9)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected Concat"); }; let shapes: Vec> = children .iter() .map(|c| { - c.output_schema() - .expect("branch schema") + c.schema + .clone() .fields .iter() .map(|c| c.name.clone()) @@ -1288,7 +1353,7 @@ fn histogram_quantiles_branches_are_union_compatible() { assert_eq!(shapes[0], shapes[1], "branches must be union-compatible"); assert_eq!(shapes[0], vec!["value".to_string(), "q".to_string()]); assert_eq!( - q.output_schema().expect("merged schema").fields.len(), + q.schema.fields.len(), 2, "the merged schema describes every branch" ); @@ -1296,11 +1361,11 @@ fn histogram_quantiles_branches_are_union_compatible() { #[test] fn histogram_quantiles_uses_the_given_label_name() { - let q = lower(r#"histogram_quantiles(h, "phi", 0.5)"#); - let QueryExpr::Concat { children, .. } = &q else { + let q = lower(r#"histogram_quantiles(h_bucket, "phi", 0.5)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { panic!("expected Concat"); }; - let QueryExpr::PromqlRelabel { dst, .. } = &children[0] else { + let NonASAPOp::PromqlRelabel { dst, .. } = children[0].expect_non_asap() else { panic!("expected PromqlRelabel"); }; assert_eq!(dst, "phi"); @@ -1309,7 +1374,7 @@ fn histogram_quantiles_uses_the_given_label_name() { #[test] fn histogram_quantiles_formats_small_quantiles_like_prometheus() { // `labels.FormatOpenMetricsFloat`: Go's %g, so exponent form below 1e-4. - let q = lower(r#"histogram_quantiles(h, "q", 0.00001)"#); + let q = lower(r#"histogram_quantiles(h_bucket, "q", 0.00001)"#); assert_eq!(quantile_branches(&q)[0].0, "1e-05"); } @@ -1317,9 +1382,9 @@ fn histogram_quantiles_formats_small_quantiles_like_prometheus() { fn histogram_quantiles_rejects_an_out_of_range_quantile() { // Same rule as `histogram_quantile(φ, …)` — one bad φ fails the whole call. for q in [ - r#"histogram_quantiles(h, "q", -0.1)"#, - r#"histogram_quantiles(h, "q", 1.01)"#, - r#"histogram_quantiles(h, "q", 0.5, NaN)"#, + r#"histogram_quantiles(h_bucket, "q", -0.1)"#, + r#"histogram_quantiles(h_bucket, "q", 1.01)"#, + r#"histogram_quantiles(h_bucket, "q", 0.5, NaN)"#, ] { assert!( lower_promql(q, AccuracyTarget::Exact).is_err(), @@ -1328,19 +1393,244 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { } } -// A subquery's `offset`/`@` shift the whole subquery, so the DAG keeps them. +// ── TimeRange.kind: instant vs range selectors ────────────────────────────────── + +#[test] +fn bare_instant_selector_is_an_instant_time_range() { + // `up` reads the latest sample per series within the workload's ingestion + // interval (1s in `support::workload`): an `Instant` lookback of that length. + let qe = lower("up"); + let NonASAPOp::TimeRange { range, kind, child } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + assert_eq!(*kind, TimeRangeKind::Instant); + assert_eq!(*range, Duration::from_secs(1)); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); +} + +#[test] +fn explicit_range_selector_is_a_range_time_range() { + // `m[5m]` keeps its own window and is a `Range` selection — both under a + // range function and as a bare matrix selector. + let qe = lower("rate(m[5m])"); + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { + panic!("expected Aggregate, got {qe:?}"); + }; + let NonASAPOp::TimeRange { range, kind, .. } = child.expect_non_asap() else { + panic!("expected TimeRange, got {child:?}"); + }; + assert_eq!(*kind, TimeRangeKind::Range); + assert_eq!(*range, Duration::from_secs(300)); + + let qe = lower("m[5m]"); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); +} + +#[test] +fn instant_and_range_selectors_of_equal_length_stay_distinct() { + // The kind is part of the shape: a 1s range selector is not the same dag as + // the 1s instant lookback injected around a bare selector. + assert_ne!(lower("up"), lower("up[1s]")); +} + +// ── the `bool` modifier → `return_bool` ───────────────────────────────────────── + +#[test] +fn vector_scalar_comparison_without_bool_filters() { + let qe = lower("up > 0"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. })); + assert!(matches!( + support::sample_expression(&qe), + ScalarExpr::Compare { + op: CompareOpKind::Gt, + .. + } + )); +} + +#[test] +fn vector_scalar_comparison_with_bool_sets_return_bool() { + let qe = lower("up > bool 0"); + assert!(matches!( + support::sample_expression(&qe), + ScalarExpr::Case { .. } + )); + assert_ne!(qe, lower("up > 0")); +} + +#[test] +fn vector_vector_comparison_with_bool_sets_return_bool() { + // `a > bool b` — the modifier lands on the vector/vector op itself, with + // the default (ignoring nothing) match. + let qe = lower("a > bool b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + lhs, + rhs, + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + assert_eq!(operator.kind, BinaryOpKind::Compare(CompareOpKind::Gt)); + assert!(matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(!lower("a > b").expect_non_asap().children().is_empty()); + assert_ne!(qe, lower("a > b")); +} + +#[test] +fn bool_modifier_composes_with_vector_matching() { + let qe = lower("a > bool on(job) b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + let vm = operator.vector_match.as_ref().expect("on(job) present"); + assert_eq!(vm.labels, vec!["job".to_string()]); +} + +// ── scalar expressions: negation, arithmetic, comparison ──────────────────────── + +#[test] +fn scalar_negation_of_time_is_a_negative_expression() { + // `-time()` is a scalar expression; its negation stays structural (the + // operand is not a constant to fold) and follows PromQL numeric rules. + let qe = support::scalar_root("-time()"); + let ScalarExpr::Negative { expr, semantics } = &qe else { + panic!("expected ScalarExpr(Negative), got {qe:?}"); + }; + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(expr.as_ref(), ScalarExpr::EvalTimestamp)); + // Scalar-shaped: no time index. +} + +#[test] +fn scalar_negation_of_a_constant_still_folds() { + // `-(2)` is constant: it folds to one literal rather than a `Negative`. + assert_eq!( + support::promql_scalar(&support::scalar_root("-(2)")), + Some(-2.0) + ); +} + +#[test] +fn scalar_arithmetic_carries_promql_semantics() { + let qe = support::scalar_root("time() - 1"); + let ScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } = &qe + else { + panic!("expected scalar(Arithmetic), got {qe:?}"); + }; + assert_eq!(*op, ArithmeticOpKind::Sub); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::EvalTimestamp)); + assert!(matches!( + right.as_ref(), + ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0 + )); +} + +#[test] +fn scalar_bool_comparison_is_a_zero_one_case_with_promql_semantics() { + // `1 < bool 2` → `Case(Compare(1 < 2) → 1.0, else 0.0)`: PromQL yields 0/1. + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { + operand, + branches, + else_expr, + } = &qe + else { + panic!("expected scalar(Case), got {qe:?}"); + }; + assert!(operand.is_none()); + let [(when, then)] = branches.as_slice() else { + panic!("expected one branch, got {branches:?}"); + }; + let ScalarExpr::Compare { + left, + op, + right, + semantics, + } = when + else { + panic!("expected a Compare condition, got {when:?}"); + }; + assert_eq!(*op, CompareOpKind::Lt); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 2.0)); + assert!(matches!(then, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!( + else_expr.as_deref(), + Some(ScalarExpr::Literal(ScalarValue::Float64(v))) if *v == 0.0 + )); +} + +#[test] +fn scalar_comparison_without_bool_is_rejected() { + // PromQL has no scalar filter: a scalar/scalar comparison needs `bool`. + for q in ["1 < 2", "time() > 0", "(1 + 1) == 2"] { + assert!( + lower_promql(q, AccuracyTarget::Exact).is_err(), + "{q} must be rejected without `bool`" + ); + } +} + +#[test] +fn label_matcher_predicates_carry_promql_semantics() { + let qe = lower(r#"up{job="api"}"#); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + let NonASAPOp::Scan { predicates, .. } = child.expect_non_asap() else { + panic!("expected Scan, got {child:?}"); + }; + assert!(matches!( + &predicates[0].0, + ScalarExpr::Compare { + semantics: ExprSemantics::Promql, + .. + } + )); +} + +// A subquery's `offset` / `@` modifier stays a `TimeShift` over the subquery. #[test] fn subquery_time_shift_is_retained() { - let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { - panic!("expected a range function"); + let root = lower("max_over_time(m[5m:1m] offset 1m)"); + let NonASAPOp::Aggregate { child, .. } = root.expect_non_asap() else { + panic!("expected a range function, got {root:?}"); }; - let QueryExpr::TimeShift { shift, child } = child.as_ref() else { + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { panic!("subquery offset was dropped: {child:?}"); }; assert_eq!(shift.offset_ms, 60_000); - assert!(matches!(child.as_ref(), QueryExpr::PromqlSubquery { .. })); assert!(matches!( - lower("max_over_time(m[5m:1m] @ 100)"), - QueryExpr::Aggregate { child, .. } if matches!(child.as_ref(), QueryExpr::TimeShift { .. }) + child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); + let root = lower("max_over_time(m[5m:1m] @ 100)"); + assert!(matches!( + root.expect_non_asap(), + NonASAPOp::Aggregate { child, .. } + if matches!(child.expect_non_asap(), NonASAPOp::TimeShift { .. }) )); } diff --git a/crates/frontend-promql/tests/scalar_design.rs b/crates/frontend-promql/tests/scalar_design.rs new file mode 100644 index 000000000..ec853a1f7 --- /dev/null +++ b/crates/frontend-promql/tests/scalar_design.rs @@ -0,0 +1,128 @@ +//! Scalar expressions never become constant-wrapper operators. +mod support; +use asap_types::ir::scalar::{ArithmeticOpKind, ScalarValue}; +use asap_types::ir::{NonASAPOp, QueryRoot, ScalarExpr}; +use asap_types::types::AccuracyTarget; + +fn root(query: &str) -> QueryRoot { + asap_frontend_promql::lower_promql_query_workload( + &support::workload(query, AccuracyTarget::Exact), + 0, + ) + .unwrap() + .remove(0) +} + +#[test] +fn standalone_scalars_are_expressions() { + for query in [ + "2", + "time()", + "scalar(sum(up)) + 1", + "1 < bool 2", + "-time()", + ] { + let QueryRoot::Scalar(expr) = root(query) else { + panic!("{query} became an operator") + }; + expr.scalar_type(&Default::default()).unwrap(); + } +} + +#[test] +fn arithmetic_projects_the_sample_and_preserves_full_identity_and_time() { + let QueryRoot::Operator(node) = root("up * 2") else { + panic!() + }; + let NonASAPOp::Project { child, cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert!(child.schema.has_promql_series_identity()); + assert!(node.schema.has_promql_series_identity()); + assert_eq!(node.schema.time_index, child.schema.time_index); + let value = node.schema.column_id("value").unwrap(); + assert!( + matches!(&cols[value].expr, ScalarExpr::Arithmetic { op: ArithmeticOpKind::Mul, right, .. } if **right == ScalarExpr::Literal(ScalarValue::Float64(2.0))) + ); + assert!(cols.iter().any(|c| matches!(&c.expr, ScalarExpr::FunctionCall { name, .. } if name == "promql_drop_metric_name"))); +} + +#[test] +fn non_bool_comparisons_keep_vector_samples_even_with_scalar_on_left() { + for query in ["up > 0", "0 < up"] { + let QueryRoot::Operator(node) = root(query) else { + panic!() + }; + let NonASAPOp::Filter { child, .. } = node.expect_non_asap() else { + panic!() + }; + assert_eq!(node.schema, child.schema); + } +} + +#[test] +fn bool_comparison_projects_zero_or_one() { + let QueryRoot::Operator(node) = root("up > bool 0") else { + panic!() + }; + let NonASAPOp::Project { cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert!(matches!( + cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::Case { .. } + )); +} + +#[test] +fn scalar_plan_dependencies_remain_visible() { + let QueryRoot::Operator(node) = root("up * scalar(sum(up))") else { + panic!() + }; + assert_eq!(node.children().len(), 2); +} + +/// Pointwise functions own scalar parameters, including vector-to-scalar reads. +#[test] +fn pointwise_functions_are_typed_scalar_projections() { + for query in [ + "abs(up)", + "round(up, scalar(sum(other)))", + "clamp(up, time() - 1, time())", + "year(up)", + "hour()", + ] { + let QueryRoot::Operator(node) = root(query) else { + panic!() + }; + let NonASAPOp::Project { cols, .. } = node.expect_non_asap() else { + panic!("{query}: expected projection") + }; + assert!(matches!( + &cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::FunctionCall { .. } + )); + node.validate_structure().unwrap(); + } +} + +/// Negation preserves the metric name and complete identity unlike multiplication. +#[test] +fn unary_minus_preserves_identity() { + let QueryRoot::Operator(node) = root("-up") else { + panic!() + }; + let NonASAPOp::Project { child, cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert_eq!(node.schema, child.schema); + assert!(matches!( + &cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::Negative { .. } + )); + for (index, col) in cols.iter().enumerate() { + if index != node.schema.column_id("value").unwrap() { + assert_eq!(col.expr, ScalarExpr::Column(index)); + } + } +} diff --git a/crates/frontend-promql/tests/support.rs b/crates/frontend-promql/tests/support.rs index 1f15b1ca2..1123c6c35 100644 --- a/crates/frontend-promql/tests/support.rs +++ b/crates/frontend-promql/tests/support.rs @@ -1,14 +1,17 @@ +use std::rc::Rc; + use asap_frontend_promql::{ lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, }; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::scalar::ScalarValue; +use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; -fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { +pub fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -35,7 +38,11 @@ fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { } } -pub fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Result { +#[allow(dead_code)] +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, PromqlError> { let mut lowered = lower_promql_workload(&workload(query, accuracy), 0)?; Ok(lowered.remove(0)) } @@ -45,8 +52,59 @@ pub fn lower_promql_with_histograms( query: &str, accuracy: AccuracyTarget, histograms: HistogramCatalog, -) -> Result { +) -> Result, PromqlError> { let mut lowered = lower_promql_workload_with_histograms(&workload(query, accuracy), histograms, 0)?; Ok(lowered.remove(0)) } + +/// The value of a bare PromQL numeric literal / folded constant at an +/// scalar position (`Literal(Float64(v))`); `None` for any +/// other shape. +#[allow(dead_code)] +pub fn promql_scalar(node: &ScalarExpr) -> Option { + match node { + ScalarExpr::Literal(ScalarValue::Float64(v)) => Some(*v), + _ => None, + } +} + +/// Time `root` under the default materialization assignment (every summary +/// at query time) and export the physical DAG — export needs every node +/// timed first. +#[allow(dead_code)] +pub fn post_asap_dag(root: &Rc) -> asap_types::ir::export::PhysicalASAPDAG { + use asap_types::ir::{apply_materialization_timings, MaterializationAssignment, TimingMemo}; + let timed = apply_materialization_timings( + root, + &MaterializationAssignment::all_query_time(), + &mut TimingMemo::new(), + ) + .expect("default materialization timings"); + asap_types::ir::export::compile_physical_asap_dag(&timed).expect("post-ASAP DAG export") +} + +#[allow(dead_code)] +pub fn scalar_root(query: &str) -> ScalarExpr { + match asap_frontend_promql::lower_promql_query_workload( + &workload(query, AccuracyTarget::Exact), + 0, + ) + .unwrap() + .remove(0) + { + asap_types::ir::QueryRoot::Scalar(expr) => expr, + _ => panic!("expected scalar root: {query}"), + } +} + +#[allow(dead_code)] +pub fn sample_expression(node: &OperatorNode) -> &ScalarExpr { + match node.expect_non_asap() { + NonASAPOp::Project { cols, .. } => { + &cols[node.schema.column_id("value").unwrap_or(cols.len() - 1)].expr + } + NonASAPOp::Filter { pred, .. } => &pred.0, + other => panic!("expected sample expression, got {other:?}"), + } +} diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index 2b4891a08..3fbf8afb0 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -1,19 +1,23 @@ use std::rc::Rc; -use asap_aware_mapping::accuracy::{ +use asap_logical_optimizer::accuracy::{ AccuracyModel, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, }; -use asap_aware_mapping::cost_model::DefaultCostModel; -use asap_aware_mapping::replacement::{default_strategies, search_workload_with_targets}; -use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; +use asap_logical_optimizer::pass1::replacement::{ + default_strategies, search_workload_with_targets, +}; +use asap_logical_optimizer::{ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG}; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::cost::cost_model::DefaultCostModel; mod support; -use asap_types::post_asap::{ - compile_post_asap_dag, cse::share_common_summary_sub_dags, AccuracyError, BoundExpr, - CompositionOperator, ErrorMetric, FieldDataType, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SketchStatistic, SummaryExpr, SummaryInputExpr, SummaryNode, +use asap_types::ir::cse::share_common_sub_dags; +use asap_types::ir::properties::{ + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, }; +use asap_types::ir::schema::{FieldDataType, SketchAlgorithm, SketchStatistic, SummaryInputExpr}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; use asap_types::types::AccuracyTarget; -use support::lower_promql; +use support::{lower_promql, post_asap_dag}; // Synthetic evidence exercises structural sharing, never runtime accuracy. struct TestEvidence; @@ -49,22 +53,22 @@ impl AccuracyModel for TestEvidence { } } -fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { +fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { let root = lower_promql(query, accuracy).unwrap(); - SketchAlgorithmStrategy::new_with_planning_inputs(&DefaultCostModel, &TestEvidence, &EqualSplitAllocator) - .replacements(&TargetSubDAG::new(&Rc::new(root))) + ASAPStrategies::new_with_planning_inputs(&TestEvidence, &EqualSplitAllocator) + .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| { - let Replacement::Summary(node) = candidate.replacement else { return None }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { return None }; - matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + let Replacement::SubDAG(node) = candidate.replacement else { return None }; + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator else { return None }; + matches!(&summary_input.operator, Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::UnivMon).then_some(node) }).expect("UnivMon candidate") } #[test] -fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { - // Equal data, grouping and window produce one state independently of readout. +fn four_evaluations_share_one_value_frequency_state_and_keep_honest_guarantees() { + // Equal data, grouping and window produce one state independently of evaluation. let accuracy = AccuracyTarget::Epsilon(0.02); let roots: Vec<_> = [ ("distinct_over_time(m[5m])", accuracy.clone()), @@ -76,26 +80,25 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { .enumerate() .map(|(id, (query, accuracy))| (id, candidate(query, accuracy))) .collect(); - let roots = share_common_summary_sub_dags(roots); + let roots = share_common_sub_dags(roots); let mut first_state = None; for (index, root) in &roots { - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - .. - } = &root.expr + }) = &root.operator else { panic!() }; if let Some(first) = &first_state { assert!( Rc::ptr_eq(first, summary_input), - "state must be shared across readouts" + "state must be shared across evaluations" ); } else { first_state = Some(Rc::clone(summary_input)); } - let SummaryExpr::SummaryAgg { input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &summary_input.operator else { panic!() }; assert!(matches!(input.item, Some(SummaryInputExpr::Column(_)))); @@ -108,7 +111,7 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { assert!(root.guarantee.as_ref().is_some_and(|g| g.is_exact())); } else { assert!(!root.guarantee.as_ref().unwrap().is_exact()); - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { panic!() }; assert!( @@ -118,12 +121,12 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { "production has no calibrated error bound" ); } - compile_post_asap_dag(root).unwrap(); + post_asap_dag(root); } } #[test] -fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { +fn uncalibrated_frequency_evaluations_do_not_bypass_accuracy_targets() { // An unmeasured heuristic remains inspectable but is never certified or // automatically selected for a caller-visible bounded-error result. for query in ["entropy_over_time(m[5m])", "l2_over_time(m[5m])"] { @@ -135,16 +138,15 @@ fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { delta: 0.01, }, ] { - let root = Rc::new(lower_promql(query, target.clone()).unwrap()); - let candidates = SketchAlgorithmStrategy::default_cost_model() - .replacements(&TargetSubDAG::new(&root)); + let root = lower_promql(query, target.clone()).unwrap(); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); let unknown = candidates .iter() .filter(|candidate| { matches!( &candidate.replacement, - Replacement::Summary(node) - if matches!(&node.expr, SummaryExpr::SummaryEstimate { .. }) + Replacement::SubDAG(node) + if matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) && node.guarantee.is_none() && candidate.has_missing_accuracy_evidence() ) @@ -165,8 +167,7 @@ fn uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets() { .candidates .iter() .any(|candidate| candidate.has_missing_accuracy_evidence())); - assert!(!space - .global_selection(&DefaultCostModel) + assert!(!global_selection(&space, &DefaultCostModel) .for_target(&space.roots[0].1) .unwrap() .chosen diff --git a/crates/frontend-sql/Cargo.toml b/crates/frontend-sql/Cargo.toml index ac6bd2f51..acfbca15d 100644 --- a/crates/frontend-sql/Cargo.toml +++ b/crates/frontend-sql/Cargo.toml @@ -9,6 +9,7 @@ edition = "2021" # #225) it consults when lowering an aggregate call — never promql-parser. [dependencies] asap-types = { path = "../types" } +asap-frontend-common = { path = "../frontend-common" } asap-sql-function-catalog = { path = "../sql-function-catalog" } datafusion = "43" # `AggIntent::Extension.payload` for ClickHouse's argMax/argMin (issue #232) @@ -16,7 +17,7 @@ datafusion = "43" serde_json = "1" [dev-dependencies] -asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-logical-optimizer = { path = "../logical-optimizer" } tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } # bgp_jan2024_workload corpus is sourced verbatim as YAML (ASAPQuery PR #561) # rather than transcribed into the flat .sql shape the other corpora use. diff --git a/crates/frontend-sql/src/error.rs b/crates/frontend-sql/src/error.rs index 04117352d..f819cfd46 100644 --- a/crates/frontend-sql/src/error.rs +++ b/crates/frontend-sql/src/error.rs @@ -1,11 +1,11 @@ use std::fmt; -use asap_types::pre_asap::ResolveDAGError; +use asap_frontend_common::ResolveDAGError; /// Errors from lowering a SQL query (parse + plan via DataFusion → the -/// canonical, unresolved DAG, built directly → -/// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved DAG, issue #179). +/// name-based [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree → +/// [`resolve_root`](asap_frontend_common::resolve_root) binds it into the +/// unified IR). /// /// Carries no PromQL type — the SQL front end never depends on the PromQL /// parser. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` @@ -28,8 +28,8 @@ pub enum SqlError { UnsupportedFeature(String), /// The workload's query language is not SQL. WrongLanguage(String), - /// Resolving the canonical unresolved DAG failed (name resolution - /// against the bound schema). + /// Resolving the name-based tree failed (name resolution against the + /// bound schema, or schema derivation). Convert(ResolveDAGError), } diff --git a/crates/frontend-sql/src/lib.rs b/crates/frontend-sql/src/lib.rs index 4ec61e69d..4d9e6ea1f 100644 --- a/crates/frontend-sql/src/lib.rs +++ b/crates/frontend-sql/src/lib.rs @@ -1,25 +1,27 @@ -//! SQL front end: parse + plan (via DataFusion) → the canonical, unresolved -//! shape, built directly (issue #179) → [`resolve_root`]. +//! SQL front end: parse + plan (via DataFusion) → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree, built directly +//! (issue #179) → [`resolve_root`]. //! -//! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the -//! canonical `QueryExpr`, generic over an unresolved -//! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational DAG; `resolve_root` runs the -//! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. +//! Emits the shared front-end tree (`UnresolvedOp` / `UnresolvedScalar`, +//! name-based [`ColumnRef`](asap_types::ir::scalar::ColumnRef)s) directly, rather +//! than a separate per-language relational tree; `resolve_root` binds it into +//! the unified [`OperatorNode`] IR, deriving every schema on the way. //! Depends on DataFusion only — never on the PromQL parser. pub mod error; pub mod sql; -use asap_types::pre_asap::resolve_root; -use asap_types::pre_asap::QueryExpr; +use std::rc::Rc; + +use asap_frontend_common::resolve_root; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{QueryLanguage, QueryWorkload, SqlDialect}; pub use error::SqlError; pub use sql::{SqlCatalog, SqlLowerer}; -/// Lower a single SQL query string to the canonical, resolved `QueryExpr`, +/// Lower a single SQL query string to the resolved, canonical operator DAG, /// parsed as `SqlDialect::DataFusionSQL`. /// /// The `catalog` supplies table schemas (used both to plan the SQL with @@ -29,7 +31,7 @@ pub async fn lower_sql( query: &str, catalog: &SqlCatalog, accuracy: AccuracyTarget, -) -> Result { +) -> Result, SqlError> { lower_sql_dialect(query, catalog, SqlDialect::DataFusionSQL, accuracy).await } @@ -46,20 +48,17 @@ pub async fn lower_sql_dialect( catalog: &SqlCatalog, dialect: SqlDialect, accuracy: AccuracyTarget, -) -> Result { +) -> Result, SqlError> { let unresolved = SqlLowerer::with_dialect(catalog, dialect) .lower(query, &accuracy) .await?; - let resolved = resolve_root(&unresolved)?; - // Binding resolves names; schema inference also checks result types such - // as temporal subtraction, whose duration unit the IR cannot represent. - resolved - .output_schema() - .map_err(|error| SqlError::InvalidExpression(error.to_string()))?; - Ok(resolved) + // Binding resolves names and derives every node's schema; result-type + // checks (such as temporal subtraction, whose duration unit the IR cannot + // represent) surface here as `ResolveDAGError::Schema`. + Ok(resolve_root(&unresolved)?) } -/// Lower every SQL batch entry in `workload` to a `QueryExpr`. +/// Lower every SQL batch entry in `workload` to an operator DAG. /// /// One `Result` per entry — errors are per-query, not fatal for the batch. /// Returns `WrongLanguage` for every entry if the workload is not SQL, and @@ -67,7 +66,7 @@ pub async fn lower_sql_dialect( pub async fn lower_sql_batch( workload: &QueryWorkload, catalog: &SqlCatalog, -) -> Vec> { +) -> Vec, SqlError>> { let entries = match &workload.query_batch { Some(e) if !e.is_empty() => e, _ => return vec![], diff --git a/crates/frontend-sql/src/sql/collection_planning.rs b/crates/frontend-sql/src/sql/collection_planning.rs index dd7106564..506f2d096 100644 --- a/crates/frontend-sql/src/sql/collection_planning.rs +++ b/crates/frontend-sql/src/sql/collection_planning.rs @@ -1,10 +1,10 @@ //! DataFusion planning adapters. Types come from the canonical signature rules; //! physical evaluation deliberately remains the query engine's responsibility. use super::types::{arrow_to_dtype, dtype_to_arrow, scalar_value_to_asap}; -use asap_types::pre_asap::scalar_type_rules::{ - element_access_type, struct_field_type, MapScalarFunction, -}; -use asap_types::pre_asap::{Field, QueryExpr, Schema}; +use asap_types::ir::scalar::scalar_type_rules::MapScalarFunction; +use asap_types::ir::scalar::{element_access_type, struct_field_type}; +use asap_types::ir::schema::{Field, Schema}; +use asap_types::ir::ScalarExpr; use datafusion::arrow::datatypes::DataType; use datafusion::common::{DataFusionError, ExprSchema, Result}; use datafusion::logical_expr::{ @@ -91,10 +91,10 @@ impl CollectionPlanningFunction { if let Some(Expr::Literal(value)) = expressions.and_then(|args| args.get(index)) { scalar_value_to_asap(value) - .map(QueryExpr::Literal) + .map(ScalarExpr::Literal) .map_err(|error| DataFusionError::Plan(error.to_string())) } else { - Ok(QueryExpr::Column(index)) + Ok(ScalarExpr::Column(index)) } }) .collect::>>()?; diff --git a/crates/frontend-sql/src/sql/expr.rs b/crates/frontend-sql/src/sql/expr.rs index 06b682062..c030dfdc7 100644 --- a/crates/frontend-sql/src/sql/expr.rs +++ b/crates/frontend-sql/src/sql/expr.rs @@ -2,12 +2,14 @@ use std::rc::Rc; use datafusion::logical_expr::{BinaryExpr, Expr, Operator}; -use asap_types::pre_asap::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; +use asap_frontend_common::UnresolvedScalar as Unresolved; +use asap_types::ir::scalar::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; +use asap_types::ir::ExprSemantics; use crate::error::SqlError as LoweringError; use super::types::{arrow_to_dtype, scalar_value_to_asap}; -use super::Unresolved; +use super::SqlLowerer; pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { match expr { @@ -24,255 +26,262 @@ pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { } } -/// Translate a DataFusion `Expr` to the canonical, unresolved DAG. -/// Returns `UnsupportedFeature` for anything not needed in v1. -pub(super) fn df_expr_to_unresolved(expr: &Expr) -> Result { - match expr { - // Preserve DataFusion's relation qualifier so a column name shared - // across a join (`a.k` vs `b.k`) resolves to the correct side. - Expr::Column(col) => Ok(Unresolved::Column(match &col.relation { - Some(rel) => ColumnRef::Qualified { - table: rel.to_string(), - name: col.name.clone(), - }, - None => ColumnRef::Named(col.name.clone()), - })), - - // Keep Arrow date literals equivalent to SQL CAST('YYYY-MM-DD' AS DATE), - // including typed nulls, without adding another canonical scalar variant. - Expr::Literal( - sv @ (datafusion::common::ScalarValue::Date32(_) - | datafusion::common::ScalarValue::Date64(_)), - ) => { - let text = sv.cast_to(&datafusion::arrow::datatypes::DataType::Utf8)?; - // Arrow formats Date64 with a time suffix; the canonical Date has - // no time-of-day, just like Date64 catalog registration as Date32. - let text = match text { - datafusion::common::ScalarValue::Utf8(Some(value)) => { - ScalarValue::Utf8(value.split('T').next().unwrap().to_owned()) - } - other => scalar_value_to_asap(&other)?, - }; - Ok(Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(text)), - to: asap_types::pre_asap::schema::DataType::Date, - try_cast: false, - }) - } - Expr::Literal(sv) => scalar_value_to_asap(sv).map(Unresolved::Literal), - - Expr::Alias(a) => df_expr_to_unresolved(&a.expr), +impl SqlLowerer<'_> { + /// Translate a DataFusion `Expr` to the name-based scalar tree. Every + /// `Compare` / `Arithmetic` / `Negative` carries `ExprSemantics::Sql`. + /// Subquery-valued expressions lower their plan as a root of its own + /// (which is why this is a method: the plan walk needs the catalog). + /// Returns `UnsupportedFeature` for anything not needed in v1. + pub(super) fn lower_expr(&self, expr: &Expr) -> Result { + let bx = |e: &Expr| self.lower_expr(e).map(Box::new); + match expr { + // Preserve DataFusion's relation qualifier so a column name shared + // across a join (`a.k` vs `b.k`) resolves to the correct side. + Expr::Column(col) => Ok(Unresolved::Column(match &col.relation { + Some(rel) => ColumnRef::Qualified { + table: rel.to_string(), + name: col.name.clone(), + }, + None => ColumnRef::Named(col.name.clone()), + })), - Expr::BinaryExpr(BinaryExpr { left, op, right }) => match op { - Operator::And => { - let parts = split_conjuncts(expr); - let lowered: Result, _> = - parts.iter().map(|e| df_expr_to_unresolved(e)).collect(); - Ok(Unresolved::BoolAnd(lowered?)) - } - Operator::Or => { - let parts = split_disjuncts(expr); - let lowered: Result, _> = - parts.iter().map(|e| df_expr_to_unresolved(e)).collect(); - Ok(Unresolved::BoolOr(lowered?)) + // Keep Arrow date literals equivalent to SQL CAST('YYYY-MM-DD' AS DATE), + // including typed nulls, without adding another canonical scalar variant. + Expr::Literal( + sv @ (datafusion::common::ScalarValue::Date32(_) + | datafusion::common::ScalarValue::Date64(_)), + ) => { + let text = sv.cast_to(&datafusion::arrow::datatypes::DataType::Utf8)?; + // Arrow formats Date64 with a time suffix; the canonical Date has + // no time-of-day, just like Date64 catalog registration as Date32. + let text = match text { + datafusion::common::ScalarValue::Utf8(Some(value)) => { + ScalarValue::Utf8(value.split('T').next().unwrap().to_owned()) + } + other => scalar_value_to_asap(&other)?, + }; + Ok(Unresolved::Cast { + expr: Box::new(Unresolved::Literal(text)), + to: asap_types::ir::schema::DataType::Date, + try_cast: false, + }) } - Operator::Eq => compare(left, CompareOpKind::Eq, right), - Operator::NotEq => compare(left, CompareOpKind::Ne, right), - Operator::Lt => compare(left, CompareOpKind::Lt, right), - Operator::LtEq => compare(left, CompareOpKind::Le, right), - Operator::Gt => compare(left, CompareOpKind::Gt, right), - Operator::GtEq => compare(left, CompareOpKind::Ge, right), - // BinaryExpr LIKE/ILIKE operators (from optimizer rewrites) - Operator::LikeMatch => compare(left, CompareOpKind::Like, right), - Operator::ILikeMatch => compare(left, CompareOpKind::ILike, right), - Operator::NotLikeMatch => compare(left, CompareOpKind::NotLike, right), - Operator::NotILikeMatch => compare(left, CompareOpKind::NotILike, right), - // Arithmetic - Operator::Plus => arith(left, ArithmeticOpKind::Add, right), - Operator::Minus => arith(left, ArithmeticOpKind::Sub, right), - Operator::Multiply => arith(left, ArithmeticOpKind::Mul, right), - Operator::Divide => arith(left, ArithmeticOpKind::Div, right), - Operator::Modulo => arith(left, ArithmeticOpKind::Mod, right), - other => Err(LoweringError::UnsupportedFeature(format!( - "operator: {other:?}" - ))), - }, + Expr::Literal(sv) => scalar_value_to_asap(sv).map(Unresolved::Literal), - // SQL LIKE / ILIKE (dedicated expr node from the SQL parser) - Expr::Like(like) => { - let op = match (like.negated, like.case_insensitive) { - (false, false) => CompareOpKind::Like, - (true, false) => CompareOpKind::NotLike, - (false, true) => CompareOpKind::ILike, - (true, true) => CompareOpKind::NotILike, - }; - compare(&like.expr, op, &like.pattern) - } + Expr::Alias(a) => self.lower_expr(&a.expr), - // Unary minus: negate literals directly; wrap others in -1 * x. - Expr::Negative(inner) => { - let inner = df_expr_to_unresolved(inner)?; - match inner { - Unresolved::Literal(ScalarValue::Int64(v)) => { - Ok(Unresolved::Literal(ScalarValue::Int64(-v))) + Expr::BinaryExpr(BinaryExpr { left, op, right }) => match op { + Operator::And => { + let parts = split_conjuncts(expr); + let lowered: Result, _> = + parts.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::BoolAnd(lowered?)) } - Unresolved::Literal(ScalarValue::Float64(v)) => { - Ok(Unresolved::Literal(ScalarValue::Float64(-v))) + Operator::Or => { + let parts = split_disjuncts(expr); + let lowered: Result, _> = + parts.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::BoolOr(lowered?)) } - other => Ok(Unresolved::Arithmetic { - op: ArithmeticOpKind::Mul, - left: Rc::new(Unresolved::Literal(ScalarValue::Int64(-1))), - right: Rc::new(other), - }), + Operator::Eq => self.compare(left, CompareOpKind::Eq, right), + Operator::NotEq => self.compare(left, CompareOpKind::Ne, right), + Operator::Lt => self.compare(left, CompareOpKind::Lt, right), + Operator::LtEq => self.compare(left, CompareOpKind::Le, right), + Operator::Gt => self.compare(left, CompareOpKind::Gt, right), + Operator::GtEq => self.compare(left, CompareOpKind::Ge, right), + // BinaryExpr LIKE/ILIKE operators (from optimizer rewrites) + Operator::LikeMatch => self.compare(left, CompareOpKind::Like, right), + Operator::ILikeMatch => self.compare(left, CompareOpKind::ILike, right), + Operator::NotLikeMatch => self.compare(left, CompareOpKind::NotLike, right), + Operator::NotILikeMatch => self.compare(left, CompareOpKind::NotILike, right), + // Arithmetic + Operator::Plus => self.arith(left, ArithmeticOpKind::Add, right), + Operator::Minus => self.arith(left, ArithmeticOpKind::Sub, right), + Operator::Multiply => self.arith(left, ArithmeticOpKind::Mul, right), + Operator::Divide => self.arith(left, ArithmeticOpKind::Div, right), + Operator::Modulo => self.arith(left, ArithmeticOpKind::Mod, right), + other => Err(LoweringError::UnsupportedFeature(format!( + "operator: {other:?}" + ))), + }, + + // SQL LIKE / ILIKE (dedicated expr node from the SQL parser) + Expr::Like(like) => { + let op = match (like.negated, like.case_insensitive) { + (false, false) => CompareOpKind::Like, + (true, false) => CompareOpKind::NotLike, + (false, true) => CompareOpKind::ILike, + (true, true) => CompareOpKind::NotILike, + }; + self.compare(&like.expr, op, &like.pattern) } - } - // SQL CASE expression - Expr::Case(c) => { - let operand = c - .expr - .as_ref() - .map(|e| df_expr_to_unresolved(e).map(Rc::new)) - .transpose()?; - let branches = c - .when_then_expr - .iter() - .map(|(when, then)| { - Ok((df_expr_to_unresolved(when)?, df_expr_to_unresolved(then)?)) + // Unary minus. (DataFusion's planner already folds `-` + // into a negative literal, so this is a non-literal operand.) + Expr::Negative(inner) => Ok(Unresolved::Negative { + expr: bx(inner)?, + semantics: ExprSemantics::Sql, + }), + + // SQL CASE expression + Expr::Case(c) => { + let operand = c.expr.as_deref().map(bx).transpose()?; + let branches = c + .when_then_expr + .iter() + .map(|(when, then)| Ok((self.lower_expr(when)?, self.lower_expr(then)?))) + .collect::, LoweringError>>()?; + let else_expr = c.else_expr.as_deref().map(bx).transpose()?; + Ok(Unresolved::Case { + operand, + branches, + else_expr, }) - .collect::, LoweringError>>()?; - let else_expr = c - .else_expr - .as_ref() - .map(|e| df_expr_to_unresolved(e).map(Rc::new)) - .transpose()?; - Ok(Unresolved::Case { - operand, - branches, - else_expr, - }) - } + } - Expr::Not(inner) => Ok(Unresolved::Not(Rc::new(df_expr_to_unresolved(inner)?))), + Expr::Not(inner) => Ok(Unresolved::Not(bx(inner)?)), - Expr::IsNull(inner) => Ok(Unresolved::IsNull(Rc::new(df_expr_to_unresolved(inner)?))), + Expr::IsNull(inner) => Ok(Unresolved::IsNull(bx(inner)?)), - Expr::IsNotNull(inner) => Ok(Unresolved::IsNotNull(Rc::new(df_expr_to_unresolved( - inner, - )?))), + Expr::IsNotNull(inner) => Ok(Unresolved::IsNotNull(bx(inner)?)), - Expr::Cast(c) => { - let inner = df_expr_to_unresolved(&c.expr)?; - let to = arrow_to_dtype(&c.data_type)?; - Ok(Unresolved::Cast { - expr: Rc::new(inner), - to, + Expr::Cast(c) => Ok(Unresolved::Cast { + expr: bx(&c.expr)?, + to: arrow_to_dtype(&c.data_type)?, try_cast: false, - }) - } + }), - // TRY_CAST returns NULL on conversion failure; preserve that semantic. - Expr::TryCast(c) => { - let inner = df_expr_to_unresolved(&c.expr)?; - let to = arrow_to_dtype(&c.data_type)?; - Ok(Unresolved::Cast { - expr: Rc::new(inner), - to, + // TRY_CAST returns NULL on conversion failure; preserve that semantic. + Expr::TryCast(c) => Ok(Unresolved::Cast { + expr: bx(&c.expr)?, + to: arrow_to_dtype(&c.data_type)?, try_cast: true, - }) - } + }), - Expr::InList(il) => { - let expr = df_expr_to_unresolved(&il.expr)?; - let list: Result, _> = il.list.iter().map(df_expr_to_unresolved).collect(); - Ok(Unresolved::InList { - expr: Rc::new(expr), - list: list?, - negated: il.negated, - }) - } + Expr::InList(il) => { + let list: Result, _> = il.list.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::InList { + expr: bx(&il.expr)?, + list: list?, + negated: il.negated, + }) + } - Expr::Between(b) => { - // Normalize: `x BETWEEN low AND high` → `x >= low AND x <= high`. - // `x NOT BETWEEN low AND high` → `x < low OR x > high`. - let x_low = compare(&b.expr, CompareOpKind::Ge, &b.low)?; - let x_high = compare(&b.expr, CompareOpKind::Le, &b.high)?; - if b.negated { - // NOT BETWEEN: invert each side - let lt = compare(&b.expr, CompareOpKind::Lt, &b.low)?; - let gt = compare(&b.expr, CompareOpKind::Gt, &b.high)?; - Ok(Unresolved::BoolOr(vec![lt, gt])) - } else { - Ok(Unresolved::BoolAnd(vec![x_low, x_high])) + Expr::Between(b) => { + // Normalize: `x BETWEEN low AND high` → `x >= low AND x <= high`. + // `x NOT BETWEEN low AND high` → `x < low OR x > high`. + if b.negated { + let lt = self.compare(&b.expr, CompareOpKind::Lt, &b.low)?; + let gt = self.compare(&b.expr, CompareOpKind::Gt, &b.high)?; + Ok(Unresolved::BoolOr(vec![lt, gt])) + } else { + let x_low = self.compare(&b.expr, CompareOpKind::Ge, &b.low)?; + let x_high = self.compare(&b.expr, CompareOpKind::Le, &b.high)?; + Ok(Unresolved::BoolAnd(vec![x_low, x_high])) + } } - } - // `NOW()` / `CURRENT_TIMESTAMP` read the SQL statement evaluation - // time. Keep this timestamp-typed leaf distinct from PromQL's - // Float64 Unix-seconds `EvalTimestamp`. Issue #184. - Expr::ScalarFunction(sf) - if sf.args.is_empty() - && matches!( - sf.func.name().to_ascii_lowercase().as_str(), - "now" | "current_timestamp" - ) => - { - Ok(Unresolved::CurrentTimestamp) - } + // `NOW()` / `CURRENT_TIMESTAMP` read the SQL statement evaluation + // time. Keep this timestamp-typed leaf distinct from PromQL's + // Float64 Unix-seconds `EvalTimestamp`. Issue #184. + Expr::ScalarFunction(sf) + if sf.args.is_empty() + && matches!( + sf.func.name().to_ascii_lowercase().as_str(), + "now" | "current_timestamp" + ) => + { + Ok(Unresolved::CurrentTimestamp) + } - Expr::ScalarFunction(sf) => { - let args: Result, _> = sf.args.iter().map(df_expr_to_unresolved).collect(); - Ok(Unresolved::FunctionCall { - name: if sf.func.name().eq_ignore_ascii_case("arrayelement") { - "asap_element_access".into() - } else if sf.func.name().eq_ignore_ascii_case("tupleelement") { - "asap_struct_field".into() - } else { - sf.func.name().to_string() - }, - args: args?, - }) - } + Expr::ScalarFunction(sf) => { + let args: Result, _> = sf.args.iter().map(|e| self.lower_expr(e)).collect(); + Ok(Unresolved::FunctionCall { + name: if sf.func.name().eq_ignore_ascii_case("arrayelement") { + "asap_element_access".into() + } else if sf.func.name().eq_ignore_ascii_case("tupleelement") { + "asap_struct_field".into() + } else { + sf.func.name().to_string() + }, + args: args?, + }) + } + + // Subquery-valued expressions. Each subquery plan is lowered as a + // root of its own; `resolve_root` binds it in its own scope, so an + // outer reference inside it has nothing to resolve against — a + // correlated subquery is rejected rather than mislowered. + Expr::ScalarSubquery(sq) => Ok(Unresolved::ScalarSubquery(Rc::new( + self.lower_uncorrelated_subquery(sq, "scalar subquery")?, + ))), + Expr::Exists(ex) => Ok(Unresolved::Exists { + subquery: Rc::new(self.lower_uncorrelated_subquery(&ex.subquery, "EXISTS")?), + negated: ex.negated, + }), + Expr::InSubquery(is) => { + let fields = is.subquery.subquery.schema().fields().len(); + if fields != 1 { + return Err(LoweringError::InvalidExpression(format!( + "IN (subquery) must select exactly one column, got {fields}" + ))); + } + Ok(Unresolved::InSubquery { + expr: bx(&is.expr)?, + subquery: Rc::new( + self.lower_uncorrelated_subquery(&is.subquery, "IN (subquery)")?, + ), + negated: is.negated, + }) + } - // Subquery-valued expressions in a predicate/projection — `x > (SELECT - // …)`, `x IN (SELECT …)`, `EXISTS (SELECT …)`. These need a subquery - // node in the unresolved expression IR (and a correlated-vs-uncorrelated - // decision); rejected cleanly until that lands rather than mislowered. - // Derived tables in `FROM` (the common nesting shape) ARE supported — - // see `lower_plan`'s `SubqueryAlias` arm. - Expr::ScalarSubquery(_) | Expr::InSubquery(_) | Expr::Exists(_) => Err( - LoweringError::UnsupportedFeature("subquery-valued expression in predicate".into()), - ), + other => Err(LoweringError::UnsupportedFeature(format!( + "expression: {}", + other + ))), + } + } - other => Err(LoweringError::UnsupportedFeature(format!( - "expression: {}", - other - ))), + fn lower_uncorrelated_subquery( + &self, + sq: &datafusion::logical_expr::Subquery, + what: &str, + ) -> Result { + if !sq.outer_ref_columns.is_empty() { + return Err(LoweringError::UnsupportedFeature(format!( + "correlated {what}" + ))); + } + self.lower_plan(&sq.subquery) } -} -pub(super) fn compare( - left: &Expr, - op: CompareOpKind, - right: &Expr, -) -> Result { - Ok(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(left)?), - op, - right: Rc::new(df_expr_to_unresolved(right)?), - }) -} + pub(super) fn compare( + &self, + left: &Expr, + op: CompareOpKind, + right: &Expr, + ) -> Result { + Ok(Unresolved::Compare { + left: Box::new(self.lower_expr(left)?), + op, + right: Box::new(self.lower_expr(right)?), + semantics: ExprSemantics::Sql, + }) + } -pub(super) fn arith( - left: &Expr, - op: ArithmeticOpKind, - right: &Expr, -) -> Result { - Ok(Unresolved::Arithmetic { - op, - left: Rc::new(df_expr_to_unresolved(left)?), - right: Rc::new(df_expr_to_unresolved(right)?), - }) + fn arith( + &self, + left: &Expr, + op: ArithmeticOpKind, + right: &Expr, + ) -> Result { + Ok(Unresolved::Arithmetic { + op, + left: Box::new(self.lower_expr(left)?), + right: Box::new(self.lower_expr(right)?), + semantics: ExprSemantics::Sql, + }) + } } pub(super) fn split_disjuncts(expr: &Expr) -> Vec<&Expr> { @@ -293,12 +302,15 @@ pub(super) fn split_disjuncts(expr: &Expr) -> Vec<&Expr> { #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::schema::DataType; + use crate::sql::SqlCatalog; + use asap_types::ir::schema::DataType; use datafusion::common::ScalarValue as DfScalarValue; // Typed Arrow dates normalize to the same typed form as SQL date casts. #[test] fn arrow_date_literals_preserve_value_and_type() { + let catalog = SqlCatalog::new(); + let lowerer = SqlLowerer::new(&catalog); for (value, expected) in [ ( DfScalarValue::Date32(Some(0)), @@ -311,15 +323,32 @@ mod tests { (DfScalarValue::Date32(None), ScalarValue::Null), (DfScalarValue::Date64(None), ScalarValue::Null), ] { - let actual = df_expr_to_unresolved(&Expr::Literal(value)).unwrap(); + let actual = lowerer.lower_expr(&Expr::Literal(value)).unwrap(); assert_eq!( actual, Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(expected)), + expr: Box::new(Unresolved::Literal(expected)), to: DataType::Date, try_cast: false, } ); } } + + // Unary minus over a non-literal is the `Negative` scalar, SQL-flavoured. + #[test] + fn unary_minus_lowers_to_negative_with_sql_semantics() { + let catalog = SqlCatalog::new(); + let lowerer = SqlLowerer::new(&catalog); + let expr = Expr::Negative(Box::new(Expr::Column( + datafusion::common::Column::new_unqualified("x"), + ))); + assert_eq!( + lowerer.lower_expr(&expr).unwrap(), + Unresolved::Negative { + expr: Box::new(Unresolved::Column(ColumnRef::Named("x".into()))), + semantics: ExprSemantics::Sql, + } + ); + } } diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index d1401988c..b5f4eb0e6 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -1,12 +1,12 @@ -//! SQL → the canonical, unresolved -//! [`UnresolvedQueryExpr`](asap_types::pre_asap::query_expr::UnresolvedQueryExpr) -//! (`QueryExpr`). +//! SQL → the name-based front-end tree +//! ([`UnresolvedOp`](asap_frontend_common::UnresolvedOp) / +//! [`UnresolvedScalar`](asap_frontend_common::UnresolvedScalar)). //! //! Parses SQL via DataFusion (over the catalog's registered tables), then -//! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with -//! unresolved `ColumnRef`s directly (issue #179) — the same DAG shape -//! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, -//! positional `QueryExpr`. Unlike PromQL's front end, SQL's +//! walks the unoptimized `LogicalPlan` and emits `UnresolvedOp` nodes with +//! unresolved `ColumnRef`s directly (issue #179) — the same tree shape +//! [`resolve_root`](asap_frontend_common::resolve_root) binds into the +//! positional, unified `OperatorNode` IR. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit //! `asap_rate`/`asap_increase` bridge is the narrow exception: it //! spells a time-series range reducer with an explicit value, time-index, and @@ -51,18 +51,21 @@ use datafusion::optimizer::{AnalyzerRule, OptimizerConfig}; use datafusion::prelude::{SessionConfig, SessionContext}; use datafusion::sql::parser::DFParser; +use asap_frontend_common::{ + resolve_root, UnresolvedOp as Unresolved, UnresolvedPredicate as Predicate, + UnresolvedProjectItem as ProjectItem, UnresolvedScalar as Scalar, UnresolvedSortKey as SortKey, +}; use asap_sql_function_catalog::{AggSemantic, Arity, RewriteKind}; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{ - GroupKeys, Predicate, ProjectItem, Reduction, SortKey, Source, - UnresolvedQueryExpr as Unresolved, WindowFrame, WindowFrameBound, WindowFrameOffset, +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::{ + GroupKeys, Reduction, Source, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, }; -use asap_types::pre_asap::schema::{DataType, FieldDataType, Schema}; -use asap_types::pre_asap::{ - resolve_column_ref, resolve_root, ColumnRef, CompareOpKind, JoinKind, RelationalSetOpKind, - ScalarValue, WindowFuncKind, -}; +use asap_types::ir::schema::{DataType, FieldDataType, Schema}; +use asap_types::ir::TimeRangeKind; + +use asap_types::ir::operator::{JoinKind, RelationalSetOpKind, WindowFuncKind}; +use asap_types::ir::scalar::{resolve_column_ref, ColumnRef, CompareOpKind, ScalarValue}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -77,7 +80,6 @@ mod types; pub use types::SqlCatalog; use self::dialect::GenericWithAggregateFilter; -use self::expr::df_expr_to_unresolved; use self::types::{arrow_to_dtype, scalar_value_to_asap, schema_to_arrow}; std::thread_local! { @@ -111,10 +113,10 @@ fn current_accuracy() -> AccuracyTarget { ACCURACY.with(|a| a.borrow().clone()) } -/// Lowers SQL strings to the canonical [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) -/// over a table [`SqlCatalog`]. Call -/// [`resolve_root`](asap_types::pre_asap::resolve_root) on the result for -/// the canonical, resolved DAG. +/// Lowers SQL strings to the name-based [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) +/// tree over a table [`SqlCatalog`]. Call +/// [`resolve_root`](asap_frontend_common::resolve_root) on the result for +/// the resolved operator DAG. pub struct SqlLowerer<'a> { catalog: &'a SqlCatalog, dialect: SqlDialect, @@ -140,7 +142,7 @@ impl<'a> SqlLowerer<'a> { Self { catalog, dialect } } - /// Parse + lower a SQL query to the canonical, unresolved shape, threading + /// Parse + lower a SQL query to the name-based tree, threading /// `accuracy` onto every approximate intent (`Count`, `Quantile`, /// `Cardinality`) as it is built. /// @@ -167,8 +169,8 @@ impl<'a> SqlLowerer<'a> { /// a rule) that isn't wanted here — e.g. it independently rejects a /// multi-column `IN (subquery)` before `lower_in_subquery`'s own arity /// check would. Going straight to `ApplyFunctionRewrites` avoids that - /// entirely: zero behavior change for every query that doesn't call a - /// catalog-listed ClickHouse builtin. + /// entirely. TypeCoercion then records implicit conversions explicitly, + /// including timestamp literals in predicates, before IR validation. pub async fn lower( &self, sql: &str, @@ -197,6 +199,8 @@ impl<'a> SqlLowerer<'a> { let plan = state.statement_to_plan(statement).await?; let rewriter = ApplyFunctionRewrites::new(vec![Arc::new(ClickHouseBuiltinRewrite)]); let plan = rewriter.analyze(plan, ctx.state().options())?; + let plan = datafusion::optimizer::analyzer::type_coercion::TypeCoercion::new() + .analyze(plan, ctx.state().options())?; // Output schemas omit predicate and nested-expression types. Check the // typed SQL plan before lowering erases fixed-duration units. plan.apply_with_subqueries(|node| { @@ -269,8 +273,8 @@ impl<'a> SqlLowerer<'a> { // *scalar* builtin — same reason as the `AggregateUDF` loop above // (DataFusion otherwise rejects the call as an unknown function // during `SqlToRel` conversion), but with no rewrite step to follow: - // `df_expr_to_unresolved`'s `Expr::ScalarFunction` arm already lowers - // any scalar call generically to `Unresolved::FunctionCall { name, + // `lower_expr`'s `Expr::ScalarFunction` arm already lowers any + // scalar call generically to `UnresolvedScalar::FunctionCall { name, // args }`, so registering the stub is the entire fix (issue #230). for builtin in asap_sql_function_catalog::CLICKHOUSE_SCALAR_BUILTINS { ctx.register_udf(clickhouse_scalar_builtin_stub_udf( @@ -302,9 +306,24 @@ impl<'a> SqlLowerer<'a> { Ok(ctx) } - fn lower_plan(&self, plan: &LogicalPlan) -> Result { + pub(super) fn lower_plan(&self, plan: &LogicalPlan) -> Result { match plan { LogicalPlan::TableScan(scan) => self.lower_table_scan(scan), + // The one empty input row of a `SELECT` without `FROM`. + LogicalPlan::EmptyRelation(empty) => Ok(Unresolved::Values { + rows: if empty.produce_one_row { + vec![vec![]] + } else { + vec![] + }, + schema: Schema { + fields: vec![], + time_index: None, + unique_keys: vec![], + closed: true, + }, + }), + LogicalPlan::Values(values) => self.lower_values(values), LogicalPlan::Filter(filter) => self.lower_filter(filter), LogicalPlan::Projection(proj) => self.lower_projection(proj), LogicalPlan::Aggregate(agg) => self.lower_aggregate(agg), @@ -373,9 +392,7 @@ impl<'a> SqlLowerer<'a> { .iter() .map(|f| ProjectItem { alias: Some(f.name().clone()), - expr: Unresolved::Column(ColumnRef::Named( - f.name().clone(), - )), + expr: Scalar::Column(ColumnRef::Named(f.name().clone())), }) .collect(); Ok(Unresolved::Project { @@ -398,108 +415,48 @@ impl<'a> SqlLowerer<'a> { /// `WHERE` — a conjunction of ordinary predicates plus, possibly, subquery /// predicates (issue #111). /// - /// `c IN (SELECT …)` and `EXISTS (…)` are not expressions over rows; they are - /// *joins*. Each such conjunct peels off into a semi- / anti-join above the - /// filter's input, and the remaining conjuncts stay as an ordinary `Filter`. + /// The ordinary conjuncts stay one predicate, folded onto a bare `Scan` + /// (`filter_or_fold`). A subquery conjunct — `c IN (SELECT …)`, `EXISTS + /// (…)`, `x > (SELECT …)` — is a row filter whose predicate reads another + /// operator (`UnresolvedScalar::InSubquery` / `Exists` / + /// `ScalarSubquery`); each one becomes its own `Filter` **above** the + /// ordinary predicate, so the shared `canonicalize` pass can turn it into + /// the join it is without having to peel it out of a conjunction or off + /// a `Scan` (it only lifts subqueries out of `Filter` / `Project`). A + /// semi-join only ever drops left rows, so the two orders agree. /// - /// The residual filter is applied **below** the joins, which is where it sat - /// before: a semi-join only ever drops left rows, so the two orders agree — - /// and keeping the fold-onto-`Scan` (`filter_or_fold`) below the joins - /// matches where the old converter folded it too. + /// The one subquery shape still lowered to a join here is a *correlated* + /// `EXISTS`: its correlation references both sides, which only a join + /// predicate can bind (a subquery referenced from a scalar position is + /// resolved as a root in its own scope). fn lower_filter(&self, filter: &logical_expr::Filter) -> Result { let mut conjuncts = Vec::new(); split_conjunction(&filter.predicate, &mut conjuncts); - let (subqueries, residual): (Vec<_>, Vec<_>) = conjuncts - .into_iter() - .partition(|e| matches!(e, Expr::InSubquery(_) | Expr::Exists(_))); + let (subqueries, residual): (Vec<_>, Vec<_>) = + conjuncts.into_iter().partition(|e| reads_subquery(e)); let input = self.lower_plan(&filter.input)?; let mut node = match rebuild_conjunction(&residual) { - Some(pred) => filter_or_fold(df_expr_to_unresolved(&pred)?, input), + Some(pred) => filter_or_fold(self.lower_expr(&pred)?, input), None => input, }; for sq in subqueries { node = match sq { - Expr::InSubquery(is) => self.lower_in_subquery(is, node)?, - Expr::Exists(ex) => self.lower_exists(ex, node)?, - _ => unreachable!("partitioned above"), + Expr::Exists(ex) if !ex.subquery.outer_ref_columns.is_empty() => { + self.lower_correlated_exists(ex, node)? + } + other => Unresolved::Filter { + pred: Predicate(self.lower_expr(other)?), + child: Rc::new(node), + }, }; } Ok(node) } - /// `c IN (SELECT k FROM …)` → a semi-join on `c = k` (issue #111). - fn lower_in_subquery( - &self, - is: &logical_expr::expr::InSubquery, - left: Unresolved, - ) -> Result { - if is.negated { - // `NOT IN` is not an anti-join. Under three-valued logic a single - // NULL among the subquery's rows makes `c NOT IN (…)` UNKNOWN for - // every `c`, so the query returns nothing — while an anti-join - // returns every unmatched left row. Reject rather than mislower. - return Err(LoweringError::UnsupportedFeature( - "NOT IN (subquery): its NULL semantics are not an anti-join".into(), - )); - } - if !is.subquery.outer_ref_columns.is_empty() { - return Err(LoweringError::UnsupportedFeature( - "correlated IN (subquery)".into(), - )); - } - let inner = is.subquery.subquery.as_ref(); - let fields = inner.schema().fields(); - if fields.len() != 1 { - return Err(LoweringError::InvalidExpression(format!( - "IN (subquery) must select exactly one column, got {}", - fields.len() - ))); - } - let key = &fields[0]; - // Project the key under a name the outer relation cannot also carry. The - // join predicate resolves against the concatenated `left ++ right` - // schema, and a bare `hosts.service` over an unqualified subquery output - // falls back to a name lookup that finds the *left's* `service` first — - // silently making the predicate `service = service`, i.e. always true. - let right = match inner { - // Rebuild the subquery's projection with the synthetic alias, so a - // computed key (`SELECT bytes + 1 …`) is named rather than becoming - // the anonymous `col_0` that nothing can reference. - LogicalPlan::Projection(p) if p.expr.len() == 1 => Unresolved::Project { - cols: vec![ProjectItem { - alias: Some(IN_SUBQUERY_KEY.to_string()), - expr: df_expr_to_unresolved(unalias(&p.expr[0]))?, - }], - qualifier: None, - child: Rc::new(self.lower_plan(&p.input)?), - }, - other => Unresolved::Project { - cols: vec![ProjectItem { - alias: Some(IN_SUBQUERY_KEY.to_string()), - expr: Unresolved::Column(ColumnRef::Named(key.name().clone())), - }], - qualifier: None, - child: Rc::new(self.lower_plan(other)?), - }, - }; - Ok(Unresolved::Join { - kind: JoinKind::Semi, - pred: Predicate(Rc::new(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(&is.expr)?), - op: CompareOpKind::Eq, - right: Rc::new(Unresolved::Column(ColumnRef::Named( - IN_SUBQUERY_KEY.to_string(), - ))), - })), - left: Rc::new(left), - right: Rc::new(right), - }) - } - /// `[NOT] EXISTS (SELECT … WHERE inner.k = outer.k)` → a semi- / anti-join /// on the correlation predicate (issue #111). - fn lower_exists( + fn lower_correlated_exists( &self, ex: &logical_expr::expr::Exists, left: Unresolved, @@ -520,12 +477,9 @@ impl<'a> SqlLowerer<'a> { // the join predicate. Whatever is left stays an ordinary inner filter. let (inner, correlation) = split_correlation(inner)?; let right = self.lower_plan(&inner)?; - // No correlation conjunct (a genuinely uncorrelated `EXISTS`) means - // the join condition is unconditionally true — same convention as an - // unconditional `JOIN` (`lower_join`, below). let pred = match correlation { - Some(e) => Predicate(Rc::new(df_expr_to_unresolved(&e)?)), - None => Predicate(Rc::new(Unresolved::Literal(ScalarValue::Boolean(true)))), + Some(e) => Predicate(self.lower_expr(&e)?), + None => Predicate(Scalar::Literal(ScalarValue::Boolean(true))), }; Ok(Unresolved::Join { kind, @@ -535,6 +489,37 @@ impl<'a> SqlLowerer<'a> { }) } + /// `VALUES (…), (…)` — one row per values row, typed by DataFusion's + /// declared schema. Row expressions have no input-column scope. + fn lower_values(&self, values: &logical_expr::Values) -> Result { + let rows = values + .values + .iter() + .map(|row| row.iter().map(|e| self.lower_expr(e)).collect()) + .collect::>, LoweringError>>()?; + let fields = values + .schema + .fields() + .iter() + .map(|f| { + Ok(asap_types::ir::schema::Field::plain( + f.name().clone(), + arrow_to_dtype(f.data_type())?, + f.is_nullable(), + )) + }) + .collect::, LoweringError>>()?; + Ok(Unresolved::Values { + rows, + schema: Schema { + fields, + time_index: None, + unique_keys: vec![], + closed: true, + }, + }) + } + /// Table leaf — carries the catalog's resolved schema directly on `Scan` /// (`schema: Some(_)`), so `resolve_root`'s SchemaResolver doesn't need to /// usage-derive it (SQL is never schemaless). Projection pushdown is left @@ -598,23 +583,17 @@ impl<'a> SqlLowerer<'a> { let mut conjuncts = join .on .iter() - .map(|(l, r)| { - Ok(Unresolved::Compare { - left: Rc::new(df_expr_to_unresolved(l)?), - op: CompareOpKind::Eq, - right: Rc::new(df_expr_to_unresolved(r)?), - }) - }) + .map(|(l, r)| self.compare(l, CompareOpKind::Eq, r)) .collect::, LoweringError>>()?; if let Some(filter) = &join.filter { - conjuncts.push(df_expr_to_unresolved(filter)?); + conjuncts.push(self.lower_expr(filter)?); } - let pred = Predicate(Rc::new(match conjuncts.len() { + let pred = Predicate(match conjuncts.len() { // No condition (a CROSS JOIN) is unconditionally true. - 0 => Unresolved::Literal(ScalarValue::Boolean(true)), + 0 => Scalar::Literal(ScalarValue::Boolean(true)), 1 => conjuncts.pop().unwrap(), - _ => Unresolved::BoolAnd(conjuncts), - })); + _ => Scalar::BoolAnd(conjuncts), + }); Ok(Unresolved::Join { kind, pred, @@ -637,6 +616,10 @@ impl<'a> SqlLowerer<'a> { .window_expr .first() .ok_or_else(|| LoweringError::InvalidExpression("empty window expression".into()))?; + let first = match first { + Expr::Alias(alias) => alias.expr.as_ref(), + other => other, + }; let Expr::WindowFunction(wf) = first else { return Err(LoweringError::InvalidExpression( "expected a window function in Window plan node".into(), @@ -646,12 +629,12 @@ impl<'a> SqlLowerer<'a> { let mut args = wf .args .iter() - .map(df_expr_to_unresolved) + .map(|e| self.lower_expr(e)) .collect::, _>>()?; // Nth_value: lift N from the (literal) 2nd arg, keep only the column. let func = if matches!(func, WindowFuncKind::NthValue(None)) { let n = match args.get(1) { - Some(Unresolved::Literal(ScalarValue::Int64(n))) if *n > 0 => *n as u64, + Some(Scalar::Literal(ScalarValue::Int64(n))) if *n > 0 => *n as u64, other => { return Err(LoweringError::InvalidExpression(format!( "NTH_VALUE requires a positive integer literal 2nd arg, got {other:?}" @@ -672,7 +655,7 @@ impl<'a> SqlLowerer<'a> { .order_by .iter() .map(|s| { - df_expr_to_unresolved(&s.expr).map(|expr| SortKey { + self.lower_expr(&s.expr).map(|expr| SortKey { expr, ascending: s.asc, nulls_first: s.nulls_first, @@ -707,7 +690,7 @@ impl<'a> SqlLowerer<'a> { let input = self.lower_plan(&proj.input)?; return Ok(match bridge { PlanningBridge::PromqlSubquery { range, resolution } => { - let child = Rc::new(temporal_bridge_projection(proj, input)?); + let child = Rc::new(self.temporal_bridge_projection(proj, input)?); Unresolved::PromqlSubquery { range, resolution: Some(resolution), @@ -740,22 +723,22 @@ impl<'a> SqlLowerer<'a> { .map(|e| match e { Expr::Alias(a) => { let expr = if temporal_input && is_temporal_output_column(&a.expr) { - Unresolved::Column(ColumnRef::Named("value".into())) + Scalar::Column(ColumnRef::Named("value".into())) } else { - df_expr_to_unresolved(&a.expr)? + self.lower_expr(&a.expr)? }; - Ok::, LoweringError>(ProjectItem { + Ok::(ProjectItem { expr, alias: Some(a.name.clone()), }) } _ => { let expr = if temporal_input && is_temporal_output_column(e) { - Unresolved::Column(ColumnRef::Named("value".into())) + Scalar::Column(ColumnRef::Named("value".into())) } else { - df_expr_to_unresolved(e)? + self.lower_expr(e)? }; - Ok::, LoweringError>(ProjectItem { expr, alias: None }) + Ok::(ProjectItem { expr, alias: None }) } }) .collect::, _>>()?; @@ -802,7 +785,7 @@ impl<'a> SqlLowerer<'a> { // reducer expression (`GROUP BY date_trunc(…)`, `SUM(a * 8)`) has no // slot. Materialize each one as a derived column in a `Project` beneath // the aggregate, then group/reduce over that column (issue #110). - let mut derived = DerivedCols::default(); + let mut derived = DerivedCols::new(self); // DataFusion strips `AS m` from a grouping expression, so the aggregate // schema's field name is what the enclosing Projection references — @@ -827,7 +810,7 @@ impl<'a> SqlLowerer<'a> { .get(i) .cloned() .unwrap_or_else(|| other.to_string()); - derived.materialize(name.clone(), df_expr_to_unresolved(other)?)?; + derived.materialize(name.clone(), self.lower_expr(other)?)?; keys.push(ColumnRef::Named(name)); } } @@ -869,7 +852,7 @@ impl<'a> SqlLowerer<'a> { .iter() .map(|f| { f.as_ref() - .map(|f| Ok(Predicate(Rc::new(df_expr_to_unresolved(f)?)))) + .map(|f| Ok(Predicate(self.lower_expr(f)?))) .transpose() }) .collect::, LoweringError>>()? @@ -922,12 +905,7 @@ impl<'a> SqlLowerer<'a> { )) })?; - let resolved_input = resolve_root(&input)?; - let input_schema = resolved_input.output_schema().map_err(|error| { - LoweringError::InvalidExpression(format!( - "cannot derive temporal aggregate input schema: {error}" - )) - })?; + let input_schema = resolve_root(&input)?.schema.clone(); let timestamp_id = resolve_column_ref(×tamp_ref, &input_schema).map_err(|error| { LoweringError::InvalidExpression(format!("{name} timestamp argument: {error}")) })?; @@ -1000,18 +978,18 @@ impl<'a> SqlLowerer<'a> { let mut cols = vec![ ProjectItem { alias: Some("ts".into()), - expr: Unresolved::Column(timestamp_ref.clone()), + expr: Scalar::Column(timestamp_ref.clone()), }, ProjectItem { alias: Some("value".into()), - expr: Unresolved::Column(value_ref.clone()), + expr: Scalar::Column(value_ref.clone()), }, ]; for group_ref in group_refs { let group_name = named_ref(&group_ref).to_string(); cols.push(ProjectItem { alias: Some(group_name), - expr: Unresolved::Column(group_ref), + expr: Scalar::Column(group_ref), }); } let child = Unresolved::Project { @@ -1019,8 +997,11 @@ impl<'a> SqlLowerer<'a> { qualifier: None, child: Rc::new(input), }; + // The explicit window is a range selector over the series, the same + // shape PromQL's `rate(m[5m])` lowers to. let child = Unresolved::TimeRange { range: Duration::from_millis(window_ms), + kind: TimeRangeKind::Range, child: Rc::new(child), }; let intent = match name.as_str() { @@ -1099,7 +1080,7 @@ impl<'a> SqlLowerer<'a> { // Reducer arguments still materialize as derived columns (#110); the // grouping keys are plain columns, so they only need carrying through. - let mut derived = DerivedCols::default(); + let mut derived = DerivedCols::new(self); for e in &distinct { derived.passthrough(e)?; } @@ -1137,10 +1118,10 @@ impl<'a> SqlLowerer<'a> { .map(|((name, dtype), e)| ProjectItem { alias: Some(name.clone()), expr: if level.contains(e) { - Unresolved::Column(ColumnRef::Named(name.clone())) + Scalar::Column(ColumnRef::Named(name.clone())) } else { - Unresolved::Cast { - expr: Rc::new(Unresolved::Literal(ScalarValue::Null)), + Scalar::Cast { + expr: Box::new(Scalar::Literal(ScalarValue::Null)), to: dtype.clone(), try_cast: false, } @@ -1148,7 +1129,7 @@ impl<'a> SqlLowerer<'a> { }) .chain(output_names.iter().map(|n| ProjectItem { alias: Some(n.clone()), - expr: Unresolved::Column(ColumnRef::Named(n.clone())), + expr: Scalar::Column(ColumnRef::Named(n.clone())), })) .collect(); Ok(Unresolved::Project { @@ -1180,7 +1161,7 @@ impl<'a> SqlLowerer<'a> { .expr .iter() .map(|s| { - df_expr_to_unresolved(&s.expr).map(|expr| SortKey { + self.lower_expr(&s.expr).map(|expr| SortKey { expr, ascending: s.asc, nulls_first: s.nulls_first, @@ -1200,8 +1181,10 @@ impl<'a> SqlLowerer<'a> { // Count-ranked `LIMIT k` over a `Sort` is promoted to the heavy-hitter // `TopK` by the shared `canonicalize` pass (issue #34), not here. Ok(Unresolved::Limit { - n: eval_fetch(&limit.fetch).unwrap_or(usize::MAX), + // No (literal) fetch is offset-only. + n: eval_fetch(&limit.fetch), offset: eval_fetch(&limit.skip).unwrap_or(0), + partition_by: GroupKeys::none(), child: Rc::new(self.lower_plan(&limit.input)?), }) } @@ -1281,49 +1264,52 @@ fn planning_bridge( /// its output slot (`... asap_promql_subquery(...) AS value ...`). This makes /// the bridge schema-preserving without silently retaining columns that SQL /// projected away. -fn temporal_bridge_projection( - projection: &logical_expr::Projection, - child: Unresolved, -) -> Result { - let cols = projection - .expr - .iter() - .map(|expr| { - if let Expr::ScalarFunction(call) = unalias(expr) { - if call - .func - .name() - .eq_ignore_ascii_case("asap_promql_subquery") - { - let Expr::Alias(alias) = expr else { - return Err(LoweringError::InvalidExpression( - "asap_promql_subquery must have an alias naming its child value column" - .into(), - )); - }; - return Ok(ProjectItem { - expr: Unresolved::Column(ColumnRef::Named(alias.name.clone())), +impl SqlLowerer<'_> { + fn temporal_bridge_projection( + &self, + projection: &logical_expr::Projection, + child: Unresolved, + ) -> Result { + let cols = projection + .expr + .iter() + .map(|expr| { + if let Expr::ScalarFunction(call) = unalias(expr) { + if call + .func + .name() + .eq_ignore_ascii_case("asap_promql_subquery") + { + let Expr::Alias(alias) = expr else { + return Err(LoweringError::InvalidExpression( + "asap_promql_subquery must have an alias naming its child value column" + .into(), + )); + }; + return Ok(ProjectItem { + expr: Scalar::Column(ColumnRef::Named(alias.name.clone())), + alias: Some(alias.name.clone()), + }); + } + } + match expr { + Expr::Alias(alias) => Ok(ProjectItem { + expr: self.lower_expr(&alias.expr)?, alias: Some(alias.name.clone()), - }); + }), + other => Ok(ProjectItem { + expr: self.lower_expr(other)?, + alias: None, + }), } - } - match expr { - Expr::Alias(alias) => Ok(ProjectItem { - expr: df_expr_to_unresolved(&alias.expr)?, - alias: Some(alias.name.clone()), - }), - other => Ok(ProjectItem { - expr: df_expr_to_unresolved(other)?, - alias: None, - }), - } + }) + .collect::, LoweringError>>()?; + Ok(Unresolved::Project { + cols, + qualifier: None, + child: Rc::new(child), }) - .collect::, LoweringError>>()?; - Ok(Unresolved::Project { - cols, - qualifier: None, - child: Rc::new(child), - }) + } } fn positive_millis_literal(expr: &Expr, argument: &str) -> Result { @@ -1408,8 +1394,8 @@ fn arity_to_signature(arity: Arity) -> Signature { // (which must become a real `AggIntent`, hence the rewrite to a native // DataFusion aggregate shape `lower_agg_intent` can classify), a scalar // function call in this IR is already deliberately opaque — -// `expr::df_expr_to_unresolved`'s `Expr::ScalarFunction` arm lowers *any* -// scalar call generically to `Unresolved::FunctionCall { name, args }`, with +// `SqlLowerer::lower_expr`'s `Expr::ScalarFunction` arm lowers *any* +// scalar call generically to `UnresolvedScalar::FunctionCall { name, args }`, with // zero name-specific logic. So teaching DataFusion's planner to accept a // ClickHouse scalar builtin's name — a stub `ScalarUDF`, registered below — // is the entire fix; the existing generic lowering already does the rest. @@ -1923,23 +1909,19 @@ fn lower_arg_selector( })) } -/// The name an `IN (subquery)`'s key column is projected under, so the join -/// predicate cannot bind it to a same-named column of the outer relation. -const IN_SUBQUERY_KEY: &str = "__asap_in_key"; - /// Fold `pred` directly onto `child.predicates` when `child` is a bare `Scan` /// (a `WHERE` directly over a table), otherwise wrap it in an ordinary /// `Filter` — canonical's invariant that a `Filter` never sits directly over a /// `Scan`. A front end emitting the canonical shape directly is responsible /// for maintaining that invariant itself (issue #179). -fn filter_or_fold(pred: Unresolved, child: Unresolved) -> Unresolved { +fn filter_or_fold(pred: Scalar, child: Unresolved) -> Unresolved { match child { Unresolved::Scan { source, mut predicates, schema, } => { - predicates.push(Predicate(Rc::new(pred))); + predicates.push(Predicate(pred)); Unresolved::Scan { source, predicates, @@ -1947,7 +1929,7 @@ fn filter_or_fold(pred: Unresolved, child: Unresolved) -> Unresolved { } } other => Unresolved::Filter { - pred: Predicate(Rc::new(pred)), + pred: Predicate(pred), child: Rc::new(other), }, } @@ -1964,6 +1946,18 @@ fn split_conjunction<'a>(expr: &'a Expr, out: &mut Vec<&'a Expr>) { } } +/// Whether `expr` reads another operator anywhere inside it (`EXISTS`, +/// `IN (…)`, a scalar subquery). +fn reads_subquery(expr: &Expr) -> bool { + expr.exists(|e| { + Ok(matches!( + e, + Expr::ScalarSubquery(_) | Expr::InSubquery(_) | Expr::Exists(_) + )) + }) + .expect("the predicate never fails") +} + /// Re-`AND` the conjuncts, or `None` when there are none left. fn rebuild_conjunction(conjuncts: &[&Expr]) -> Option { conjuncts @@ -2112,9 +2106,9 @@ fn expand_grouping_set(gs: &logical_expr::GroupingSet) -> Vec> { /// The projection also has to carry through the plain columns the aggregate /// still references, since a `Project` replaces its child's schema rather than /// extending it. -#[derive(Default)] -struct DerivedCols { - cols: Vec>, +struct DerivedCols<'l> { + lowerer: &'l SqlLowerer<'l>, + cols: Vec, /// Whether any column is genuinely derived. Without one the aggregate keeps /// its original child, so DAGs that lower today keep their exact shape. any: bool, @@ -2123,12 +2117,21 @@ struct DerivedCols { collision: Option, } -impl DerivedCols { +impl<'l> DerivedCols<'l> { + fn new(lowerer: &'l SqlLowerer<'l>) -> Self { + Self { + lowerer, + cols: Vec::new(), + any: false, + collision: None, + } + } + /// Add `alias := expr`, or note a collision if `alias` already means /// something else. `Project` carries one relation qualifier for all its /// columns, so `a.k` and `b.k` cannot both survive it — but that only /// matters when a projection gets inserted at all. - fn push(&mut self, alias: String, expr: Unresolved) { + fn push(&mut self, alias: String, expr: Scalar) { let existing = self .cols .iter() @@ -2151,12 +2154,12 @@ impl DerivedCols { let Expr::Column(c) = unalias(expr) else { return Ok(()); }; - self.push(c.name.clone(), df_expr_to_unresolved(expr)?); + self.push(c.name.clone(), self.lowerer.lower_expr(expr)?); Ok(()) } /// A genuinely derived column: `alias` now names `expr`'s value. - fn materialize(&mut self, alias: String, expr: Unresolved) -> Result<(), LoweringError> { + fn materialize(&mut self, alias: String, expr: Scalar) -> Result<(), LoweringError> { self.any = true; self.push(alias, expr); Ok(()) @@ -2178,7 +2181,7 @@ impl DerivedCols { let mut rewritten = agg_fn.clone(); for arg in &mut rewritten.args { let alias = unalias(arg).to_string(); - self.materialize(alias.clone(), df_expr_to_unresolved(arg)?)?; + self.materialize(alias.clone(), self.lowerer.lower_expr(arg)?)?; *arg = Expr::Column(DfColumn::new_unqualified(alias)); } return Ok(Expr::AggregateFunction(rewritten)); @@ -2200,12 +2203,12 @@ impl DerivedCols { } match agg_col_name(&agg_fn.args) { Some(name) => { - self.push(name, df_expr_to_unresolved(arg)?); + self.push(name, self.lowerer.lower_expr(arg)?); Ok(expr.clone()) } None => { let alias = unalias(arg).to_string(); - self.materialize(alias.clone(), df_expr_to_unresolved(arg)?)?; + self.materialize(alias.clone(), self.lowerer.lower_expr(arg)?)?; let mut agg_fn = agg_fn.clone(); agg_fn.args[0] = Expr::Column(DfColumn::new_unqualified(alias)); Ok(Expr::AggregateFunction(agg_fn)) @@ -2276,7 +2279,7 @@ fn expr_to_group_ref(expr: &Expr) -> Result { match expr { // Preserve the relation qualifier so a GROUP BY / PARTITION BY key over a // join (`b.k` vs `a.k`) resolves to the correct side — the same rule the - // scalar predicate path uses (`df_expr_to_unresolved`). + // scalar predicate path uses (`lower_expr`). Expr::Column(col) => Ok(match &col.relation { Some(rel) => ColumnRef::Qualified { table: rel.to_string(), diff --git a/crates/frontend-sql/src/sql/types.rs b/crates/frontend-sql/src/sql/types.rs index 1d178daf4..6dcaa60f0 100644 --- a/crates/frontend-sql/src/sql/types.rs +++ b/crates/frontend-sql/src/sql/types.rs @@ -9,8 +9,8 @@ use datafusion::arrow::datatypes::{ }; use datafusion::common::ScalarValue as DfScalarValue; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::ScalarValue; +use asap_types::ir::scalar::ScalarValue; +use asap_types::ir::schema::{DataType, Field, Schema}; use crate::error::SqlError as LoweringError; @@ -364,9 +364,10 @@ mod bottom_map_tests { use super::*; #[test] fn empty_map_bottom_types_roundtrip_without_string_defaults() { - let (map, nullable) = asap_types::pre_asap::scalar_type_rules::MapScalarFunction::Construct - .output_type(&[]) - .unwrap(); + let (map, nullable) = + asap_types::ir::scalar::scalar_type_rules::MapScalarFunction::Construct + .output_type(&[]) + .unwrap(); assert!(!nullable); let arrow = dtype_to_arrow(&map); assert_eq!(arrow_to_dtype(&arrow).unwrap(), map); diff --git a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs index 23f757985..21cbbc6ef 100644 --- a/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs +++ b/crates/frontend-sql/tests/bgp_analytics/bgp_analytics.rs @@ -35,9 +35,12 @@ //! `Err`, never panics. The pinned per-query outcomes document today's real //! coverage so a regression (or a future improvement) is visible, not silent. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::operator::{AggIntent, GroupKeys}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use datafusion::error::DataFusionError; @@ -92,7 +95,7 @@ fn queries() -> Vec { .collect() } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result, LoweringError> { lower_sql_dialect( q, &catalog(), @@ -218,19 +221,19 @@ async fn corpus_lowering_matches_the_pinned_per_query_outcome() { ); } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match node.expect_non_asap() { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } @@ -260,7 +263,7 @@ async fn top_k_queries_are_count_grouped_by_prefix() { idx + 1 ); assert!( - matches!(qe, QueryExpr::Limit { .. }), + matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. }), "q{} ({label}) top-k shape keeps the LIMIT at the root: {qe:?}", idx + 1 ); diff --git a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs index 6e382bb57..2276fae24 100644 --- a/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs +++ b/crates/frontend-sql/tests/bgp_jan2024_workload/bgp_jan2024_workload.rs @@ -14,7 +14,7 @@ //! tally** by outcome category -- see the module doc on [`Category`] for why. use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; use datafusion::error::DataFusionError; @@ -70,7 +70,7 @@ fn catalog() -> SqlCatalog { .with_table("bgp.bgp_updates", updates) } -async fn lower(q: &str) -> Result { +async fn lower(q: &str) -> Result, SqlError> { lower_sql_dialect( q, &catalog(), @@ -91,6 +91,8 @@ async fn lower(q: &str) -> Result { /// is that signal, ratcheted so a category shifting size is visible. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] enum Category { + /// A planned expression lacks a faithful registered IR type contract. + InvalidRepresentation, Lowered, /// `DataFusionError::Plan` -- almost entirely "unknown function" for a /// ClickHouse-only builtin (`uniqExact`, `countIf`, `splitByChar`, ...). @@ -119,6 +121,7 @@ fn categorize(err: &SqlError) -> Category { SqlError::DataFusion(DataFusionError::SQL(_, _)) => Category::Parse, SqlError::DataFusion(DataFusionError::NotImplemented(_)) => Category::NotImplemented, SqlError::UnsupportedFeature(_) => Category::UnsupportedFeature, + SqlError::Convert(_) => Category::InvalidRepresentation, _ => Category::Other, } } @@ -194,8 +197,11 @@ async fn corpus_lowering_matches_the_pinned_aggregate_tally() { // 152 -> 154: `ScalarValue::Interval` (this branch) converts the // `INTERVAL x unit` literal the two `toStartOfInterval(...)` queries // carry. - expect(Category::Lowered, 154); - expect(Category::Plan, 40); + // Previously admitted ClickHouse stubs used placeholder Float64 types. + // Unregistered functions and incompatible operands now fail closed. + expect(Category::Lowered, 105); + expect(Category::InvalidRepresentation, 53); + expect(Category::Plan, 41); expect(Category::Schema, 0); expect(Category::Parse, 0); // One query that used to fail at `uniqExact` (`Plan`) now clears that @@ -207,7 +213,7 @@ async fn corpus_lowering_matches_the_pinned_aggregate_tally() { // Typed Map access lowers one prior gap; six array accesses now fail // during typed planning because the Map adapter rejects array inputs. expect(Category::NotImplemented, 0); - expect(Category::UnsupportedFeature, 6); + expect(Category::UnsupportedFeature, 1); // Was 2: the two `toStartOfInterval(...)` queries whose `INTERVAL`-literal // conversion gap the `toStartOfInterval` note above describes. Both now // lower end to end and are counted in `Lowered`. diff --git a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs index caed56594..284107b51 100644 --- a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs +++ b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs @@ -19,9 +19,12 @@ //! Schema: `packets(srcip, dstip, srcport, dstport, proto, time, pkt_len)`; //! flow / 5-tuple = `(srcip, dstip, srcport, dstport, proto)`. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::operator::{AggIntent, GroupKeys}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/synthetic_packet_trace_queries.sql"); @@ -67,110 +70,61 @@ fn queries() -> Vec { // ── DAG helpers ────────────────────────────────────────────────────────────── -/// Every `AggIntent` in the DAG, root-to-leaf. -fn intents(e: &QueryExpr) -> Vec { - let mut out = Vec::new(); - fn go(e: &QueryExpr, out: &mut Vec) { - match e { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::Project { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - QueryExpr::Concat { children, .. } => children.iter().for_each(|c| go(c, out)), - QueryExpr::PromqlVectorFromScalar(inner) | QueryExpr::PromqlScalarFromVector(inner) => { - go(inner, out) - } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - // Scalar expression variants (issue #205): `AggIntent` only ever - // lives in `Aggregate.measures`, never nested inside a scalar - // expression DAG, so there's nothing to recurse into here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} - } - } - go(e, &mut out); - out +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + +/// Every `AggIntent` in the DAG, root-to-leaf (every reachable node — +/// `AggIntent` only ever lives in `Aggregate.measures`). +fn intents(e: &Rc) -> Vec { + OperatorNode::reachable(e) + .iter() + .filter_map(|node| match op(node) { + NonASAPOp::Aggregate { measures, .. } => Some(measures.clone()), + _ => None, + }) + .flatten() + .collect() } /// The first `Aggregate`'s `(by, measures)` along the single-child spine. SQL /// never lowers to `Reduction::PerEntity` (it has no per-series concept), so /// `expect_reduce()` here is a safe, load-bearing assumption for these tests. -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::SQLWindowFunc { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } /// Whether a `SQLWindowFunc` (analytic `OVER (…)`) node appears anywhere. -fn has_window_func(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::SQLWindowFunc { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => has_window_func(child), +fn has_window_func(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::SQLWindowFunc { .. } => true, + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => has_window_func(child), _ => false, } } -async fn lower(q: &str) -> QueryExpr { +async fn lower(q: &str) -> Rc { lower_sql(q, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) diff --git a/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs b/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs index 83cff5ec9..85878c340 100644 --- a/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs +++ b/crates/frontend-sql/tests/data_quality_check/tpch_deequ.rs @@ -23,7 +23,7 @@ //! which is the same narrowing sidra's own catalog file makes. use asap_frontend_sql::{lower_sql, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/tpch_deequ_queries.sql"); diff --git a/crates/frontend-sql/tests/maintained_population.rs b/crates/frontend-sql/tests/maintained_population.rs index c8269fd2a..dc0d611d2 100644 --- a/crates/frontend-sql/tests/maintained_population.rs +++ b/crates/frontend-sql/tests/maintained_population.rs @@ -1,18 +1,18 @@ //! SQL and PromQL use the same shared-state rule without sharing membership semantics. -use asap_aware_mapping::maintained_population::MaintainedPopulationStrategy; use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::{ - post_asap::{ - compile_post_asap_dag, - maintained_population::{MaintainedPopulation, PopulationInput}, - share_common_summary_sub_dags, SummaryExpr, ValueOperation, - }, - pre_asap::{DataType, Field, QueryExpr, Schema}, - types::AccuracyTarget, +use asap_logical_optimizer::pass1::maintained_population::MaintainedPopulationStrategy; +use asap_types::ir::cse::share_common_sub_dags; +use asap_types::ir::export::compile_physical_asap_dag; +use asap_types::ir::operator::maintained_population::{MaintainedPopulation, PopulationInput}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{ + apply_materialization_timings, ASAPOp, MaterializationAssignment, NonASAPOp, Operator, + OperatorNode, TimingMemo, }; +use asap_types::types::AccuracyTarget; use std::rc::Rc; -async fn aggregate(q: &str) -> Rc { +async fn aggregate(q: &str) -> Rc { let catalog = SqlCatalog::new().with_table( "samples", Schema::new(vec![ @@ -20,38 +20,38 @@ async fn aggregate(q: &str) -> Rc { Field::plain("job", DataType::Utf8, false), ]), ); - let root = lower_sql(q, &catalog, AccuracyTarget::Exact).await.unwrap(); - Rc::new(root) + lower_sql(q, &catalog, AccuracyTarget::Exact).await.unwrap() } -fn population( - mut node: &asap_types::post_asap::SummaryNode, -) -> ( - &Rc, - &MaintainedPopulation, -) { - while let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { .. }, - .. - } = &node.expr - { +/// The `MaintainPopulation` node a candidate's evaluation reads, and its spec. +fn population(mut node: &OperatorNode) -> (&Rc, &MaintainedPopulation) { + while let Operator::NonASAP(NonASAPOp::Project { child, .. }) = &node.operator { node = child; } - let SummaryExpr::ValueOperation { child, .. } = &node.expr else { - panic!("readout") + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &node.operator else { + panic!("evaluation") }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!("state") }; (child, population) } -// Quantile parameters are readout identity, while source, value column and grouping are state identity. +/// Export `plan` the way the planner does: assign the default materialization +/// timings, then compile the timed DAG. +fn compile(plan: &Rc) -> Result<(), String> { + let timed = apply_materialization_timings( + plan, + &MaterializationAssignment::all_query_time(), + &mut TimingMemo::new(), + ) + .map_err(|e| e.to_string())?; + compile_physical_asap_dag(&timed) + .map(|_| ()) + .map_err(|e| e.to_string()) +} + +// Quantile parameters are evaluation identity, while source, value column and grouping are state identity. #[tokio::test] async fn sql_quantiles_share_rows_without_promql_lookback() { let roots = vec![ @@ -59,7 +59,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { aggregate("SELECT approx_percentile_cont(latency, 0.99) FROM samples").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -67,7 +67,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); } let (a, spec) = population(&plans[0].1); let (b, _) = population(&plans[1].1); @@ -107,9 +107,9 @@ async fn sql_filters_separate_populations() { assert_ne!(population(&a).1.input, population(&b).1.input); } -// All four scalar readouts can share the same non-null numeric SQL population. +// All four scalar evaluations can share the same non-null numeric SQL population. #[tokio::test] -async fn sql_scalar_readouts_share_membership() { +async fn sql_scalar_evaluations_share_membership() { let mut roots = Vec::new(); for function in [ "median(latency)", @@ -120,7 +120,7 @@ async fn sql_scalar_readouts_share_membership() { roots.push(aggregate(&format!("SELECT {function} FROM samples")).await); } let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -128,27 +128,29 @@ async fn sql_scalar_readouts_share_membership() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } } -// A readout cannot reinterpret a label column as its numeric population. +// A evaluation cannot reinterpret a label column as its numeric population. #[tokio::test] async fn malformed_table_population_fails_validation() { let root = aggregate("SELECT median(latency) FROM samples").await; let rule = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)); let mut candidate = rule.candidate(&root).unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &mut Rc::make_mut(&mut candidate).expr else { + let Operator::NonASAP(NonASAPOp::Project { child, .. }) = + &mut Rc::make_mut(&mut candidate).operator + else { unreachable!() }; - let SummaryExpr::ValueOperation { child, .. } = &mut Rc::make_mut(child).expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = + &mut Rc::make_mut(child).operator + else { unreachable!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &mut Rc::make_mut(child).expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = + &mut Rc::make_mut(child).operator else { unreachable!() }; @@ -156,7 +158,7 @@ async fn malformed_table_population_fails_validation() { unreachable!() }; *value_column = 1; - assert!(compile_post_asap_dag(&candidate).is_err()); + assert!(compile(&candidate).is_err()); } // SQL ORDER BY value DESC LIMIT k uses the same maximum-k state contract. @@ -167,7 +169,7 @@ async fn sql_topk_limits_share_maximum_k() { aggregate("SELECT * FROM samples ORDER BY latency DESC LIMIT 5").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -175,7 +177,7 @@ async fn sql_topk_limits_share_maximum_k() { .collect(), ); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); + compile(plan).unwrap(); assert_eq!(population(plan).1.max_k, 5); assert!(Rc::ptr_eq(population(&plans[0].1).0, population(plan).0)); } diff --git a/crates/frontend-sql/tests/netflow/netflow.rs b/crates/frontend-sql/tests/netflow/netflow.rs index 22236b850..9d06d46a6 100644 --- a/crates/frontend-sql/tests/netflow/netflow.rs +++ b/crates/frontend-sql/tests/netflow/netflow.rs @@ -4,9 +4,12 @@ //! aggregate over a netflow table, a time predicate, optional grouping, //! optional `ORDER BY`/`LIMIT`, plus the nested aggregate shape. +use std::rc::Rc; + use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr}; +use asap_types::ir::operator::{AggIntent, GroupKeys}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::types::AccuracyTarget; const CORPUS: &str = include_str!("data/netflow.sql"); @@ -109,11 +112,10 @@ async fn netflow_sql_corpus_lowers_to_expected_intents() { ); for (idx, (query, expected)) in queries.iter().zip(EXPECTED).enumerate() { + // A successful `lower_sql` already derived every node's schema. let qe = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|err| panic!("q{} failed to lower:\n{query}\n{err}", idx + 1)); - qe.output_schema() - .unwrap_or_else(|err| panic!("q{} schema derivation failed: {err}", idx + 1)); assert!( has_scan_predicate(&qe), "q{} should retain the netflow time predicate on the Scan: {qe:?}", @@ -123,7 +125,7 @@ async fn netflow_sql_corpus_lowers_to_expected_intents() { } } -fn assert_expected(qe: &QueryExpr, expected: Expected, case_no: usize) { +fn assert_expected(qe: &Rc, expected: Expected, case_no: usize) { match expected { Expected::Quantile { q, by } => { let (actual_by, measures) = first_aggregate(qe).expect("expected Aggregate"); @@ -190,53 +192,58 @@ impl AggKind { } } -fn first_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + +fn first_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => first_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_aggregate(child), _ => None, } } -fn has_scan_predicate(qe: &QueryExpr) -> bool { +fn has_scan_predicate(qe: &Rc) -> bool { any_node( qe, - |node| matches!(node, QueryExpr::Scan { predicates, .. } if !predicates.is_empty()), + |node| matches!(op(node), NonASAPOp::Scan { predicates, .. } if !predicates.is_empty()), ) } -fn has_topk(qe: &QueryExpr, k: usize) -> bool { +fn has_topk(qe: &Rc, k: usize) -> bool { any_node(qe, |node| { matches!( - node, - QueryExpr::Aggregate { measures, .. } + op(node), + NonASAPOp::Aggregate { measures, .. } if measures.iter().any(|agg| matches!(agg, AggIntent::TopK { k: actual, .. } if *actual == k)) ) }) } fn aggregate_by_with( - qe: &QueryExpr, + qe: &Rc, by: &'static [usize], pred: impl Fn(&AggIntent) -> bool, ) -> bool { let expected_by = GroupKeys::by(by.to_vec()); let mut found = false; visit(qe, &mut |node| { - if let QueryExpr::Aggregate { + if let NonASAPOp::Aggregate { reduction, measures, .. - } = node + } = op(node) { found |= *reduction.expect_reduce() == expected_by && measures.iter().any(&pred); } @@ -244,78 +251,26 @@ fn aggregate_by_with( found } -fn all_intents(qe: &QueryExpr) -> Vec { +fn all_intents(qe: &Rc) -> Vec { let mut intents = Vec::new(); visit(qe, &mut |node| { - if let QueryExpr::Aggregate { measures, .. } = node { + if let NonASAPOp::Aggregate { measures, .. } = op(node) { intents.extend(measures.iter().cloned()); } }); intents } -fn any_node(qe: &QueryExpr, pred: impl Fn(&QueryExpr) -> bool) -> bool { +fn any_node(qe: &Rc, pred: impl Fn(&OperatorNode) -> bool) -> bool { let mut found = false; visit(qe, &mut |node| found |= pred(node)); found } -fn visit(qe: &QueryExpr, f: &mut impl FnMut(&QueryExpr)) { - f(qe); - match qe { - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::TimeRange { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlRelabel { child, .. } - | QueryExpr::PromqlSeriesSample { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlInfoEnrich { child, .. } => visit(child, f), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - visit(lhs, f); - visit(rhs, f); - } - QueryExpr::Concat { children, .. } => { - for child in children { - visit(child, f); - } - } - QueryExpr::PromqlVectorFromScalar(child) | QueryExpr::PromqlScalarFromVector(child) => { - visit(child, f) - } - QueryExpr::Scan { .. } - | QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => {} - // Scalar expression variants (issue #205) aren't relational nodes; - // this visitor only walks the relational DAG, so stop here. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => {} +/// Every reachable operator node, parents before children — including the +/// operators referenced from scalar positions (subqueries). +fn visit(qe: &Rc, f: &mut impl FnMut(&OperatorNode)) { + for node in OperatorNode::reachable(qe) { + f(&node); } } diff --git a/crates/frontend-sql/tests/pearson_corr.rs b/crates/frontend-sql/tests/pearson_corr.rs index 618f6c38f..ea648bece 100644 --- a/crates/frontend-sql/tests/pearson_corr.rs +++ b/crates/frontend-sql/tests/pearson_corr.rs @@ -2,7 +2,12 @@ use std::rc::Rc; use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::pre_asap::{AggIntent, DataType, Field, QueryExpr, Schema}; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{ + apply_materialization_timings, export::compile_physical_asap_dag, MaterializationAssignment, + NonASAPOp, OperatorNode, ScalarExpr, TimingMemo, +}; use asap_types::types::AccuracyTarget; fn catalog() -> SqlCatalog { @@ -16,20 +21,20 @@ fn catalog() -> SqlCatalog { .with_table("b", schema) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> Rc { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap() } -fn aggregate(query: &QueryExpr) -> (&[AggIntent], &QueryExpr) { - match query { - QueryExpr::Aggregate { +fn aggregate(query: &OperatorNode) -> (&[AggIntent], &OperatorNode) { + match query.expect_non_asap() { + NonASAPOp::Aggregate { measures, child, .. } => (measures, child), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } => aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } => aggregate(child), other => panic!("expected aggregate, got {other:?}"), } } @@ -46,17 +51,14 @@ async fn corr_materializes_both_arguments() { let query = lower(sql).await; let (measures, child) = aggregate(&query); assert_eq!(measures, &[AggIntent::PearsonCorr { left: 0, right: 1 }]); - let QueryExpr::Project { cols, .. } = child else { + let NonASAPOp::Project { cols, .. } = child.expect_non_asap() else { panic!("derived inputs") }; assert_eq!(cols.len(), 2); assert!(cols .iter() - .any(|col| !matches!(col.expr, QueryExpr::Column(_)))); - assert_eq!( - query.output_schema().unwrap().fields[0].dtype, - DataType::Float64 - ); + .any(|col| !matches!(col.expr, ScalarExpr::Column(_)))); + assert_eq!(query.schema.fields[0].dtype, DataType::Float64); } } @@ -66,11 +68,11 @@ async fn corr_preserves_qualified_join_inputs() { let query = lower("SELECT corr(a.x, b.x) FROM a JOIN b ON a.g = b.g").await; let (measures, child) = aggregate(&query); assert_eq!(measures[0].input_cols(), vec![0, 1]); - let QueryExpr::Project { cols, .. } = child else { + let NonASAPOp::Project { cols, .. } = child.expect_non_asap() else { panic!("paired projection") }; - assert_eq!(cols[0].expr, QueryExpr::Column(0)); - assert_eq!(cols[1].expr, QueryExpr::Column(3)); + assert_eq!(cols[0].expr, ScalarExpr::Column(0)); + assert_eq!(cols[1].expr, ScalarExpr::Column(3)); } // Grouping and sibling reducers cannot drop either correlation argument. @@ -82,12 +84,12 @@ async fn corr_coexists_with_grouping_having_and_other_measures() { .iter() .find(|m| matches!(m, AggIntent::PearsonCorr { .. })) .unwrap(); - let schema = child.output_schema().unwrap(); + let schema = &child.schema; for id in pair.input_cols() { assert!(id < schema.fields.len()); } assert!(measures.iter().any(|m| matches!(m, AggIntent::Sum { .. }))); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[1].name, "r"); assert_eq!(output.fields[1].dtype, DataType::Float64); assert!(output.fields[1].nullable); @@ -99,8 +101,8 @@ async fn corr_repeated_input_and_serialization() { let query = lower("SELECT corr(x, x) FROM a").await; assert_eq!(aggregate(&query).0[0].input_cols(), vec![0, 0]); let encoded = serde_json::to_string(&query).unwrap(); - let decoded: QueryExpr = serde_json::from_str(&encoded).unwrap(); - assert_eq!(query, decoded); + let decoded: OperatorNode = serde_json::from_str(&encoded).unwrap(); + assert_eq!(*query, decoded); } // Unsupported modifiers and window calls fail instead of silently changing semantics. @@ -128,10 +130,10 @@ async fn corr_filter_is_a_measure_filter() { let query = lower("SELECT corr(x, y) FILTER (WHERE g > 0) FROM a").await; let (measures, _) = aggregate(&query); assert_eq!(measures[0].input_cols(), vec![0, 1]); - fn filters(query: &QueryExpr) -> &[Option] { - match query { - QueryExpr::Aggregate { filters, .. } => filters, - QueryExpr::Project { child, .. } | QueryExpr::Filter { child, .. } => filters(child), + fn filters(query: &OperatorNode) -> &[Option] { + match query.expect_non_asap() { + NonASAPOp::Aggregate { filters, .. } => filters, + NonASAPOp::Project { child, .. } | NonASAPOp::Filter { child, .. } => filters(child), other => panic!("expected aggregate, got {other:?}"), } } @@ -145,12 +147,17 @@ async fn corr_filter_is_a_measure_filter() { // Exact fallback retains the complete typed query and compiles to a post-ASAP DAG. #[tokio::test] async fn corr_survives_exact_plan_compilation() { - let query = Rc::new(lower("SELECT corr(x, y) AS r FROM a").await); - let plan = asap_aware_mapping::replacement::keep_pre_asap(&query).unwrap(); + let query = lower("SELECT corr(x, y) AS r FROM a").await; + let plan = asap_logical_optimizer::pass1::replacement::retain_exact(&query).unwrap(); assert!(plan.guarantee.as_ref().unwrap().is_exact()); - let asap_types::post_asap::SummaryExpr::KeepPreAsap(retained) = &plan.expr else { - panic!("expected exact fallback"); - }; - assert_eq!(aggregate(retained).0, aggregate(&query).0); - asap_types::post_asap::compile_post_asap_dag(&plan).unwrap(); + // The exact fallback is the query's own operator DAG, no ASAP node added. + assert!(!plan.contains_asap(), "expected exact fallback"); + assert_eq!(aggregate(&plan).0, aggregate(&query).0); + let timed = apply_materialization_timings( + &plan, + &MaterializationAssignment::all_query_time(), + &mut TimingMemo::new(), + ) + .unwrap(); + compile_physical_asap_dag(&timed).unwrap(); } diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index 883966db9..07987542f 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -1,16 +1,26 @@ -//! End-to-end SQL → unresolved → canonical DAG lowering tests (positional IR). +//! End-to-end SQL → unresolved → resolved operator DAG lowering tests. //! //! Validates the DataFusion front end: SQL parses + plans, lowers directly to -//! the canonical, unresolved shape (`QueryExpr`, issue #179), and -//! the shared `resolve_root` produces the positional, resolved canonical -//! DAG (the same resolver the PromQL path uses). - -use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; -use asap_types::pre_asap::schema::{DataType, Field, FieldDataType, Schema}; -use asap_types::pre_asap::{ - AggIntent, CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, Reduction, ScalarValue, - Source, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, +//! the name-based `UnresolvedOp` tree (issue #179), and the shared +//! `resolve_root` produces the positional, canonical `OperatorNode` DAG (the +//! same resolver the PromQL path uses). Every node's schema is derived during +//! resolution, so a successful `lower` already proves schema derivation is +//! total over the tree. + +use asap_types::ir::Predicate; +use std::rc::Rc; + +use asap_frontend_common::{UnresolvedOp, UnresolvedScalar}; +use asap_frontend_sql::{ + lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError, SqlLowerer, }; +use asap_types::ir::operator::{ + AggIntent, GroupKeys, JoinKind, Reduction, Source, WindowFrameBound, WindowFrameOffset, + WindowFrameUnits, WindowFuncKind, +}; +use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; +use asap_types::ir::schema::{DataType, Field, FieldDataType, Schema}; +use asap_types::ir::{ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; @@ -43,37 +53,21 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str) -> QueryExpr { +async fn lower(sql: &str) -> Rc { lower_sql(sql, &catalog(), AccuracyTarget::Exact) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) } +/// The operator of a front-end node: a front-end DAG never holds an ASAP node. +fn op(node: &OperatorNode) -> &NonASAPOp { + node.expect_non_asap() +} + #[tokio::test] -async fn planning_subquery_bridge_reuses_canonical_promql_subquery() { - let query = lower( - "SELECT max(value) FROM (\ - SELECT asap_promql_subquery(21600000, 60000) AS value FROM (\ - SELECT sum(bytes) AS value FROM metrics))", - ) - .await; - let QueryExpr::Project { child, .. } = query else { - panic!("expected outer SQL projection"); - }; - let QueryExpr::Aggregate { child, .. } = child.as_ref() else { - panic!("expected outer max aggregate, got {child:?}"); - }; - let QueryExpr::PromqlSubquery { - range, - resolution, - child, - } = child.as_ref() - else { - panic!("expected canonical subquery bridge, got {child:?}"); - }; - assert_eq!(*range, std::time::Duration::from_secs(6 * 60 * 60)); - assert_eq!(*resolution, Some(std::time::Duration::from_secs(60))); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); +async fn planning_subquery_bridge_rejects_a_relation_without_vector_conversion() { + let result = lower_sql("SELECT max(value) FROM (SELECT asap_promql_subquery(21600000, 60000) AS value FROM (SELECT sum(bytes) AS value FROM metrics))", &catalog(), AccuracyTarget::Exact).await; + assert!(result.is_err()); } #[tokio::test] @@ -83,12 +77,12 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { SELECT service AS le, sum(bytes) AS value FROM metrics GROUP BY service)", ) .await; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = query + } = op(&query) else { panic!("expected canonical histogram aggregate"); }; @@ -98,7 +92,7 @@ async fn planning_histogram_bridge_reuses_classic_bucket_intent() { measures.as_slice(), [AggIntent::HistogramQuantile { q, le: 0 }] if (*q - 0.95).abs() < 1e-12 )); - assert!(matches!(child.as_ref(), QueryExpr::Project { .. })); + assert!(matches!(op(child), NonASAPOp::Project { .. })); } #[tokio::test] @@ -134,31 +128,31 @@ async fn planning_relation_bridges_reject_ambiguous_shapes() { } /// Find the first `Aggregate` node along the single-child spine. -fn find_aggregate(qe: &QueryExpr) -> Option<(&GroupKeys, &Vec)> { - match qe { - QueryExpr::Aggregate { +fn find_aggregate(node: &OperatorNode) -> Option<(&GroupKeys, &Vec)> { + match op(node) { + NonASAPOp::Aggregate { reduction, measures, .. } => Some((reduction.expect_reduce(), measures)), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_aggregate(child), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_aggregate(child), _ => None, } } /// The first `Aggregate` node itself, for tests that need its child. -fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Aggregate { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find_aggregate_node(child), +fn find_aggregate_node(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Aggregate { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => find_aggregate_node(child), _ => None, } } @@ -166,47 +160,47 @@ fn find_aggregate_node(qe: &QueryExpr) -> Option<&QueryExpr> { /// The names of the columns the first `Aggregate`'s reducers read, resolved /// against its child's schema, plus whether that child is a materializing /// `Project` (issue #110). -fn reducer_input_names(qe: &QueryExpr) -> (Vec, bool) { - let QueryExpr::Aggregate { +fn reducer_input_names(node: &OperatorNode) -> (Vec, bool) { + let NonASAPOp::Aggregate { measures, child, .. - } = find_aggregate_node(qe).expect("expected an Aggregate") + } = op(find_aggregate_node(node).expect("expected an Aggregate")) else { unreachable!() }; - let schema = child.output_schema().expect("child schema"); + let schema = &child.schema; let names = measures .iter() .flat_map(|a| a.input_cols()) .map(|id| schema.fields[id].name.clone()) .collect(); - (names, matches!(**child, QueryExpr::Project { .. })) + (names, matches!(op(child), NonASAPOp::Project { .. })) } /// Find the first `Join` node along the single-child spine. -fn find_join(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Join { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_join(child), +fn find_join(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Join { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_join(child), _ => None, } } /// The first `Filter` node along the single-child spine. -fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::Filter { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_filter(child), +fn find_filter(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::Filter { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_filter(child), _ => None, } } @@ -215,11 +209,11 @@ fn find_filter(qe: &QueryExpr) -> Option<&QueryExpr> { async fn select_star_with_where_folds_predicate_onto_scan() { // SELECT * elides the projection; WHERE folds onto the Scan predicates. let qe = lower("SELECT * FROM metrics WHERE service = 'api'").await; - let QueryExpr::Scan { + let NonASAPOp::Scan { source, predicates, schema, - } = &qe + } = op(&qe) else { panic!("expected Scan at root, got {qe:?}"); }; @@ -254,9 +248,7 @@ async fn projection_over_aggregate_resolves_output_types_via_output_names() { // onto the canonical Aggregate so the Project resolves real types — not // the Utf8 fallback that an unresolved column would get. let qe = lower("SELECT SUM(bytes), AVG(latency) FROM metrics").await; - let schema = qe - .output_schema() - .expect("root projection schema derivation"); + let schema = &qe.schema; assert_eq!(schema.fields.len(), 2); assert_eq!( schema.fields[0].dtype, @@ -284,7 +276,7 @@ async fn single_agg_group_by_keeps_key_in_output_schema() { )); // Both the group key and the aggregate resolve in the root projection schema. - let schema = qe.output_schema().expect("root projection schema"); + let schema = &qe.schema; assert_eq!(schema.fields.len(), 2); assert_eq!( schema.fields[0].dtype, @@ -315,7 +307,7 @@ async fn count_ranked_topk_is_heavy_hitter() { "count-ranked topk → heavy-hitter TopK, got {measures:?}" ); // The inner child is the explicit Count, grouped by service (col 1). - let QueryExpr::Aggregate { child, .. } = &qe else { + let NonASAPOp::Aggregate { child, .. } = op(&qe) else { panic!("expected outer Aggregate, got {qe:?}"); }; let (inner_by, inner_measures) = find_aggregate(child).expect("expected inner Count aggregate"); @@ -426,7 +418,7 @@ async fn select_distinct_lowers_to_distinct_with_positional_cols() { // (not name-based ColumnRefs). DataFusion's `Distinct::All` dedups on every // column, so `cols` is empty here — but the field type is now `Vec`. let qe = lower("SELECT DISTINCT service FROM metrics").await; - let QueryExpr::Dedup { cols, .. } = &qe else { + let NonASAPOp::Dedup { cols, .. } = op(&qe) else { panic!("expected a Dedup at the root, got {qe:?}"); }; let _: &Vec = cols; // compile-time: positional ids, not ColumnRefs @@ -441,34 +433,35 @@ async fn inner_join_lowers_to_join_over_two_scans() { FROM metrics JOIN hosts ON metrics.service = hosts.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the DAG"); - let QueryExpr::Join { + let join = find_join(&qe).expect("expected a Join in the tree"); + let NonASAPOp::Join { kind, left, right, .. - } = join + } = op(join) else { unreachable!("find_join only returns Join"); }; assert_eq!(*kind, JoinKind::Inner); - assert!(matches!(left.as_ref(), QueryExpr::Scan { .. })); - assert!(matches!(right.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(op(left), NonASAPOp::Scan { .. })); + assert!(matches!(op(right), NonASAPOp::Scan { .. })); } /// The two `ColumnId`s an equijoin predicate `Column(l) = Column(r)` binds to, /// returned sorted so the assertion is independent of left/right ordering. -fn join_eq_columns(join: &QueryExpr) -> [usize; 2] { - let QueryExpr::Join { pred, .. } = join else { +fn join_eq_columns(join: &OperatorNode) -> [usize; 2] { + let NonASAPOp::Join { pred, .. } = op(join) else { unreachable!("expected a Join"); }; - let QueryExpr::Compare { + let ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, - } = pred.0.as_ref() + .. + } = &pred.0 else { panic!("expected an equijoin Compare, got {:?}", pred.0); }; match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(l), QueryExpr::Column(r)) => { + (ScalarExpr::Column(l), ScalarExpr::Column(r)) => { let mut cols = [*l, *r]; cols.sort_unstable(); cols @@ -567,12 +560,12 @@ async fn qualified_where_over_join_resolves_to_right_side() { ) .await; let filter = find_filter(&qe).expect("expected a Filter over the join"); - let QueryExpr::Filter { pred, .. } = filter else { + let NonASAPOp::Filter { pred, .. } = op(filter) else { unreachable!("find_filter only returns Filter"); }; assert!( - matches!(pred.0.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Eq, .. } - if matches!(left.as_ref(), QueryExpr::Column(4))), + matches!(&pred.0, ScalarExpr::Compare { left, op: CompareOpKind::Eq, .. } + if matches!(left.as_ref(), ScalarExpr::Column(4))), "hosts.service must bind to concatenated position 4 (not the first `service`), got {:?}", pred.0 ); @@ -636,20 +629,24 @@ async fn aggregate_over_join_binds_against_concatenated_schema() { } // ── Issue #111: IN / EXISTS subquery predicates become semi / anti joins ──── +// +// The front end now leaves them as `UnresolvedScalar::{InSubquery, Exists}` +// filter conjuncts; the shared `canonicalize` pass (run by `resolve_root`) +// lowers each to the semi-/anti-join, so the resolved DAG a test sees is the +// same join shape the front end used to emit directly. /// The first `Join` node's `(kind, predicate, left column count)`. -fn join_parts(qe: &QueryExpr) -> (&JoinKind, &QueryExpr, usize) { - let QueryExpr::Join { +fn join_parts(node: &OperatorNode) -> (&JoinKind, &ScalarExpr, usize) { + let NonASAPOp::Join { kind, pred, left, right: _, - } = find_join(qe).expect("expected a Join") + } = op(find_join(node).expect("expected a Join")) else { unreachable!() }; - let left_len = left.output_schema().expect("left schema").fields.len(); - (kind, pred.0.as_ref(), left_len) + (kind, &pred.0, left.schema.fields.len()) } #[tokio::test] @@ -664,14 +661,15 @@ async fn in_subquery_lowers_to_a_semi_join() { // The predicate resolves against `left ++ right`. Both relations have a // `service` column, so a name-based lookup would bind *both* sides to the // left's — silently making this `service = service`, always true. The key is - // projected under a synthetic name to make that impossible. - let QueryExpr::Compare { left, right, .. } = pred else { + // bound positionally to the subquery's column (right after the left's), + // which makes that impossible. + let ScalarExpr::Compare { left, right, .. } = pred else { panic!("expected a comparison, got {pred:?}"); }; - assert_eq!(**left, QueryExpr::Column(1), "outer service"); + assert_eq!(**left, ScalarExpr::Column(1), "outer service"); assert_eq!( **right, - QueryExpr::Column(left_len), + ScalarExpr::Column(left_len), "the subquery key, not the outer column again" ); } @@ -682,20 +680,14 @@ async fn a_semi_join_outputs_only_the_left_schema() { let qe = lower("SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)").await; let join = find_join(&qe).expect("expected a Join"); - let names: Vec<_> = join - .output_schema() - .expect("join schema") - .fields - .iter() - .map(|c| c.name.clone()) - .collect(); + let names: Vec<_> = join.schema.fields.iter().map(|c| c.name.clone()).collect(); assert_eq!(names, ["ts", "service", "latency", "bytes"]); } #[tokio::test] async fn a_subquery_key_that_is_an_expression_still_binds() { - // `SELECT bytes + 1 …` has no column name of its own; it is projected under - // the synthetic key rather than becoming an unreferenceable `col_0`. + // `SELECT bytes + 1 …` has no column name of its own; the join key binds + // to it positionally rather than through an unreferenceable `col_0`. let qe = lower("SELECT service FROM metrics WHERE bytes IN (SELECT bytes + 1 FROM metrics)").await; assert_eq!(join_parts(&qe).0, &JoinKind::Semi); @@ -723,13 +715,13 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { AND service IN (SELECT service FROM hosts)", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } => scan_has_predicate(child), - QueryExpr::Join { left, right, .. } => { + fn scan_has_predicate(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } => scan_has_predicate(child), + NonASAPOp::Join { left, right, .. } => { scan_has_predicate(left) || scan_has_predicate(right) } _ => false, @@ -743,16 +735,16 @@ async fn an_ordinary_conjunct_still_folds_onto_the_scan() { } /// Find the first `SQLWindowFunc` node along the single-child spine. -fn find_windowfunc(qe: &QueryExpr) -> Option<&QueryExpr> { - match qe { - QueryExpr::SQLWindowFunc { .. } => Some(qe), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => find_windowfunc(child), +fn find_windowfunc(node: &OperatorNode) -> Option<&OperatorNode> { + match op(node) { + NonASAPOp::SQLWindowFunc { .. } => Some(node), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => find_windowfunc(child), _ => None, } } @@ -766,12 +758,12 @@ async fn window_function_lowers_to_positional_windowfunc() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { + let NonASAPOp::SQLWindowFunc { func, partition_by, order_by, .. - } = win + } = op(win) else { unreachable!("find_windowfunc only returns SQLWindowFunc"); }; @@ -780,14 +772,14 @@ async fn window_function_lowers_to_positional_windowfunc() { assert_eq!(order_by.len(), 1); assert_eq!( order_by[0].expr, - QueryExpr::Column(3), + ScalarExpr::Column(3), "ORDER BY bytes → col 3" ); assert!(!order_by[0].ascending, "DESC"); // The window output column is appended to the schema (Int64 for ROW_NUMBER), // and the enclosing projection resolves it (output_name threading). - let schema = qe.output_schema().expect("root schema"); + let schema = &qe.schema; assert!( schema.fields.iter().any(|c| c.dtype == DataType::Int64), "row_number output column present, got {:?}", @@ -799,11 +791,11 @@ async fn window_function_lowers_to_positional_windowfunc() { async fn window_aggregate_lowers_to_windowfunc() { let qe = lower("SELECT service, SUM(bytes) OVER (PARTITION BY service) FROM metrics").await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::Sum); - assert_eq!(args, &vec![QueryExpr::Column(3)], "SUM(bytes) → arg col 3"); + assert_eq!(args, &vec![ScalarExpr::Column(3)], "SUM(bytes) → arg col 3"); } // ── Window frames (issue #268) ─────────────────────────────────────────────── @@ -827,8 +819,8 @@ async fn window_frame_is_captured_not_dropped() { ) .await; - let frame_of = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { frame, .. } = find_windowfunc(qe).unwrap() else { + let frame_of = |node: &OperatorNode| { + let NonASAPOp::SQLWindowFunc { frame, .. } = op(find_windowfunc(node).unwrap()) else { unreachable!(); }; frame @@ -868,9 +860,9 @@ async fn range_interval_frame_is_preserved() { RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW) FROM metrics", ) .await; - let QueryExpr::SQLWindowFunc { + let NonASAPOp::SQLWindowFunc { frame: Some(frame), .. - } = find_windowfunc(&qe).unwrap() + } = op(find_windowfunc(&qe).unwrap()) else { panic!("expected a window function with a concrete frame"); }; @@ -900,10 +892,10 @@ async fn range_numeric_frames_remain_scalar_offsets() { ) .await; - let start_bound = |qe: &QueryExpr| { - let QueryExpr::SQLWindowFunc { + let start_bound = |node: &OperatorNode| { + let NonASAPOp::SQLWindowFunc { frame: Some(frame), .. - } = find_windowfunc(qe).unwrap() + } = op(find_windowfunc(node).unwrap()) else { panic!("expected a window function with a concrete frame"); }; @@ -938,43 +930,17 @@ async fn groups_frame_is_rejected() { // ── Nested query functions: derived tables / inline views (issue #27) ─────────── -/// Collect every `AggIntent` in the DAG, root-to-leaf. -fn all_intents(qe: &QueryExpr) -> Vec { - let mut out = Vec::new(); - fn go(qe: &QueryExpr, out: &mut Vec) { - match qe { - QueryExpr::Aggregate { - measures, child, .. - } => { - out.extend(measures.iter().cloned()); - go(child, out); - } - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Dedup { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::SQLWindowFunc { child, .. } - | QueryExpr::PromqlSubquery { child, .. } => go(child, out), - QueryExpr::BinaryOp { lhs, rhs, .. } - | QueryExpr::Join { - left: lhs, - right: rhs, - .. - } - | QueryExpr::SetOp { - left: lhs, - right: rhs, - .. - } => { - go(lhs, out); - go(rhs, out); - } - _ => {} - } - } - go(qe, &mut out); - out +/// Collect every `AggIntent` in the DAG, root-to-leaf (every reachable node, +/// including operators referenced from scalar positions). +fn all_intents(root: &Rc) -> Vec { + OperatorNode::reachable(root) + .iter() + .filter_map(|node| match op(node) { + NonASAPOp::Aggregate { measures, .. } => Some(measures.clone()), + _ => None, + }) + .flatten() + .collect() } #[tokio::test] @@ -997,9 +963,9 @@ async fn derived_table_aggregate_over_aggregate_nests() { intents.iter().any(|i| matches!(i, AggIntent::Sum { .. })), "inner SUM survives, got {intents:?}" ); - // The whole nested DAG's output schema derives without error (positional - // resolution is total across the derived-table boundary). - assert_eq!(qe.output_schema().unwrap().fields.len(), 1); + // The whole nested tree's output schema derives (positional resolution + // is total across the derived-table boundary). + assert_eq!(qe.schema.fields.len(), 1); } #[tokio::test] @@ -1037,26 +1003,23 @@ async fn filter_over_derived_aggregate_resolves_alias_column() { assert!(all_intents(&qe) .iter() .any(|i| matches!(i, AggIntent::Sum { .. }))); - // Schema derivation is total across the boundary. - let _ = qe.output_schema().expect("nested schema derivation"); + // Schema derivation is total across the boundary: the root carries one. + assert_eq!(qe.schema.fields.len(), 2); } #[tokio::test] -async fn scalar_subquery_in_predicate_is_rejected() { - // A subquery-*valued* expression (`x > (SELECT …)`) needs a subquery node in - // the unresolved expression IR (and a correlated/uncorrelated decision); - // rejected cleanly until that lands. Derived tables in FROM (the common nesting - // shape) ARE supported — see the tests above. - let res = lower_sql( - "SELECT service FROM metrics WHERE bytes > (SELECT AVG(bytes) FROM metrics)", - &catalog(), - AccuracyTarget::Exact, - ) - .await; +async fn scalar_subquery_in_predicate_lowers_through_a_cross_join() { + let qe = + lower("SELECT service FROM metrics WHERE bytes > (SELECT AVG(bytes) FROM metrics)").await; + let filter = find_filter(&qe).unwrap(); + let NonASAPOp::Filter { pred, child } = op(filter) else { + panic!() + }; + assert!(matches!(op(child), NonASAPOp::Scan { .. })); assert!( - res.is_err(), - "scalar subquery in predicate should be rejected" + matches!(&pred.0,ScalarExpr::Compare { right,.. } if matches!(right.as_ref(),ScalarExpr::ScalarSubquery(_))) ); + qe.validate_structure().unwrap(); } #[tokio::test] @@ -1072,15 +1035,15 @@ async fn correlated_exists_lifts_its_correlation_into_the_join() { .await; let (kind, pred, left_len) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - let QueryExpr::Compare { left, right, .. } = pred else { + let ScalarExpr::Compare { left, right, .. } = pred else { panic!("expected the correlation as a comparison, got {pred:?}"); }; assert_eq!( **left, - QueryExpr::Column(left_len), + ScalarExpr::Column(left_len), "h.service (right side)" ); - assert_eq!(**right, QueryExpr::Column(1), "m.service (left side)"); + assert_eq!(**right, ScalarExpr::Column(1), "m.service (left side)"); } #[tokio::test] @@ -1099,24 +1062,62 @@ async fn an_uncorrelated_exists_is_an_unconditional_semi_join() { let qe = lower("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; let (kind, pred, _) = join_parts(&qe); assert_eq!(kind, &JoinKind::Semi); - assert_eq!(*pred, QueryExpr::Literal(ScalarValue::Boolean(true))); + assert_eq!(*pred, ScalarExpr::Literal(ScalarValue::Boolean(true))); } #[tokio::test] -async fn not_in_subquery_is_rejected_rather_than_mislowered_as_an_anti_join() { - // `NOT IN` is *not* an anti-join. Under three-valued logic a single NULL - // among the subquery's rows makes `c NOT IN (…)` UNKNOWN for every `c`, so - // the query returns nothing — while an anti-join returns every unmatched - // left row. Rejecting is the only correct option until the nullability is - // proven, and `NOT EXISTS` is the safe spelling. - let err = lower_sql( - "SELECT service FROM metrics WHERE service NOT IN (SELECT service FROM hosts)", - &catalog(), - AccuracyTarget::Exact, +async fn where_exists_resolves_to_a_semi_join_over_the_subquery() { + // The front end emits `Filter { Exists(s) }`; the resolved DAG is the + // `Semi` join with the subquery (a filtered `hosts` scan) on the right. + let qe = lower( + "SELECT service FROM metrics WHERE EXISTS (SELECT service FROM hosts WHERE region = 'eu')", ) - .await - .expect_err("NOT IN must not lower to an anti-join"); - assert!(format!("{err}").contains("NOT IN"), "got {err}"); + .await; + let NonASAPOp::Project { child, .. } = op(&qe) else { + panic!("expected the SELECT list as a Project, got {qe:?}"); + }; + let NonASAPOp::Join { + kind, + pred, + left, + right, + } = op(child) + else { + panic!("expected the Semi join directly under the Project, got {child:?}"); + }; + assert_eq!(*kind, JoinKind::Semi); + assert_eq!(pred.0, ScalarExpr::Literal(ScalarValue::Boolean(true))); + assert!( + matches!(op(left), NonASAPOp::Scan { .. }), + "left is metrics" + ); + let NonASAPOp::Project { child: scan, .. } = op(right) else { + panic!("expected the subquery's projection on the right, got {right:?}"); + }; + assert!( + matches!(op(scan), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1), + "the subquery's WHERE stays on its own Scan, got {scan:?}" + ); + assert_eq!( + child.schema.fields.len(), + 4, + "a semi join outputs the left's columns alone" + ); +} + +#[tokio::test] +async fn not_in_subquery_is_rejected_rather_than_mislowered_as_an_anti_join() { + let qe = + lower("SELECT service FROM metrics WHERE service NOT IN (SELECT service FROM hosts)").await; + let filter = find_filter(&qe).unwrap(); + let NonASAPOp::Filter { pred, .. } = op(filter) else { + panic!() + }; + assert!(matches!( + pred.0, + ScalarExpr::InSubquery { negated: true, .. } + )); + qe.validate_structure().unwrap(); } #[tokio::test] @@ -1132,6 +1133,177 @@ async fn a_correlated_in_subquery_is_rejected() { assert!(format!("{err}").contains("correlated IN"), "got {err}"); } +// ── Subquery-valued expressions at the `UnresolvedOp` level ───────────────── + +/// `SqlLowerer::lower` output, before `resolve_root`. +async fn lower_unresolved(sql: &str) -> UnresolvedOp { + let catalog = catalog(); + SqlLowerer::new(&catalog) + .lower(sql, &AccuracyTarget::Exact) + .await + .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) +} + +#[tokio::test] +async fn scalar_subquery_in_projection_lowers_to_a_scalar_subquery_item() { + // An uncorrelated `(SELECT max(v) FROM t2)` in the SELECT list is a + // `ScalarSubquery` projection item reading its own lowered plan; the + // cross-join rewrite is `canonicalize`'s job, not the front end's. + let tree = lower_unresolved("SELECT (SELECT max(latency) FROM metrics) FROM hosts").await; + let UnresolvedOp::Project { cols, child, .. } = &tree else { + panic!("expected the SELECT list as a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Scan { source: Source::Table { table_ref }, .. } + if table_ref == "hosts"), + "the outer relation stays the projection's child, got {child:?}" + ); + assert_eq!(cols.len(), 1); + let UnresolvedScalar::ScalarSubquery(sub) = &cols[0].expr else { + panic!("expected a ScalarSubquery item, got {:?}", cols[0].expr); + }; + let UnresolvedOp::Project { child: inner, .. } = sub.as_ref() else { + panic!("expected the subquery's own SELECT list, got {sub:?}"); + }; + assert!( + matches!(inner.as_ref(), UnresolvedOp::Aggregate { measures, .. } + if matches!(measures.as_slice(), [AggIntent::Max { .. }])), + "the subquery plan is lowered as a root of its own, got {inner:?}" + ); +} + +#[tokio::test] +async fn exists_and_in_subqueries_lower_to_scalar_filter_conjuncts() { + // The front end no longer builds the semi join itself: `EXISTS` / `IN + // (…)` are `Filter` predicates reading the subquery operator. + let tree = + lower_unresolved("SELECT service FROM metrics WHERE EXISTS (SELECT 1 FROM hosts)").await; + let UnresolvedOp::Project { child, .. } = &tree else { + panic!("expected a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } + if matches!(pred.0, UnresolvedScalar::Exists { negated: false, .. })), + "expected Filter {{ Exists }}, got {child:?}" + ); + + let tree = lower_unresolved( + "SELECT service FROM metrics WHERE service IN (SELECT service FROM hosts)", + ) + .await; + let UnresolvedOp::Project { child, .. } = &tree else { + panic!("expected a Project, got {tree:?}"); + }; + assert!( + matches!(child.as_ref(), UnresolvedOp::Filter { pred, .. } + if matches!(pred.0, UnresolvedScalar::InSubquery { negated: false, .. })), + "expected Filter {{ InSubquery }}, got {child:?}" + ); +} + +// ── `SELECT` without `FROM`, unary minus, SQL expression semantics ────────── + +#[tokio::test] +async fn select_without_from_projects_over_one_empty_row() { + // `SELECT 1` has no table: DataFusion's `EmptyRelation` is one empty + // input row, which the SELECT list projects a literal over. + let qe = lower("SELECT 1").await; + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert_eq!(cols.len(), 1); + assert_eq!(cols[0].expr, ScalarExpr::Literal(ScalarValue::Int64(1))); + let NonASAPOp::Values { rows, schema } = op(child) else { + panic!("expected Values under the Project, got {child:?}"); + }; + assert_eq!(rows, &vec![Vec::::new()], "one empty row"); + assert!(schema.fields.is_empty() && schema.closed); + assert_eq!(qe.schema.fields.len(), 1); + assert_eq!(qe.schema.fields[0].dtype, DataType::Int64); +} + +#[tokio::test] +async fn values_lowers_to_one_row_per_values_row() { + let qe = lower("SELECT * FROM (VALUES (1, 'a'), (2, 'b')) AS v(n, s)").await; + let values = OperatorNode::reachable(&qe) + .into_iter() + .find(|n| matches!(op(n), NonASAPOp::Values { .. })) + .expect("expected a Values node"); + let NonASAPOp::Values { rows, schema } = op(&values) else { + unreachable!() + }; + assert_eq!(rows.len(), 2); + assert_eq!( + rows[1], + vec![ + ScalarExpr::Literal(ScalarValue::Int64(2)), + ScalarExpr::Literal(ScalarValue::Utf8("b".into())), + ] + ); + assert_eq!(schema.fields.len(), 2); + assert_eq!(schema.fields[0].dtype, DataType::Int64); + assert_eq!(schema.fields[1].dtype, DataType::Utf8); + assert_eq!( + qe.schema + .fields + .iter() + .map(|f| f.name.as_str()) + .collect::>(), + ["n", "s"] + ); +} + +#[tokio::test] +async fn unary_minus_lowers_to_negative() { + // `-x` over a column is the `Negative` scalar (a negative *literal* is + // folded by DataFusion's planner before lowering). + let qe = lower("SELECT -latency FROM metrics").await; + let NonASAPOp::Project { cols, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert_eq!( + cols[0].expr, + ScalarExpr::Negative { + expr: Box::new(ScalarExpr::Column(2)), + semantics: ExprSemantics::Sql, + } + ); + assert_eq!(qe.schema.fields[0].dtype, DataType::Float64); +} + +#[tokio::test] +async fn sql_comparisons_and_arithmetic_carry_sql_semantics() { + let qe = lower("SELECT bytes * 8 FROM metrics WHERE latency > 1.5").await; + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { + panic!("expected Project at root, got {qe:?}"); + }; + assert!( + matches!( + &cols[0].expr, + ScalarExpr::Arithmetic { + semantics: ExprSemantics::Sql, + .. + } + ), + "got {:?}", + cols[0].expr + ); + let NonASAPOp::Scan { predicates, .. } = op(child) else { + panic!("expected the WHERE folded onto the Scan, got {child:?}"); + }; + assert!( + matches!( + &predicates[0].0, + ScalarExpr::Compare { + semantics: ExprSemantics::Sql, + .. + } + ), + "got {:?}", + predicates[0].0 + ); +} + // ── Issue #115: Quantile / Cardinality carry their input column ───────────── #[tokio::test] @@ -1281,20 +1453,20 @@ async fn time_bucketing_group_by_lowers_to_a_derived_key() { let qe = lower("SELECT date_trunc('minute', ts) AS m, SUM(bytes) FROM metrics GROUP BY m").await; let node = find_aggregate_node(&qe).expect("expected an Aggregate"); - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction, measures, child, .. - } = node + } = op(node) else { unreachable!() }; assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(op(child), NonASAPOp::Project { .. }), "expected a materializing Project beneath the Aggregate" ); - let schema = child.output_schema().expect("child schema"); + let schema = &child.schema; assert_eq!(reduction, &Reduction::by(vec![0])); assert!( schema.fields[0].name.contains("date_trunc"), @@ -1317,14 +1489,14 @@ async fn time_bucketing_keeps_the_scan_predicate() { WHERE bytes > 10 GROUP BY m", ) .await; - fn scan_has_predicate(qe: &QueryExpr) -> bool { - match qe { - QueryExpr::Scan { predicates, .. } => !predicates.is_empty(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => scan_has_predicate(child), + fn scan_has_predicate(node: &OperatorNode) -> bool { + match op(node) { + NonASAPOp::Scan { predicates, .. } => !predicates.is_empty(), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => scan_has_predicate(child), _ => false, } } @@ -1341,13 +1513,13 @@ async fn a_plain_group_by_inserts_no_projection() { "SELECT COUNT(*) FROM metrics", ] { let qe = lower(q).await; - let QueryExpr::Aggregate { child, .. } = - find_aggregate_node(&qe).expect("expected an Aggregate") + let NonASAPOp::Aggregate { child, .. } = + op(find_aggregate_node(&qe).expect("expected an Aggregate")) else { unreachable!() }; assert!( - !matches!(**child, QueryExpr::Project { .. }), + !matches!(op(child), NonASAPOp::Project { .. }), "{q} should not gain a projection" ); } @@ -1356,14 +1528,14 @@ async fn a_plain_group_by_inserts_no_projection() { #[tokio::test] async fn a_shared_expression_is_materialized_once() { let qe = lower("SELECT SUM(bytes * 2), MIN(bytes * 2) FROM metrics").await; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = find_aggregate_node(&qe).expect("expected an Aggregate") + } = op(find_aggregate_node(&qe).expect("expected an Aggregate")) else { unreachable!() }; assert_eq!( - child.output_schema().expect("child schema").fields.len(), + child.schema.fields.len(), 1, "the two reducers should share one derived column" ); @@ -1373,38 +1545,32 @@ async fn a_shared_expression_is_materialized_once() { // ── Issue #118: multi-level grouping expands into one Aggregate per level ─── /// The branches of the first `Concat` along the single-child spine. -fn merge_branches(qe: &QueryExpr) -> &Vec { - fn find(qe: &QueryExpr) -> Option<&Vec> { - match qe { - QueryExpr::Concat { children, .. } => Some(children), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => find(child), +fn merge_branches(node: &OperatorNode) -> &Vec> { + fn find(node: &OperatorNode) -> Option<&Vec>> { + match op(node) { + NonASAPOp::Concat { children, .. } => Some(children), + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => find(child), _ => None, } } - find(qe).expect("expected a Concat") + find(node).expect("expected a Concat") } /// `(group keys, column names)` of each merged grouping level. -fn grouping_levels(qe: &QueryExpr) -> Vec<(GroupKeys, Vec)> { - merge_branches(qe) +fn grouping_levels(node: &OperatorNode) -> Vec<(GroupKeys, Vec)> { + merge_branches(node) .iter() .map(|b| { - let QueryExpr::Project { child, .. } = b else { + let NonASAPOp::Project { child, .. } = op(b) else { panic!("expected a Project per level, got {b:?}"); }; - let QueryExpr::Aggregate { reduction, .. } = child.as_ref() else { + let NonASAPOp::Aggregate { reduction, .. } = op(child) else { panic!("expected an Aggregate under the Project, got {child:?}"); }; - let names = b - .output_schema() - .expect("level schema") - .fields - .iter() - .map(|c| c.name.clone()) - .collect(); + let names = b.schema.fields.iter().map(|c| c.name.clone()).collect(); (reduction.expect_reduce().clone(), names) }) .collect() @@ -1463,9 +1629,7 @@ async fn omitted_grouping_keys_become_typed_nulls() { } // The `()` level projects `service` as a Utf8 null, not a Float64 one. - let schema = merge_branches(&qe)[1] - .output_schema() - .expect("level schema"); + let schema = &merge_branches(&qe)[1].schema; assert_eq!(schema.fields[0].name, "service"); assert_eq!( schema.fields[0].dtype, @@ -1483,8 +1647,7 @@ async fn grouping_levels_are_union_compatible() { let shapes: Vec<_> = merge_branches(&qe) .iter() .map(|b| { - b.output_schema() - .expect("level schema") + b.schema .fields .iter() .map(|c| (c.name.clone(), c.dtype.clone())) @@ -1534,12 +1697,12 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { // #110's materializing Project sits beneath every level's Aggregate. let qe = lower("SELECT service, SUM(bytes * 8) FROM metrics GROUP BY ROLLUP(service)").await; for b in merge_branches(&qe) { - let QueryExpr::Project { child, .. } = b else { + let NonASAPOp::Project { child, .. } = op(b) else { panic!("expected a Project per level"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { measures, child, .. - } = child.as_ref() + } = op(child) else { panic!("expected an Aggregate"); }; @@ -1548,7 +1711,7 @@ async fn multi_level_grouping_composes_with_a_derived_reducer_argument() { [AggIntent::Sum { col: Some(_) }] )); assert!( - matches!(**child, QueryExpr::Project { .. }), + matches!(op(child), NonASAPOp::Project { .. }), "the derived-column projection should sit under each level" ); } @@ -1611,7 +1774,7 @@ async fn array_agg_is_deliberately_rejected() { // ── Issue #225: catalog-driven ClickHouse builtins (countIf, generalizing // uniqExact from #221) ─────────────────────────────────────────────────── -async fn lower_clickhouse(sql: &str) -> QueryExpr { +async fn lower_clickhouse(sql: &str) -> Rc { lower_sql_dialect( sql, &catalog(), @@ -1622,20 +1785,20 @@ async fn lower_clickhouse(sql: &str) -> QueryExpr { .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) } -fn temporal_aggregate(qe: &QueryExpr) -> (&AggIntent, std::time::Duration, &QueryExpr) { - match qe { - QueryExpr::Aggregate { +fn temporal_aggregate(node: &OperatorNode) -> (&AggIntent, std::time::Duration, &OperatorNode) { + match op(node) { + NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, child, .. } => { - let QueryExpr::TimeRange { range, child } = child.as_ref() else { + let NonASAPOp::TimeRange { range, child, .. } = op(child) else { panic!("temporal Aggregate must directly wrap TimeRange, got {child:?}"); }; (&measures[0], *range, child) } - QueryExpr::Project { child, .. } | QueryExpr::Filter { child, .. } => { + NonASAPOp::Project { child, .. } | NonASAPOp::Filter { child, .. } => { temporal_aggregate(child) } other => panic!("expected temporal Aggregate, got {other:?}"), @@ -1656,15 +1819,15 @@ async fn explicit_temporal_aggregates_share_promql_intents_and_timerange() { let (intent, range, child) = temporal_aggregate(&qe); assert_eq!(intent, &expected); assert_eq!(range, std::time::Duration::from_secs(300)); - assert!(matches!(child, QueryExpr::Project { child, .. } - if matches!(child.as_ref(), QueryExpr::Scan { predicates, .. } if predicates.len() == 1))); + assert!(matches!(op(child), NonASAPOp::Project { child, .. } + if matches!(op(child), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1))); - let QueryExpr::Project { cols, .. } = &qe else { + let NonASAPOp::Project { cols, .. } = op(&qe) else { panic!("SELECT list must remain a Project, got {qe:?}"); }; - assert!(matches!(cols[0].expr, QueryExpr::Column(2))); + assert!(matches!(cols[0].expr, ScalarExpr::Column(2))); assert_eq!(cols[1].alias.as_deref(), Some("v")); - assert!(matches!(cols[1].expr, QueryExpr::Column(1))); + assert!(matches!(cols[1].expr, ScalarExpr::Column(1))); } } @@ -1825,20 +1988,20 @@ async fn project_filter_and_outer_aggregate_preserve_temporal_child() { ) r WHERE v >= 0", ) .await; - let QueryExpr::Project { child, .. } = &qe else { + let NonASAPOp::Project { child, .. } = op(&qe) else { panic!("expected outer SELECT Project, got {qe:?}"); }; - let QueryExpr::Aggregate { + let NonASAPOp::Aggregate { reduction: Reduction::Reduce(_), measures, child, .. - } = child.as_ref() + } = op(child) else { panic!("expected outer Aggregate, got {child:?}"); }; assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); - let QueryExpr::Filter { child, .. } = child.as_ref() else { + let NonASAPOp::Filter { child, .. } = op(child) else { panic!("derived-table WHERE must remain above the inner query, got {child:?}"); }; let (intent, range, _) = temporal_aggregate(child); @@ -2002,13 +2165,13 @@ async fn lag_in_frame_lowers_to_its_own_kind_not_lag() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, args, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, args, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LagInFrame); assert_eq!( args, - &vec![QueryExpr::Column(3)], + &vec![ScalarExpr::Column(3)], "lagInFrame(bytes) → arg col 3" ); } @@ -2021,7 +2184,7 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { ) .await; let win = find_windowfunc(&qe).expect("expected a SQLWindowFunc node"); - let QueryExpr::SQLWindowFunc { func, .. } = win else { + let NonASAPOp::SQLWindowFunc { func, .. } = op(win) else { unreachable!(); }; assert_eq!(*func, WindowFuncKind::LeadInFrame); @@ -2034,13 +2197,13 @@ async fn lead_in_frame_lowers_to_its_own_kind_not_lead() { async fn now_in_predicate_lowers_to_current_timestamp() { // SELECT * folds WHERE onto Scan.predicates (no explicit Filter node). let qe = lower("SELECT * FROM metrics WHERE ts < NOW()").await; - let QueryExpr::Scan { predicates, .. } = &qe else { + let NonASAPOp::Scan { predicates, .. } = op(&qe) else { panic!("expected Scan at root, got {qe:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } + if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), "NOW() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2051,13 +2214,13 @@ async fn now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { let qe = lower_clickhouse("SELECT * FROM metrics WHERE ts < now()").await; - let QueryExpr::Scan { predicates, .. } = &qe else { + let NonASAPOp::Scan { predicates, .. } = op(&qe) else { panic!("expected Scan at root, got {qe:?}"); }; assert_eq!(predicates.len(), 1); assert!( - matches!(predicates[0].0.as_ref(), QueryExpr::Compare { right, .. } - if matches!(right.as_ref(), QueryExpr::CurrentTimestamp)), + matches!(&predicates[0].0, ScalarExpr::Compare { right, .. } + if matches!(right.as_ref(), ScalarExpr::Cast { expr, to: DataType::Timestamp, .. } if matches!(expr.as_ref(), ScalarExpr::CurrentTimestamp))), "now() must lower to CurrentTimestamp, got {:?}", predicates[0].0 ); @@ -2066,12 +2229,16 @@ async fn clickhouse_now_in_predicate_lowers_to_current_timestamp() { #[tokio::test] async fn current_timestamp_lowers_to_typed_current_timestamp_leaf() { let qe = lower("SELECT CURRENT_TIMESTAMP FROM metrics").await; - let QueryExpr::Project { cols, .. } = &qe else { + let NonASAPOp::Project { cols, child, .. } = op(&qe) else { panic!("expected Project at root, got {qe:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); - let schema = cols[0].expr.output_schema().expect("timestamp schema"); - assert_eq!(schema.fields[0].dtype, DataType::Timestamp); + assert!(matches!(&cols[0].expr, ScalarExpr::CurrentTimestamp)); + let (dtype, _) = cols[0] + .expr + .scalar_type(&child.schema) + .expect("timestamp type"); + assert_eq!(dtype, DataType::Timestamp); + assert_eq!(qe.schema.fields[0].dtype, DataType::Timestamp); } // A `count` over a non-null input is a plain row count; over a nullable @@ -2114,10 +2281,7 @@ async fn count_null_semantics_become_a_measure_filter() { aggregate_filters(&qe) ); }; - assert!( - matches!(cond.as_ref(), QueryExpr::IsNotNull(_)), - "{sql}: {cond:?}" - ); + assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{sql}: {cond:?}"); } // Only the second measure is filtered. let qe = lower_sql( @@ -2164,7 +2328,7 @@ async fn grouped_map_column_preserves_map_type() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().fields[0].dtype, map); + assert_eq!(query.schema.fields[0].dtype, map); } #[tokio::test] @@ -2202,10 +2366,7 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { .await .unwrap(); assert_eq!(function, operator, "{call}"); - assert_eq!( - function.output_schema().unwrap(), - operator.output_schema().unwrap() - ); + assert_eq!(function.schema, operator.schema); } let nullable = lower_sql_dialect( "SELECT modulo(n, 3) AS value FROM numbers", @@ -2215,8 +2376,8 @@ async fn clickhouse_modulo_uses_native_arithmetic_types_and_nullability() { ) .await .unwrap() - .output_schema() - .unwrap(); + .schema + .clone(); assert_eq!(nullable.fields[0].dtype, DataType::Int64); assert!(nullable.fields[0].nullable); } @@ -2255,7 +2416,7 @@ async fn original_o11y_map_queries_lower_with_typed_results() { ) .await .unwrap_or_else(|e| panic!("{sql}: {e}")); - let schema = query.output_schema().unwrap(); + let schema = &query.schema; assert!( schema .fields @@ -2287,7 +2448,7 @@ async fn clickhouse_modulo_preserves_projection_names_and_outer_references() { ) .await .unwrap(); - assert_eq!(query.output_schema().unwrap().fields[0].name, name); + assert_eq!(query.schema.fields[0].name, name); } } @@ -2317,7 +2478,7 @@ async fn clickhouse_map_access_keeps_generated_names_and_rejects_variant_coercio ) .await .unwrap(); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[0].name, "arrayElement(labels, 'job')"); assert_eq!(output.fields[0].dtype, DataType::Utf8); assert!(!output.fields[0].nullable); @@ -2369,7 +2530,7 @@ async fn arg_selector_result_schema_tracks_selected_argument() { ) .await .unwrap(); - let schema = query.output_schema().unwrap(); + let schema = &query.schema; assert_eq!(schema.fields[0].dtype, dtype); assert_eq!(schema.fields[0].nullable, nullable); } @@ -2403,7 +2564,7 @@ async fn clickhouse_list_element_uses_canonical_typed_access() { ) .await .unwrap(); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[0].dtype, DataType::Int64); assert_eq!(output.fields[0].nullable, nullable); let serialized = serde_json::to_string(&query).unwrap(); @@ -2462,7 +2623,7 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { ) .await .unwrap(); - let output = query.output_schema().unwrap(); + let output = &query.schema; assert_eq!(output.fields[0].dtype, dtype); assert_eq!(output.fields[0].nullable, nullable); assert!(serde_json::to_string(&query) @@ -2489,7 +2650,7 @@ async fn clickhouse_tuple_element_preserves_declared_field_metadata() { #[tokio::test] async fn corr_result_is_nullable_float() { let query = lower("SELECT corr(latency, bytes) AS correlation FROM metrics").await; - let schema = query.output_schema().unwrap(); + let schema = &query.schema; assert_eq!(schema.fields[0].name, "correlation"); assert_eq!(schema.fields[0].dtype, DataType::Float64); assert!(schema.fields[0].nullable); @@ -2513,8 +2674,8 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = - find_aggregate_node(&composite).expect("expected an Aggregate") + let NonASAPOp::Aggregate { measures, .. } = + op(find_aggregate_node(&composite).expect("expected an Aggregate")) else { unreachable!() }; @@ -2530,8 +2691,8 @@ async fn composite_distinct_counts_tuples() { ) .await .unwrap(); - let QueryExpr::Aggregate { measures, .. } = - find_aggregate_node(&single).expect("expected an Aggregate") + let NonASAPOp::Aggregate { measures, .. } = + op(find_aggregate_node(&single).expect("expected an Aggregate")) else { unreachable!() }; @@ -2589,8 +2750,10 @@ async fn distinct_with_derived_sibling() { // ── Issue #466: per-measure FILTER predicates ───────────────────────────────── /// The first `Aggregate`'s `filters`, positional against its child. -fn aggregate_filters(qe: &QueryExpr) -> &[Option] { - let Some(QueryExpr::Aggregate { filters, .. }) = find_aggregate_node(qe) else { +fn aggregate_filters(qe: &OperatorNode) -> &[Option] { + let Some(NonASAPOp::Aggregate { filters, .. }) = + find_aggregate_node(qe).map(|n| n.expect_non_asap()) + else { panic!("expected an Aggregate, got {qe:?}"); }; filters @@ -2620,15 +2783,17 @@ async fn conditional_count_lowers_to_a_filtered_measure() { panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); }; assert!( - matches!(cond.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Gt, .. } - if matches!(left.as_ref(), QueryExpr::Column(2))), + matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Gt, .. } + if matches!(left.as_ref(), ScalarExpr::Column(2))), "latency > 1.0 against the scan, got {cond:?}" ); - let Some(QueryExpr::Aggregate { child, .. }) = find_aggregate_node(&qe) else { + let Some(NonASAPOp::Aggregate { child, .. }) = + find_aggregate_node(&qe).map(|n| n.expect_non_asap()) + else { unreachable!() }; assert!( - matches!(child.as_ref(), QueryExpr::Scan { .. }), + matches!(child.expect_non_asap(), NonASAPOp::Scan { .. }), "{child:?}" ); } @@ -2642,9 +2807,9 @@ async fn filter_clause_lowers_to_a_measure_filter() { panic!("expected [Some, None], got {:?}", aggregate_filters(&qe)); }; assert!( - matches!(cond.as_ref(), QueryExpr::Compare { left, op: CompareOpKind::Eq, right } - if matches!(left.as_ref(), QueryExpr::Column(1)) - && matches!(right.as_ref(), QueryExpr::Literal(ScalarValue::Utf8(s)) if s == "a")), + matches!(cond, ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, .. } + if matches!(left.as_ref(), ScalarExpr::Column(1)) + && matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(s)) if s == "a")), "{cond:?}" ); } @@ -2658,7 +2823,7 @@ async fn count_of_a_nullable_expression_filters_nulls() { let [Some(Predicate(cond))] = aggregate_filters(&qe) else { panic!("expected [Some], got {:?}", aggregate_filters(&qe)); }; - assert!(matches!(cond.as_ref(), QueryExpr::IsNotNull(_)), "{cond:?}"); + assert!(matches!(cond, ScalarExpr::IsNotNull(_)), "{cond:?}"); assert!( matches!( find_aggregate(&qe).unwrap().1.as_slice(), @@ -2673,23 +2838,25 @@ async fn count_of_a_nullable_expression_filters_nulls() { #[tokio::test] async fn measure_filter_columns_survive_a_derived_column_projection() { let qe = lower("SELECT sum(bytes * 2) FILTER (WHERE latency > 1.0) FROM metrics").await; - let Some(QueryExpr::Aggregate { child, .. }) = find_aggregate_node(&qe) else { + let Some(NonASAPOp::Aggregate { child, .. }) = + find_aggregate_node(&qe).map(|n| n.expect_non_asap()) + else { unreachable!() }; assert!( - matches!(child.as_ref(), QueryExpr::Project { .. }), + matches!(child.expect_non_asap(), NonASAPOp::Project { .. }), "{child:?}" ); let [Some(Predicate(cond))] = aggregate_filters(&qe) else { panic!("expected [Some], got {:?}", aggregate_filters(&qe)); }; - let QueryExpr::Compare { left, .. } = cond.as_ref() else { + let ScalarExpr::Compare { left, .. } = cond else { panic!("{cond:?}"); }; - let QueryExpr::Column(id) = left.as_ref() else { + let ScalarExpr::Column(id) = left.as_ref() else { panic!("{left:?}"); }; - assert_eq!(child.output_schema().unwrap().fields[*id].name, "latency"); + assert_eq!(child.schema.fields[*id].name, "latency"); } // `GROUP BY ROLLUP` fans one measure list out into one `Aggregate` per level; diff --git a/crates/frontend-sql/tests/temporal_types.rs b/crates/frontend-sql/tests/temporal_types.rs index a6b4f1187..fb9f97607 100644 --- a/crates/frontend-sql/tests/temporal_types.rs +++ b/crates/frontend-sql/tests/temporal_types.rs @@ -1,8 +1,6 @@ use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_types::{ - pre_asap::schema::{DataType, Field, Schema}, - types::AccuracyTarget, -}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::types::AccuracyTarget; fn catalog() -> SqlCatalog { SqlCatalog::new().with_table( "t", @@ -38,10 +36,7 @@ async fn date_shifts_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Date - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Date); } } // Interval literals and explicit interval casts must both cross the Arrow bridge. @@ -54,10 +49,7 @@ async fn interval_cast_lowers_like_interval_literal() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Interval - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Interval); } } @@ -90,11 +82,7 @@ async fn negative_intervals_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Interval, - "{query}" - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Interval, "{query}"); } } @@ -108,9 +96,6 @@ async fn sql_date_literals_keep_their_type() { let node = lower_sql(query, &catalog(), AccuracyTarget::Exact) .await .unwrap(); - assert_eq!( - node.output_schema().unwrap().fields[0].dtype, - DataType::Date - ); + assert_eq!(node.schema.fields[0].dtype, DataType::Date); } } diff --git a/crates/integration-tests/Cargo.toml b/crates/integration-tests/Cargo.toml index 5a5de9902..0d2f938bf 100644 --- a/crates/integration-tests/Cargo.toml +++ b/crates/integration-tests/Cargo.toml @@ -7,12 +7,15 @@ edition = "2021" asap-types = { path = "../types" } asap-frontend-promql = { path = "../frontend-promql" } asap-frontend-sql = { path = "../frontend-sql" } -asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-plan-selection = { path = "../plan-selection" } +asap-logical-optimizer = { path = "../logical-optimizer" } +asap-physical-optimizer = { path = "../physical-optimizer" } [dev-dependencies] +asap-planner = { path = "../planner" } asap_sketchlib = { workspace = true } serde_json = "1" tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } -asap-physical-operators = { path = "../asap-physical-operators" } +asap-executor = { path = "../executor" } futures = "0.3" diff --git a/crates/integration-tests/src/lib.rs b/crates/integration-tests/src/lib.rs index be8e259bc..5435730f6 100644 --- a/crates/integration-tests/src/lib.rs +++ b/crates/integration-tests/src/lib.rs @@ -12,21 +12,34 @@ //! here derives or computes expected outputs. pub mod fixtures { - use asap_frontend_promql::lower_promql_workload; - use asap_types::pre_asap::schema::{DataType, Field, Schema}; - use asap_types::pre_asap::QueryExpr; + + use asap_types::ir::schema::{DataType, Field, Schema}; + use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; + use std::rc::Rc; /// Lower one query through the plan-ready workload API using the test /// suite's declared one-second source cadence. pub fn lower_promql( query: &str, accuracy: AccuracyTarget, - ) -> Result { + ) -> Result, asap_frontend_promql::PromqlError> { + match lower_promql_root(query, accuracy)? { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + _ => Err(asap_frontend_promql::PromqlError::UnsupportedFeature( + "expected vector root".into(), + )), + } + } + + pub fn lower_promql_root( + query: &str, + accuracy: AccuracyTarget, + ) -> Result { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -51,7 +64,7 @@ pub mod fixtures { ..Default::default() }), }; - let mut lowered = lower_promql_workload(&workload, 0)?; + let mut lowered = asap_frontend_promql::lower_promql_query_workload(&workload, 0)?; Ok(lowered.remove(0)) } @@ -83,3 +96,41 @@ pub mod fixtures { } } } + +/// Timing and export helpers for post-ASAP plans. +pub mod post_asap { + use asap_types::ir::export::{compile_physical_asap_dag, PhysicalASAPDAG}; + use asap_types::ir::{ + apply_materialization_timings, MaterializationAssignment, OperatorNode, TimingMemo, + }; + use std::rc::Rc; + + /// Time `root` under the default assignment (every summary computed at + /// query time). Returns the timed copy; read `node.timing` on it. + pub fn timed(root: &Rc) -> Rc { + timed_with(root, &MaterializationAssignment::all_query_time()) + } + + /// Time `root` with every summary maintained at ingestion time. + pub fn maintained(root: &Rc) -> Rc { + timed_with(root, &MaterializationAssignment::all_ingestion_time()) + } + + fn timed_with( + root: &Rc, + assignment: &MaterializationAssignment, + ) -> Rc { + apply_materialization_timings(root, assignment, &mut TimingMemo::new()) + .expect("materialization timing failed") + } + + /// Time `root` (default assignment), then export the physical DAG. + pub fn post_asap_dag(root: &Rc) -> PhysicalASAPDAG { + compile_physical_asap_dag(&timed(root)).expect("post-ASAP DAG export failed") + } + + /// Time `root` with every summary maintained, then export the physical DAG. + pub fn maintained_post_asap_dag(root: &Rc) -> PhysicalASAPDAG { + compile_physical_asap_dag(&maintained(root)).expect("post-ASAP DAG export failed") + } +} diff --git a/crates/integration-tests/tests/aggregate.rs b/crates/integration-tests/tests/aggregate.rs index 051eeb6b7..7010d0485 100644 --- a/crates/integration-tests/tests/aggregate.rs +++ b/crates/integration-tests/tests/aggregate.rs @@ -1,47 +1,54 @@ -//! `QueryExpr::Aggregate` — cross-series aggregation tests. +//! `NonASAPOp::Aggregate` — cross-series aggregation tests. //! //! topk/bottomk are omitted — dispatch is deferred. //! -//! Cross-series aggregates lower to a single `Aggregate` node with no -//! `TimeRange` child (range functions use `TimeRange` — see `time_range.rs`). -//! Group keys land on `Aggregate.by` as positional `ColumnId`s. -//! Single-stat PromQL aggregates always get `output_names: [""]` (no alias) -//! and `having: None`. +//! Cross-series aggregates lower to a single `Aggregate` node over the +//! instant-selector `TimeRange` (range functions use a `Range` selector — +//! see `time_range.rs`). Group keys land on `Aggregate.reduction` as +//! positional `ColumnId`s. Single-stat PromQL aggregates always get +//! `output_names: [""]` (no alias) and `having: None`. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::operator::{AggIntent, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(child), + kind: TimeRangeKind::Instant, + child, }), - } + }) } // #5 — sum with no group keys @@ -169,7 +176,7 @@ fn q_stdvar_no_group() { ); } -// #10 — cross-series quantile; no TimeRange node (no range window) +// #10 — cross-series quantile; instant selector, no range window #[test] fn q10_quantile_cross_series() { assert_eq!( diff --git a/crates/integration-tests/tests/binary_op.rs b/crates/integration-tests/tests/binary_op.rs index 35f1c632d..7bd267e1a 100644 --- a/crates/integration-tests/tests/binary_op.rs +++ b/crates/integration-tests/tests/binary_op.rs @@ -1,92 +1,122 @@ -//! `QueryExpr::BinaryOp` — arithmetic, comparison, and vector-match tests. +//! `NonASAPOp::BinaryOp` — arithmetic, comparison, and vector-match tests. //! //! Each side of a `BinaryOp` is bound independently by the SchemaResolver, so each //! gets its own scan schema derived from the labels it references. -//! `VectorMatch` labels (e.g. `on(job)`) are carried as strings on the node -//! and are NOT resolved to column ids — the SchemaResolver does not see them. +//! `VectorMatch` labels (e.g. `on(job)`) are carried as strings on the +//! operator and are NOT resolved to column ids — the SchemaResolver does not +//! see them. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, GroupSide, QueryExpr, Reduction, - Source, VectorGrouping, VectorMatch, VectorMatchKind, +use asap_types::ir::operator::{ + AggIntent, BinaryOpKind, GroupSide, Reduction, Source, VectorGrouping, VectorMatch, + VectorMatchKind, }; +use asap_types::ir::scalar::{ArithmeticOpKind, CompareOpKind}; +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::TimeRange { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(source_scan(metric, labels)), - } + kind: TimeRangeKind::Instant, + child: source_scan(metric, labels), + }) } -fn source_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn source_scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn rate_agg(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn rate_agg(metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Rate], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(source_scan(metric, &[])), + kind: TimeRangeKind::Range, + child: source_scan(metric, &[]), }), - } + }) } -fn sum_by_job(metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn sum_by_job(metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(scan(metric, &["job"])), - } + child: scan(metric, &["job"]), + }) +} + +/// A PromQL binary operator: no checked-division flags, no `bool` modifier. +fn binary( + kind: BinaryOpKind, + vector_match: Option, + lhs: Rc, + rhs: Rc, +) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + }, + return_bool: false, + lhs, + rhs, + }) } // #18 — arithmetic binary op between two bare scans; no vector match #[test] fn q18_div_bare_scans() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + scan("http_requests_total", &[]), + scan("http_requests_total", &[]), + ); assert_eq!(lower("http_requests_total / http_requests_total"), expected); } // #19 — add with on(job) vector match; match labels are strings, not column ids #[test] fn q19_add_with_on_match() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: Some(VectorMatch { + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: None, }), - }; + scan("http_requests_total", &[]), + scan("http_requests_total", &[]), + ); assert_eq!( lower("http_requests_total + on(job) http_requests_total"), expected @@ -96,12 +126,12 @@ fn q19_add_with_on_match() { // #20 — divide two rate aggregates over different metrics #[test] fn q20_div_two_rates() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(rate_agg("http_requests_total")), - rhs: Rc::new(rate_agg("http_errors_total")), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + rate_agg("http_requests_total"), + rate_agg("http_errors_total"), + ); assert_eq!( lower("rate(http_requests_total[5m]) / rate(http_errors_total[5m])"), expected, @@ -113,12 +143,12 @@ fn q20_div_two_rates() { fn q_gt_comparison() { assert_eq!( lower("http_requests_total > http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Gt), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -126,12 +156,12 @@ fn q_gt_comparison() { fn q_lt_comparison() { assert_eq!( lower("http_requests_total < http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Lt), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Lt), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -139,12 +169,12 @@ fn q_lt_comparison() { fn q_ge_comparison() { assert_eq!( lower("http_requests_total >= http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Ge), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Ge), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -152,12 +182,12 @@ fn q_ge_comparison() { fn q_le_comparison() { assert_eq!( lower("http_requests_total <= http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Le), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: None, - } + binary( + BinaryOpKind::Compare(CompareOpKind::Le), + None, + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -166,16 +196,16 @@ fn q_le_comparison() { fn q_add_with_ignoring() { assert_eq!( lower("http_requests_total + ignoring(job) http_errors_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("http_errors_total", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + Some(VectorMatch { kind: VectorMatchKind::Ignoring, labels: vec!["job".into()], grouping: None, }), - } + scan("http_requests_total", &[]), + scan("http_errors_total", &[]), + ) ); } @@ -184,11 +214,9 @@ fn q_add_with_ignoring() { fn q_mul_group_left() { assert_eq!( lower("http_requests_total * on(job) group_left() node_info"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("http_requests_total", &[])), - rhs: Rc::new(scan("node_info", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: Some(VectorGrouping { @@ -196,7 +224,9 @@ fn q_mul_group_left() { labels: vec![], }), }), - } + scan("http_requests_total", &[]), + scan("node_info", &[]), + ) ); } @@ -205,11 +235,9 @@ fn q_mul_group_left() { fn q_mul_group_right() { assert_eq!( lower("node_info * on(job) group_right() http_requests_total"), - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("node_info", &[])), - rhs: Rc::new(scan("http_requests_total", &[])), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + Some(VectorMatch { kind: VectorMatchKind::On, labels: vec!["job".into()], grouping: Some(VectorGrouping { @@ -217,7 +245,9 @@ fn q_mul_group_right() { labels: vec![], }), }), - } + scan("node_info", &[]), + scan("http_requests_total", &[]), + ) ); } @@ -225,12 +255,12 @@ fn q_mul_group_right() { // each side: Aggregate{Sum, by=[2]} over Scan([ts, value, job]) #[test] fn q21_div_two_sum_by_job() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(sum_by_job("http_requests_total")), - rhs: Rc::new(sum_by_job("http_errors_total")), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + sum_by_job("http_requests_total"), + sum_by_job("http_errors_total"), + ); assert_eq!( lower("sum by (job) (http_requests_total) / sum by (job) (http_errors_total)"), expected, @@ -238,34 +268,29 @@ fn q21_div_two_sum_by_job() { } // #36 — unary negation lowers as `expr * -1`: a Mul BinaryOp of the vector -// against PromqlScalarBridge(-1), no vector match. The vector side keeps its schema. +// against a `ScalarExpr(-1)` leaf, no vector match. The vector side keeps +// its schema. #[test] fn q36_unary_negation_is_multiply_by_minus_one() { - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("some_metric", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), - vector_match: None, + let root = lower("-some_metric"); + let NonASAPOp::Project { cols, child, .. } = root.expect_non_asap() else { + panic!() }; - assert_eq!(lower("-some_metric"), expected); + assert!(child.schema.has_promql_series_identity()); + assert!(matches!(&cols[1].expr, ScalarExpr::Negative { .. })); } // #36 — negation nested inside an aggregate argument (issue #27 nesting): // `sum(-m)` → Aggregate{Sum} over the `m * -1` BinaryOp. #[test] fn q36_sum_of_negation_nests() { - let expected = QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec!["".into()], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), - lhs: Rc::new(scan("node_cpu_seconds_total", &[])), - rhs: Rc::new(QueryExpr::promql_scalar(-1.0)), - vector_match: None, - }), + let root = lower("sum(-node_cpu_seconds_total)"); + let NonASAPOp::Aggregate { + child, measures, .. + } = root.expect_non_asap() + else { + panic!() }; - assert_eq!(lower("sum(-node_cpu_seconds_total)"), expected); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); } diff --git a/crates/integration-tests/tests/cse.rs b/crates/integration-tests/tests/cse.rs index bb11eee2d..e1b35029c 100644 --- a/crates/integration-tests/tests/cse.rs +++ b/crates/integration-tests/tests/cse.rs @@ -2,38 +2,38 @@ //! #223). //! //! Drives the full staged pipeline this issue lands: two independently -//! lowered `QueryExpr` DAGs → `share_common_sub_dags` (stage 1, -//! `asap-types::pre_asap::cse`, run internally by `search_workload`) → -//! `search_workload` (stage 2, `asap-aware-mapping`) — and asserts the +//! lowered `OperatorNode` DAGs → `share_common_sub_dags` (stage 1, +//! `asap-types::ir::cse`, run internally by `search_workload`) → +//! `search_workload` (stage 2, `asap-logical-optimizer`) — and asserts the //! sharing that stage 1 decides survives into stage 2's discovered //! `CandidateLogicalASAPDAGs` as one genuinely shared `TargetSubDAGCandidates`, not just one shared -//! `Rc`. This is the "real caller" the issue's landing plan +//! `Rc`. This is the "real caller" the issue's landing plan //! requires before `share_common_sub_dags` is allowed to exist at all (its -//! predecessor, `asap-plan::cse::dedupe_subtrees`, was deleted in #192 for +//! predecessor, `asap-plan::cse::dedupe_sub-DAGs`, was deleted in #192 for //! being unwired dead code). //! //! Committing to one final, physically-materialized answer for a whole //! workload (the former `implement_workload`/`implement_workload_with`, //! which this test file used to drive instead of `search_workload`) is out -//! of `asap-aware-mapping`'s scope — see that crate's `lib.rs` `## Status` +//! of `asap-logical-optimizer`'s scope — see that crate's `lib.rs` `## Status` //! section — so these tests assert on the discovered `CandidateLogicalASAPDAGs` shape -//! directly, the same way `asap-aware-mapping::replacement`'s own +//! directly, the same way `asap-logical-optimizer::pass1::replacement`'s own //! `shared_aggregate_across_two_roots_gets_both_strategies_candidates` test //! does, just exercised through the crate's public API from this external //! integration-test crate. use std::rc::Rc; -use asap_aware_mapping::{search_workload, Replacement}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_logical_optimizer::{is_logical_rewrite, search_workload, Replacement}; +use asap_types::ir::NonASAPOp; use asap_types::types::AccuracyTarget; /// Two workload entries that happen to submit the exact same query (a /// realistic case — two dashboards, or a query fired both standalone and as -/// part of a larger batch) collapse onto one shared `Rc` after +/// part of a larger batch) collapse onto one shared `Rc` after /// `search_workload`'s internal `share_common_sub_dags` pass, and onto one -/// genuinely-shared [`TargetSubDAGCandidates`](asap_aware_mapping::TargetSubDAGCandidates) — carrying +/// genuinely-shared [`TargetSubDAGCandidates`](asap_logical_optimizer::TargetSubDAGCandidates) — carrying /// every candidate discovered for it exactly once, not once per root — no /// second structural-equality pass at the post-ASAP layer needed for this /// kind of sharing. @@ -41,7 +41,7 @@ use asap_types::types::AccuracyTarget; fn duplicate_workload_queries_collapse_onto_one_memo_group() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema carries // a provable unique key — the legality gate `share_common_sub_dags` - // enforces (see `asap-types::pre_asap::cse`'s module doc) — and its + // enforces (see `asap-types::ir::cse`'s module doc) — and its // `ExactAggregate(Sum)` realization is deterministic regardless of the // accuracy target, so this pins the sharing mechanism itself rather than // any one particular summary-family choice. @@ -57,20 +57,20 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { "fixture sanity: identical query text lowers identically" ); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); // roots[0] and roots[1] must have merged onto the same Rc — the // `share_common_sub_dags` pass `search_workload` runs internally. assert!( Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1), - "search_workload must collapse the two identical roots onto one Rc" + "search_workload must collapse the two identical roots onto one Rc" ); // The single shared root is one discovered TargetSubDAG, holding one - // TargetSubDAGCandidates with consumer_count 2 — SketchAlgorithmStrategy's one + // TargetSubDAGCandidates with consumer_count 2 — ASAPStrategies's one // ExactAggregate candidate *and* SharedSubDAGStrategy's share-vs- // recompute pair, exactly as `shared_aggregate_across_two_roots_gets_both_strategies_candidates` - // (asap-aware-mapping::replacement's own equivalent, internal test) + // (asap-logical-optimizer::pass1::replacement's own equivalent, internal test) // pins for the same fixture shape. let group = space .candidates_for_target(&space.roots[0].1) @@ -79,19 +79,21 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { assert_eq!( group.candidates.len(), 3, - "1 ExactAggregate Summary + 2 Rewrite (share/recompute): {:?}", + "1 ExactAggregate summary + 2 logical rewrites (share/recompute): {:?}", group.candidates ); + // A bound summary is a `Subtree` with an ASAP operator in it; a logical + // rewrite is a `Subtree` with none (`is_logical_rewrite`). let summary_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Summary(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap())) .count(); let rewrite_count = group .candidates .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if is_logical_rewrite(n))) .count(); assert_eq!(summary_count, 1); assert_eq!(rewrite_count, 2); @@ -100,12 +102,14 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // "false-positive dedup" failure mode `is_duplicate_rewrite` exists to // prevent): one shares the group's own target `Rc`, the other is a // structurally-identical but independently-built `Rc`. - let one_is_the_target = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target)), - ); - let one_is_not = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if !Rc::ptr_eq(rc, &group.target)), - ); + let one_is_the_target = group.candidates.iter().any(|c| { + matches!(&c.replacement, Replacement::SubDAG(rc) + if is_logical_rewrite(rc) && Rc::ptr_eq(rc, &group.target)) + }); + let one_is_not = group.candidates.iter().any(|c| { + matches!(&c.replacement, Replacement::SubDAG(rc) + if is_logical_rewrite(rc) && !Rc::ptr_eq(rc, &group.target)) + }); assert!(one_is_the_target && one_is_not); } @@ -121,7 +125,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { .expect("query b failed to lower"); assert_ne!(a, b, "fixture sanity: the two queries differ"); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); assert!(!Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let group_a = space @@ -140,7 +144,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { /// Single-query CSE (a repeated sub-expression within one query) also /// survives through `search_workload`: the two grouped-`Aggregate` branches -/// of a `BinaryOp` collapse to one shared `Rc` in the internal +/// of a `BinaryOp` collapse to one shared `Rc` in the internal /// `share_common_sub_dags` pass, and to one shared `TargetSubDAGCandidates` (with /// `consumer_count == 2`, one per branch) here. #[test] @@ -148,16 +152,16 @@ fn single_query_repeated_subexpression_shares_one_memo_group() { let query = "sum by (job) (http_requests_total) / sum by (job) (http_requests_total)"; let expr = lower_promql(query, AccuracyTarget::Exact).expect("query failed to lower"); - let space = search_workload(vec![("q", Rc::new(expr))]); + let space = search_workload(vec![("q", expr)]); let [(_, root)] = space.roots.as_slice() else { panic!("expected 1 root"); }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = root.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { panic!("expected a BinaryOp root, got {root:?}"); }; assert!( Rc::ptr_eq(lhs, rhs), - "the two identical sum-by-job branches must collapse onto one Rc" + "the two identical sum-by-job branches must collapse onto one Rc" ); let group = space diff --git a/crates/integration-tests/tests/exact_composition.rs b/crates/integration-tests/tests/exact_composition.rs index 917ef7b3a..fbe65a0dd 100644 --- a/crates/integration-tests/tests/exact_composition.rs +++ b/crates/integration-tests/tests/exact_composition.rs @@ -1,42 +1,50 @@ //! Issue #171 — composing exact operators with summary plans across -//! explicit update/readout boundaries, end to end through -//! `search_workload_with` → `CandidateLogicalASAPDAGs::global_selection` → +//! explicit update/evaluation boundaries, end to end through +//! `search_workload_with` → `candidate_selection::global_selection` → //! `GlobalSelection::assemble_selected_dag` → `dag_export`. //! //! Covers the issue's integration matrix: both nesting directions, grouped //! fine-to-coarse and identity folds, one inner summary shared by several -//! queries, phase-aware summary construction, a runtime without -//! the capability, a cost model without statistics, and pre/post-ASAP +//! queries, phase-aware summary construction, unknown runtime support, +//! a cost model without statistics, and pre/post-ASAP //! schemas plus shared `Rc` identity — along with pins for every //! already-supported exact-accumulator nesting. use std::rc::Rc; -use asap_aware_mapping::cost_model::{ - CostProvenance, CostUnit, ExactCompositionCostInputs, ExactCompositionCostRequest, - ValueOperationCapabilities, -}; -use asap_aware_mapping::replacement::{ - default_strategies_with, search_workload_with, Replacement, ReplacementProvenance, - ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, +use asap_integration_tests::fixtures::lower_promql; +use asap_integration_tests::post_asap::{maintained, post_asap_dag, timed}; +use asap_logical_optimizer::pass1::exact_composition::ExactOperation; +use asap_logical_optimizer::pass1::replacement::{ + default_strategies, search_workload_with, ASAPStrategies, Replacement, ReplacementProvenance, + ReplacementStrategy, TargetSubDAG, }; -use asap_aware_mapping::{ - CostModel, DefaultCostModel, EvaluationRate, ExplanationKind, OperationPlacement, +use asap_logical_optimizer::{ExplanationKind, OperationPlacement}; +use asap_plan_selection::candidate_selection::{global_selection, runtime_support_evidence}; +use asap_plan_selection::cost::cost_model::{ + CostProvenance, CostUnit, ExactCompositionCostInputs, ExactCompositionCostRequest, }; -use asap_integration_tests::fixtures::lower_promql; +use asap_plan_selection::{CostModel, DefaultCostModel, EvaluationRate}; use asap_types::dag_export; -use asap_types::post_asap::{ - validate_execution_data_states, ExactKind, ExactOperation, ExecutionDataState, ExecutionTiming, - FieldDataType, SketchAlgorithm, SummaryExpr, SummaryNode, SummaryUpdate, -}; -use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::export::{NonASAPOpKind, PhysicalASAPOperatorPayload}; +use asap_types::ir::operator::agg_intent::{default_quantile, AggIntent}; +use asap_types::ir::operator::operator_properties::{Reduction, Source}; +use asap_types::ir::properties::timing::data_state; +use asap_types::ir::properties::{ExecutionDataState, ExecutionTiming}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{ExactKind, FieldDataType, SketchAlgorithm, SummaryUpdate}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, TimeRangeKind}; + use asap_types::types::AccuracyTarget; // ── fixtures ──────────────────────────────────────────────────────────── -fn metric_scan(labels: &[&str]) -> QueryExpr { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn metric_scan(labels: &[&str]) -> Rc { let mut columns = vec![ Field::plain("ts", DataType::Timestamp, false), Field::plain("value", DataType::Float64, false), @@ -46,17 +54,17 @@ fn metric_scan(labels: &[&str]) -> QueryExpr { .iter() .map(|n| Field::plain(*n, DataType::Utf8, true)), ); - QueryExpr::Scan { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "latency".into(), }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), - } + }) } -fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], @@ -66,8 +74,8 @@ fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc }) } -fn per_entity(intent: AggIntent, child: Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { +fn per_entity(intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec![], @@ -78,11 +86,11 @@ fn per_entity(intent: AggIntent, child: Rc) -> Rc { } /// `quantile by (zone, host) (latency)` — the fine-grained inner summary. -fn fine_quantile() -> Rc { +fn fine_quantile() -> Rc { agg( vec![2, 3], default_quantile(0.99), - Rc::new(metric_scan(&["zone", "host"])), + metric_scan(&["zone", "host"]), ) } @@ -94,10 +102,9 @@ struct StatsModel; /// Search, selection, and materialization must retain the caller's proven rule. #[test] fn custom_accuracy_rule_survives_root_target_and_materialization() { - use asap_aware_mapping::{AccuracyModel, DefaultAccuracyModel, PropagationStats}; - use asap_types::post_asap::{ - AccuracyError, CompositionOperator, ExactOperation, ResultGuarantee, SketchStatistic, - }; + use asap_logical_optimizer::{AccuracyModel, DefaultAccuracyModel, PropagationStats}; + use asap_types::ir::properties::{AccuracyError, CompositionOperator, ResultGuarantee}; + use asap_types::ir::schema::SketchStatistic; struct Model; impl AccuracyModel for Model { fn exact_operation_rule(&self, _: &ExactOperation) -> Option { @@ -125,17 +132,13 @@ fn custom_accuracy_rule_survives_root_target_and_materialization() { } } let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); - let space = asap_aware_mapping::replacement::search_workload_with_targets( + let space = asap_logical_optimizer::pass1::replacement::search_workload_with_targets( vec![("q", root, Some(AccuracyTarget::Exact))], - &default_strategies_with(&StatsModel), + &default_strategies(), &Model, ); - let selection = space.global_selection(&StatsModel); - assert!(selection - .for_target(&space.roots[0].1) - .unwrap() - .composition - .is_some()); + let selection = global_selection(&space, &StatsModel); + assert!(selection.composition(&space.roots[0].1).is_some()); let node = selection .assemble_selected_dag(&space.roots[0].1) .unwrap() @@ -149,17 +152,13 @@ fn custom_accuracy_rule_survives_root_target_and_materialization() { #[test] fn root_target_rejects_unproven_composition() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); - let space = asap_aware_mapping::replacement::search_workload_with_targets( + let space = asap_logical_optimizer::pass1::replacement::search_workload_with_targets( vec![("q", root, Some(AccuracyTarget::Exact))], - &default_strategies_with(&StatsModel), - &asap_aware_mapping::DefaultAccuracyModel, + &default_strategies(), + &asap_logical_optimizer::DefaultAccuracyModel, ); - let selection = space.global_selection(&StatsModel); - assert!(selection - .for_target(&space.roots[0].1) - .unwrap() - .composition - .is_none()); + let selection = global_selection(&space, &StatsModel); + assert!(selection.composition(&space.roots[0].1).is_none()); } impl CostModel for StatsModel { @@ -203,32 +202,6 @@ impl CostModel for StatsModel { } } -/// Same statistics, but the runtime advertises no mixed-execution shape. -struct NoCapabilityModel; - -impl CostModel for NoCapabilityModel { - fn allow_uncosted_legacy_selection(&self) -> bool { - true - } - - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - fn value_operation_capabilities(&self) -> ValueOperationCapabilities { - ValueOperationCapabilities::NONE - } - fn exact_composition_cost_inputs( - &self, - request: &ExactCompositionCostRequest<'_>, - ) -> ExactCompositionCostInputs { - StatsModel.exact_composition_cost_inputs(request) - } -} - /// Complete cost evidence does not imply runtime support evidence. struct UnknownCapabilityModel; @@ -251,26 +224,20 @@ impl CostModel for UnknownCapabilityModel { #[test] fn unknown_runtime_capability_keeps_candidate_but_prevents_selection() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); - let space = plan(vec![("q", root)], &UnknownCapabilityModel); + let space = plan(vec![("q", root)]); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); assert!(group.candidates.iter().any(|candidate| { matches!(candidate.replacement, Replacement::ExactComposition(_)) - && candidate - .runtime_support_evidence(&UnknownCapabilityModel) - .is_none() + && runtime_support_evidence(candidate, &UnknownCapabilityModel).is_none() && UnknownCapabilityModel .candidate_cost( candidate, - &asap_aware_mapping::TargetSubDAG::new(&space.roots[0].1), + &asap_logical_optimizer::TargetSubDAG::new(&space.roots[0].1), ) .is_none() })); - let selection = space.global_selection(&UnknownCapabilityModel); - assert!(selection - .for_target(&space.roots[0].1) - .unwrap() - .composition - .is_none()); + let selection = global_selection(&space, &UnknownCapabilityModel); + assert!(selection.composition(&space.roots[0].1).is_none()); assert!(selection .assemble_selected_dag(&space.roots[0].1) .unwrap() @@ -278,34 +245,43 @@ fn unknown_runtime_capability_keeps_candidate_but_prevents_selection() { } fn plan( - roots: Vec<(&'static str, Rc)>, - cost_model: &dyn CostModel, -) -> asap_aware_mapping::CandidateLogicalASAPDAGs<&'static str> { - search_workload_with(roots, &default_strategies_with(cost_model)) + roots: Vec<(&'static str, Rc)>, +) -> asap_logical_optimizer::CandidateLogicalASAPDAGs<&'static str> { + search_workload_with(roots, &default_strategies()) } -fn is_plain(node: &SummaryNode) -> bool { +fn is_plain(node: &OperatorNode) -> bool { node.schema .fields .iter() .all(|f| matches!(f.dtype, FieldDataType::Plain(_))) } -fn names(node: &SummaryNode) -> Vec<&str> { +fn names(node: &OperatorNode) -> Vec<&str> { node.schema.fields.iter().map(|f| f.name.as_str()).collect() } +/// The composed query-time shape: an exact `Aggregate` directly over a +/// summary evaluation, at query time. +fn is_query_time_fold(node: &OperatorNode) -> bool { + matches!( + node.non_asap(), + Some(NonASAPOp::Aggregate { child, .. }) + if matches!(child.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) + ) +} + // ── step 1: pin every already-supported exact-accumulator nesting ─────── #[test] fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { use std::time::Duration; - let cases: Vec<(Rc, ExactKind)> = vec![ + let cases: Vec<(Rc, ExactKind)> = vec![ ( agg( vec![2], AggIntent::Sum { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Sum, ), @@ -315,7 +291,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { AggIntent::Count { accuracy: AccuracyTarget::Exact, }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Count, ), @@ -323,7 +299,7 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { agg( vec![2], AggIntent::Min { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Min, ), @@ -331,16 +307,17 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { agg( vec![2], AggIntent::Max { col: None }, - Rc::new(metric_scan(&["zone"])), + metric_scan(&["zone"]), ), ExactKind::Max, ), ( per_entity( AggIntent::Rate, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ), ExactKind::Rate, @@ -348,9 +325,10 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { ( per_entity( AggIntent::Increase, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ), ExactKind::Increase, @@ -359,40 +337,43 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { for (inner, kind) in cases { let outer = agg(vec![], default_quantile(0.9), inner); let target = TargetSubDAG::new(&outer); - let candidates = SketchAlgorithmStrategy::default_cost_model().replacements(&target); - let Replacement::Summary(root) = &candidates[0].replacement else { + let candidates = ASAPStrategies::default().replacements(&target); + let Replacement::SubDAG(root) = &candidates[0].replacement else { unreachable!() }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected KLL readout, got {:?}", root.expr); + // Timing is not stored on the plan: time it with the outer summary + // maintained (which also validates every edge) and inspect the copy. + let root = maintained(root); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected KLL evaluation, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("expected outer SummaryAgg"); }; - let SummaryExpr::ValueOperation { - child, - operation: asap_types::post_asap::ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } = &child.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: finalized }) = &child.operator else { panic!("{kind:?}: missing maintenance finalization"); }; + assert_eq!( + child.timing, + Some(ExecutionTiming::IngestionTime), + "{kind:?}: finalization runs at maintenance time" + ); assert!( matches!( - &child.expr, - SummaryExpr::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. } if *k == kind + &finalized.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(k, _), .. }) if *k == kind ), "{kind:?}: expected the exact accumulator under its finalization, got {:?}", - child.expr + finalized.operator ); - validate_execution_data_states(root).expect("accumulator state composes under maintenance"); } } -// ── direction 1: outer exact fold over an inner summary readout ──────── +// ── direction 1: outer exact fold over an inner summary evaluation ──────── -/// Before this PR both `max`/`avg` over a quantile collapsed into one -/// opaque `KeepPreAsap`. Now: the outer group holds an `ValueOperationAtQueryTime` +/// `max`/`avg` over a quantile does not collapse into one opaque kept +/// sub-DAG: the outer group holds an `ValueOperationAtQueryTime` /// candidate referencing the inner target, the inner group keeps its own /// sketch candidates, and with statistics the pair is committed and /// materializes as `ValueOperationAtQueryTime → SummaryEstimate → SummaryAgg`. @@ -400,9 +381,9 @@ fn every_exact_accumulator_is_finalized_before_an_outer_sketch() { fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { for intent in [AggIntent::Max { col: None }, AggIntent::Avg { col: None }] { let root = agg(vec![0], intent.clone(), fine_quantile()); - let space = plan(vec![("q", Rc::clone(&root))], &StatsModel); + let space = plan(vec![("q", Rc::clone(&root))]); let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: inner, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child: inner, .. }) = root.non_asap() else { unreachable!() }; @@ -419,21 +400,20 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { inner_group .candidates .iter() - .any(|c| matches!(&c.replacement, Replacement::Summary(n) - if matches!(n.expr, SummaryExpr::SummaryEstimate { .. }))), - "{intent:?}: the inner quantile keeps its own readout candidates" + .any(|c| matches!(&c.replacement, Replacement::SubDAG(n) + if matches!(n.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })))), + "{intent:?}: the inner quantile keeps its own evaluation candidates" ); - let selection = space.global_selection(&StatsModel); + let selection = global_selection(&space, &StatsModel); let selected = selection.for_target(&root).unwrap(); let chosen = selected.chosen.expect("a decision"); assert_eq!( chosen.provenance, ReplacementProvenance::ValueOperationAtQueryTime ); - let decision = selected - .composition - .as_ref() + let decision = selection + .composition(&root) .expect("composition provenance"); assert!(Rc::ptr_eq(decision.child_target, inner)); assert!(decision.cost_rate < decision.baseline_rate); @@ -449,18 +429,16 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { )); let composed = selection.assemble_selected_dag(&root).unwrap().unwrap(); - let SummaryExpr::ValueOperation { - child, - timing: ExecutionTiming::QueryTime, - .. - } = &composed.expr - else { + let Some(NonASAPOp::Aggregate { child, .. }) = composed.non_asap() else { panic!( "{intent:?}: expected ValueOperationAtQueryTime root, got {:?}", - composed.expr + composed.operator ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); assert!( child.guarantee.is_some(), "child has its KLL rank guarantee" @@ -472,15 +450,18 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { assert!(is_plain(&composed)); assert_eq!( names(&composed), - root.output_schema() - .unwrap() + root.schema .fields .iter() .map(|c| c.name.as_str()) .collect::>(), "the composed plan's schema is the pre-ASAP target's own" ); - validate_execution_data_states(&composed).unwrap(); + assert_eq!( + timed(&composed).timing, + Some(ExecutionTiming::QueryTime), + "{intent:?}: the exact fold runs at query time" + ); } } @@ -490,13 +471,9 @@ fn max_and_avg_over_quantile_compose_at_query_time_with_statistics() { fn avg_over_quantile_keeps_the_sum_over_count_rewrite_as_a_competitor() { // `by (zone)` over `by (zone)`: the averaged column resolves to the // non-null quantile output, which is what the rewrite requires. - let inner = agg( - vec![2], - default_quantile(0.99), - Rc::new(metric_scan(&["zone"])), - ); + let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); let root = agg(vec![0], AggIntent::Avg { col: None }, inner); - let space = plan(vec![("q", root)], &StatsModel); + let space = plan(vec![("q", root)]); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); let provenances: Vec<_> = group.candidates.iter().map(|c| c.provenance).collect(); assert!(provenances.contains(&ReplacementProvenance::LogicalRewrite)); @@ -508,33 +485,31 @@ fn avg_over_quantile_keeps_the_sum_over_count_rewrite_as_a_competitor() { /// is the same, only the fold's row multiplicity differs. #[test] fn identity_and_genuine_multi_row_folds_both_compose() { - let identity_inner = agg( - vec![2], - default_quantile(0.99), - Rc::new(metric_scan(&["zone"])), - ); + let identity_inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); for (label, inner) in [ ("identity", identity_inner), ("fine-to-coarse", fine_quantile()), ] { let root = agg(vec![0], AggIntent::Max { col: None }, inner); - let space = plan(vec![("q", root)], &StatsModel); + let space = plan(vec![("q", root)]); let root = &space.roots[0].1; - let composed = space - .global_selection(&StatsModel) + let composed = global_selection(&space, &StatsModel) .assemble_selected_dag(root) .unwrap() .unwrap(); assert!( matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } + composed.non_asap(), + Some(NonASAPOp::Aggregate { child, .. }) + if matches!(child.operator, Operator::ASAP(ASAPOp::SummaryEstimate { .. })) ), "{label}: {:?}", - composed.expr + composed.operator + ); + assert_eq!( + timed(&composed).timing, + Some(ExecutionTiming::QueryTime), + "{label}" ); assert_eq!(names(&composed), vec!["zone", "max"], "{label}"); } @@ -543,17 +518,17 @@ fn identity_and_genuine_multi_row_folds_both_compose() { /// One inner quantile consumed by two outer folds in two queries: CSE /// collapses the inner target onto one `Rc`, both compositions commit to /// the *same* child candidate, and both materializations share one -/// `Rc` for it — the summary is maintained once. +/// `Rc` for it — the summary is maintained once. #[test] fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let max = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); let min = agg(vec![0], AggIntent::Min { col: None }, fine_quantile()); - let space = plan(vec![("max", max), ("min", min)], &StatsModel); - let selection = space.global_selection(&StatsModel); + let space = plan(vec![("max", max), ("min", min)]); + let selection = global_selection(&space, &StatsModel); - let roots: Vec> = space.roots.iter().map(|(_, r)| Rc::clone(r)).collect(); - let inner_of = |r: &Rc| match r.as_ref() { - QueryExpr::Aggregate { child, .. } => Rc::clone(child), + let roots: Vec> = space.roots.iter().map(|(_, r)| Rc::clone(r)).collect(); + let inner_of = |r: &Rc| match r.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) => Rc::clone(child), _ => unreachable!(), }; assert!( @@ -568,14 +543,7 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { let decisions: Vec<_> = roots .iter() - .map(|r| { - selection - .for_target(r) - .unwrap() - .composition - .as_ref() - .expect("both roots compose") - }) + .map(|r| selection.composition(r).expect("both roots compose")) .collect(); assert!(std::ptr::eq( decisions[0].child_candidate.unwrap(), @@ -589,17 +557,20 @@ fn a_shared_inner_summary_is_materialized_once_for_several_outer_folds() { .iter() .map(|r| selection.assemble_selected_dag(r).unwrap().unwrap()) .collect(); - let child_of = |n: &Rc| match &n.expr { - SummaryExpr::ValueOperation { - child, - timing: ExecutionTiming::QueryTime, - .. - } => Rc::clone(child), - other => panic!("expected ValueOperationAtQueryTime, got {other:?}"), + let child_of = |n: &Rc| match n.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) + if matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ) => + { + Rc::clone(child) + } + _ => panic!("expected ValueOperationAtQueryTime, got {:?}", n.operator), }; assert!( Rc::ptr_eq(&child_of(&composed[0]), &child_of(&composed[1])), - "both folds compose over the same Rc" + "both folds compose over the same Rc" ); } @@ -614,15 +585,16 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { use std::time::Duration; let deriv = per_entity( AggIntent::Deriv, - Rc::new(QueryExpr::TimeRange { + node(NonASAPOp::TimeRange { range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["zone"])), + kind: TimeRangeKind::Range, + child: metric_scan(&["zone"]), }), ); let root = agg(vec![], default_quantile(0.99), deriv); - let space = plan(vec![("q", root)], &StatsModel); + let space = plan(vec![("q", root)]); let root = Rc::clone(&space.roots[0].1); - let QueryExpr::Aggregate { child: deriv, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child: deriv, .. }) = root.non_asap() else { unreachable!() }; assert!(space @@ -632,44 +604,36 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { .iter() .any(|c| c.provenance == ReplacementProvenance::ValueOperationAtIngestionTime)); - let selection = space.global_selection(&StatsModel); + let selection = global_selection(&space, &StatsModel); let deriv_sel = selection.for_target(deriv).unwrap(); assert_eq!( deriv_sel.chosen.unwrap().provenance, ReplacementProvenance::ValueOperationAtIngestionTime ); - let decision = deriv_sel.composition.as_ref().unwrap(); + let decision = selection.composition(deriv).unwrap(); assert!(decision.child_candidate.is_none(), "function input is raw"); assert!(decision.cost_rate < decision.baseline_rate); let composed = selection.assemble_selected_dag(&root).unwrap().unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &composed.expr else { - panic!("expected readout root, got {:?}", composed.expr); + // Walk the timed copy, with the outer summary maintained at ingestion time. + let composed = maintained(&composed); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &composed.operator else { + panic!("expected evaluation root, got {:?}", composed.operator); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("expected SummaryAgg"); }; - let SummaryExpr::ValueOperation { - child: raw, - timing: ExecutionTiming::IngestionTime, - .. - } = &child.expr - else { + let Some(NonASAPOp::Aggregate { child: raw, .. }) = child.non_asap() else { panic!( "expected ValueOperationAtIngestionTime under the maintained summary, got {:?}", - child.expr + child.operator ); }; - assert!(matches!(raw.expr, SummaryExpr::KeepPreAsap(_))); - let assignment = validate_execution_data_states(&composed).unwrap(); - assert_eq!( - assignment.data_state_of(child), - Some(ExecutionDataState::INGESTION_ROWS) - ); - assert_eq!( - assignment.data_state_of(raw), - Some(ExecutionDataState::INGESTION_ROWS) - ); + // The raw input is kept as-is. + assert!(matches!(raw.non_asap(), Some(NonASAPOp::TimeRange { .. }))); + assert!(!raw.contains_asap()); + assert_eq!(data_state(child), Some(ExecutionDataState::INGESTION_ROWS)); + assert_eq!(data_state(raw), Some(ExecutionDataState::INGESTION_ROWS)); } // ── rejection, capability, statistics ─────────────────────────────────── @@ -679,51 +643,33 @@ fn outer_summary_over_an_exact_function_composes_at_ingestion_time() { #[test] fn summary_construction_follows_its_value_input_phase() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); - let space = plan(vec![("q", Rc::clone(&root))], &StatsModel); - let post = space - .global_selection(&StatsModel) + let space = plan(vec![("q", Rc::clone(&root))]); + let post = global_selection(&space, &StatsModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let illegal = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: post, - family: FieldDataType::ExactAggregate( - ExactKind::Max, - asap_types::post_asap::ExactParams::Max, - ), - input: SummaryUpdate::column(asap_types::pre_asap::ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: Default::default(), - filter: None, - }, - schema: asap_types::post_asap::Schema::lifted(vec![], None), - guarantee: None, - }); - let state = asap_types::post_asap::produced_data_state(&illegal.expr).unwrap(); + let illegal = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: post, + family: FieldDataType::ExactAggregate( + ExactKind::Max, + asap_types::ir::schema::ExactParams::Max, + ), + input: SummaryUpdate::column(asap_types::ir::scalar::ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: Default::default(), + filter: None, + }), + Schema::lifted(vec![], None), + ) + .with_guarantee(None), + ); + // Even under an ingestion-time consumer the state is built at query + // time, because a evaluation sits below it. + let state = asap_types::ir::planned_data_state(&illegal, ExecutionTiming::IngestionTime); assert_eq!(state.timing, ExecutionTiming::QueryTime); - asap_types::post_asap::validate_execution_data_states_at(&illegal, state).unwrap(); -} - -#[test] -fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { - let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); - let space = plan(vec![("q", root)], &NoCapabilityModel); - let root = Rc::clone(&space.roots[0].1); - let group = space.candidates_for_target(&root).unwrap(); - assert!(group - .candidates - .iter() - .all(|c| !matches!(c.replacement, Replacement::ExactComposition(_)))); - let selection = space.global_selection(&NoCapabilityModel); - assert!(selection.for_target(&root).unwrap().composition.is_none()); - let node = selection.assemble_selected_dag(&root).unwrap().unwrap(); - assert!(!matches!(node.expr, SummaryExpr::ValueOperation { .. })); - // The inner quantile is still independently selectable. - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { - unreachable!() - }; - assert!(selection.for_target(child).unwrap().chosen.is_some()); + asap_types::ir::validate_maintained(&illegal, state.timing).unwrap(); } /// Without statistics (the built-in model) the composition is *proposed* @@ -731,9 +677,9 @@ fn a_runtime_without_mixed_execution_gets_no_composition_candidates() { /// site keeps a non-composed alternative, and the inner summary stays /// independently selectable. #[test] -fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { +fn missing_cost_statistics_preserve_the_conservative_retain_exact() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); - let space = plan(vec![("q", root)], &DefaultCostModel); + let space = plan(vec![("q", root)]); let root = Rc::clone(&space.roots[0].1); assert!(space .candidates_for_target(&root) @@ -741,17 +687,17 @@ fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { .candidates .iter() .any(|c| c.provenance == ReplacementProvenance::ValueOperationAtQueryTime)); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); let selected = selection.for_target(&root).unwrap(); - assert!(selected.composition.is_none()); + assert!(selection.composition(&root).is_none()); assert!(!matches!( selected.chosen.map(|c| &c.replacement), Some(Replacement::ExactComposition(_)) )); let node = selection.assemble_selected_dag(&root).unwrap().unwrap(); - assert!(!matches!(node.expr, SummaryExpr::ValueOperation { .. })); + assert!(!is_query_time_fold(&node)); - let explanations = asap_aware_mapping::explain_replacements(vec![("q", (*root).clone())]); + let explanations = asap_logical_optimizer::explain_replacements(vec![("q", Rc::clone(&root))]); assert!(explanations .iter() .any(|e| e.kind == ExplanationKind::ExactComposition)); @@ -762,21 +708,27 @@ fn missing_cost_statistics_preserve_the_conservative_keep_pre_asap() { #[test] fn dag_export_carries_explicit_stage_and_plain_schema_for_a_composed_plan() { let root = agg(vec![0], AggIntent::Max { col: None }, fine_quantile()); - let space = plan(vec![("q", root)], &StatsModel); + let space = plan(vec![("q", root)]); let root = &space.roots[0].1; - let composed = space - .global_selection(&StatsModel) + let composed = global_selection(&space, &StatsModel) .assemble_selected_dag(root) .unwrap() .unwrap(); let dag = dag_export::export_summary(&composed); let node = &dag.nodes[dag.root as usize]; - assert_eq!(node.kind, "ValueOperation"); - assert_eq!(node.detail["timing"], "query_time"); - assert!(node.detail["operation"] - .as_str() - .unwrap() - .starts_with("Exact(Aggregate")); + assert_eq!(node.kind, "aggregate"); + assert!(node.detail["measures"].is_array()); + // Timing is explicit in the wire-6 DAG: the root is a relational + // aggregate placed at query time. + let wire = post_asap_dag(&composed); + let wire_root = wire.nodes.iter().find(|n| n.id == wire.roots[0]).unwrap(); + assert!(matches!( + wire_root.payload, + PhysicalASAPOperatorPayload::Relational { + operator: NonASAPOpKind::Aggregate { .. } + } + )); + assert_eq!(wire_root.output_state.timing, ExecutionTiming::QueryTime); // Pre-ASAP export of the same target still describes the same columns. let pre = dag_export::export(root); @@ -798,9 +750,9 @@ fn promql_max_by_zone_over_quantile_over_time_composes() { AccuracyTarget::Epsilon(0.01), ) .unwrap(); - let space = plan(vec![("q", Rc::new(expr))], &StatsModel); + let space = plan(vec![("q", expr)]); let root = &space.roots[0].1; - let selection = space.global_selection(&StatsModel); + let selection = global_selection(&space, &StatsModel); let selected = selection.for_target(root).unwrap(); assert_eq!( selected.chosen.map(|c| c.provenance), @@ -815,15 +767,10 @@ fn promql_max_by_zone_over_quantile_over_time_composes() { .collect::>() ); let composed = selection.assemble_selected_dag(root).unwrap().unwrap(); - assert!(matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } - )); + assert!(is_query_time_fold(&composed), "{:?}", composed.operator); + assert_eq!(timed(&composed).timing, Some(ExecutionTiming::QueryTime)); assert_eq!( - selected.composition.as_ref().map(|d| d.inputs.unit), + selection.composition(root).map(|d| d.inputs.unit), Some(CostUnit::CostUnitsPerSecond) ); let _ = OperationPlacement::Read; diff --git a/crates/integration-tests/tests/frontend_timestamps.rs b/crates/integration-tests/tests/frontend_timestamps.rs index 388261da6..58de9ade9 100644 --- a/crates/integration-tests/tests/frontend_timestamps.rs +++ b/crates/integration-tests/tests/frontend_timestamps.rs @@ -1,9 +1,9 @@ //! Cross-frontend evaluation-time semantics (issues #46 and #184). use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_integration_tests::fixtures::lower_promql; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::QueryExpr; +use asap_integration_tests::fixtures::lower_promql_root; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{NonASAPOp, ScalarExpr}; use asap_types::types::AccuracyTarget; /// PromQL exposes its evaluation time as Unix seconds, whereas SQL exposes @@ -11,10 +11,21 @@ use asap_types::types::AccuracyTarget; /// but must remain distinguishable in the shared IR and type inference. #[tokio::test] async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { - let promql = lower_promql("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); - assert!(matches!(promql, QueryExpr::EvalTimestamp)); - let promql_schema = promql.output_schema().expect("PromQL time() schema"); - assert_eq!(promql_schema.fields[0].dtype, DataType::Float64); + let promql = lower_promql_root("time()", AccuracyTarget::Exact).expect("lower PromQL time()"); + assert!( + matches!( + promql, + asap_types::ir::QueryRoot::Scalar(ScalarExpr::EvalTimestamp) + ), + "expected a bare evaluation-time scalar, got {promql:?}" + ); + assert_eq!( + ScalarExpr::EvalTimestamp + .scalar_type(&Schema::default()) + .unwrap() + .0, + DataType::Float64 + ); let catalog = SqlCatalog::new().with_table( "metrics", @@ -27,13 +38,14 @@ async fn promql_eval_time_and_sql_current_timestamp_remain_distinct() { ) .await .expect("lower SQL CURRENT_TIMESTAMP"); - let QueryExpr::Project { cols, .. } = sql else { + let Some(NonASAPOp::Project { cols, child, .. }) = sql.non_asap() else { panic!("expected SQL projection, got {sql:?}"); }; - assert!(matches!(&cols[0].expr, QueryExpr::CurrentTimestamp)); - let sql_schema = cols[0] + assert!(matches!(&cols[0].expr, ScalarExpr::CurrentTimestamp)); + let (sql_dtype, _) = cols[0] .expr - .output_schema() - .expect("SQL CURRENT_TIMESTAMP schema"); - assert_eq!(sql_schema.fields[0].dtype, DataType::Timestamp); + .scalar_type(&child.schema) + .expect("SQL CURRENT_TIMESTAMP type"); + assert_eq!(sql_dtype, DataType::Timestamp); + assert_eq!(sql.schema.fields[0].dtype, DataType::Timestamp); } diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index b48d14bf7..cf9a495c4 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -1,7 +1,7 @@ //! Maintenance -> stored pane state -> independently bound query execution. mod physical_common; -use asap_physical_operators::{ - operators::{Operator, ReadoutQuery}, +use asap_executor::{ + operators::{Operator, SummaryEvaluation}, physical_planner::{CompiledPhysicalDAG, InputContract, Source}, plan::{PhysicalDAG, PhysicalOperator, PlanProperties}, runtime::{Input, Limits, OutputStream, RunContext, Scope}, @@ -9,11 +9,9 @@ use asap_physical_operators::{ values::{Batch, SchemaRef, Value}, AggregateCore, Error, }; -use asap_types::{ - post_asap::{ - Field, FieldDataType, Schema, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, - }, - pre_asap::DataType, +use asap_types::ir::schema::{ + DataType, Field, FieldDataType, Schema, SketchAlgorithm, SketchKind, SketchParams, + SketchStatistic, }; use futures::{executor::block_on, StreamExt}; use std::{ @@ -32,8 +30,8 @@ fn family(k: u32) -> FieldDataType { } fn raw_schema() -> SchemaRef { Arc::new(Schema { - closed: true, unique_keys: vec![], + closed: false, fields: vec![Field { table: None, name: "value".into(), @@ -98,11 +96,11 @@ fn restore(schema: SchemaRef, states: &[Arc]) -> Batch { ) .unwrap() } -fn readout(schema: SchemaRef, q: f64) -> Operator { - Operator::readout( +fn evaluation(schema: SchemaRef, q: f64) -> Operator { + Operator::evaluation( schema, 0, - ReadoutQuery::Sketch(SketchStatistic::Quantile { q }), + SummaryEvaluation::Sketch(SketchStatistic::Quantile { q }), ) .unwrap() } @@ -159,21 +157,21 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { ), ), (6, (vec![5], merge.clone())), - (7, (vec![6], readout(schema.clone(), 0.5))), - (8, (vec![6], readout(schema.clone(), 0.99))), + (7, (vec![6], evaluation(schema.clone(), 0.5))), + (8, (vec![6], evaluation(schema.clone(), 0.99))), ]), vec![6, 7, 8], ) .unwrap(); - let layout = asap_types::post_asap::PaneLayout { + let layout = asap_types::physical::PaneLayout { pane_width_ms: 60_000, pane_origin_ms: Some(0), }; assert!( - asap_types::post_asap::validate_pane_coverage( + asap_types::physical::validate_pane_coverage( &layout, Some(330_000), - &asap_types::post_asap::WindowEdgeCoverage::PaneAligned + &asap_types::physical::WindowEdgeCoverage::PaneAligned ) .is_err(), "moving window edges require residual computation" @@ -181,10 +179,10 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { for offset in [0, 1] { let restored = restore(schema.clone(), &panes[offset..offset + 5]); let evaluation_time_ms = (5 + offset as i64) * 60_000; - asap_types::post_asap::validate_pane_coverage( + asap_types::physical::validate_pane_coverage( &layout, Some(evaluation_time_ms), - &asap_types::post_asap::WindowEdgeCoverage::PaneAligned, + &asap_types::physical::WindowEdgeCoverage::PaneAligned, ) .unwrap(); let inputs: BTreeMap<_, _> = (0..5) @@ -247,8 +245,10 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { }, ) .unwrap(); - dag.add(2, vec![1], readout(schema.clone(), 0.5)).unwrap(); - dag.add(3, vec![1], readout(schema.clone(), 0.99)).unwrap(); + dag.add(2, vec![1], evaluation(schema.clone(), 0.5)) + .unwrap(); + dag.add(3, vec![1], evaluation(schema.clone(), 0.99)) + .unwrap(); let outputs = block_on(futures::future::join_all( dag.execute( &[2, 3], diff --git a/crates/integration-tests/tests/nested.rs b/crates/integration-tests/tests/nested.rs index 1e3e34ebd..2bd8b8747 100644 --- a/crates/integration-tests/tests/nested.rs +++ b/crates/integration-tests/tests/nested.rs @@ -1,8 +1,8 @@ //! Multi-node pipeline tests — nested `Aggregate`, `TimeRange`, `BinaryOp`, and `Scan`. //! //! Key invariant: `rate`/`increase` are label-preserving (per-series), so an -//! outer `Aggregate.by` resolves its group keys against the inner aggregate's -//! output schema, which still carries all label columns. +//! outer `Aggregate` reduction resolves its group keys against the inner +//! aggregate's output schema, which still carries all label columns. //! //! Label column ordering is always alphabetical, so in a query that references //! both `job` and `status`: @@ -13,56 +13,108 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{ - AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, GroupKeys, Predicate, - PromQLVectorSetOpKind, QueryExpr, Reduction, ScalarValue, Source, TimeShift, VectorMatch, - VectorMatchKind, +use asap_types::ir::operator::{ + AggIntent, AtModifier, BinaryOpKind, GroupKeys, PromQLVectorSetOpKind, Reduction, Source, + TimeShift, VectorMatch, VectorMatchKind, +}; +use asap_types::ir::scalar::{ArithmeticOpKind, CompareOpKind, ScalarValue}; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, TimeRangeKind, }; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str, predicates: Vec, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { + metric: metric.into(), + }, + predicates, + schema: metric_schema(labels), + }) +} + +fn instant(child: Rc) -> Rc { + node(NonASAPOp::TimeRange { + range: Duration::from_secs(1), + kind: TimeRangeKind::Instant, + child, + }) +} + +fn range(secs: u64, child: Rc) -> Rc { + node(NonASAPOp::TimeRange { + range: Duration::from_secs(secs), + kind: TimeRangeKind::Range, + child, + }) +} + +fn eq_pred(col_id: usize, value: &str) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col_id)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Literal(ScalarValue::Utf8(value.into()))), + semantics: ExprSemantics::Promql, + }) +} + +fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(child), - } + child, + }) } -fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { +fn agg_per_entity(intent: AggIntent, child: Rc) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(child), - } + child, + }) +} + +fn binary( + kind: BinaryOpKind, + vector_match: Option, + lhs: Rc, + rhs: Rc, +) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind, + vector_match, + }, + return_bool: false, + lhs, + rhs, + }) } // #22 — sum by job over rate; outer by=[2] resolves against rate's // label-preserving output schema [ts, value, job] #[test] fn q22_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["job"])), ); let expected = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); assert_eq!( @@ -76,25 +128,12 @@ fn q22_sum_by_job_over_rate() { // predicate on status (col 3); group key job (col 2) #[test] fn q23_sum_by_job_over_filtered_scan() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; - let expected = agg( - vec![2], - AggIntent::Sum { col: None }, - QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], ); + let expected = agg(vec![2], AggIntent::Sum { col: None }, instant(scan)); assert_eq!( lower(r#"sum by (job) (http_requests_total{status="200"})"#), expected @@ -108,54 +147,30 @@ fn q23_sum_by_job_over_filtered_scan() { // schema [ts, value, job]; outer by=[2] (job) #[test] fn q25_div_over_complex_sub_dags() { - let lhs_scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; + let lhs_scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], + ); let lhs = agg( vec![2], AggIntent::Sum { col: None }, - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(lhs_scan), - }, - ), + agg_per_entity(AggIntent::Rate, range(300, lhs_scan)), ); - let rhs_scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_errors_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; + let rhs_scan = scan("http_errors_total", vec![], &["job"]); let rhs = agg( vec![2], AggIntent::Sum { col: None }, - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(rhs_scan), - }, - ), + agg_per_entity(AggIntent::Rate, range(300, rhs_scan)), ); - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: Rc::new(lhs), - rhs: Rc::new(rhs), - vector_match: None, - }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + None, + lhs, + rhs, + ); assert_eq!( lower( r#"sum by (job) (rate(http_requests_total{status="200"}[5m])) / sum by (job) (rate(http_errors_total[5m]))"# @@ -171,19 +186,9 @@ fn q25_div_over_complex_sub_dags() { // label-preserving output schema; the outer `max` has no grouping. #[test] fn q27_max_over_sum_by_job_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["job"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["job"])), ); let sum_by_job = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); let expected = agg(vec![], AggIntent::Max { col: None }, sum_by_job); @@ -201,25 +206,12 @@ fn q27_max_over_sum_by_job_over_rate() { // Scan schema: [ts(0), value(1), group(2), job(3)] (labels alphabetical). #[test] fn q53_outer_group_key_absent_from_nested_aggregate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api-server".into()))), - }))], - schema: metric_schema(&["group", "job"]), - }; - let inner = agg( - vec![2], - AggIntent::Sum { col: None }, - QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests", + vec![eq_pred(3, "api-server")], + &["group", "job"], ); + let inner = agg(vec![2], AggIntent::Sum { col: None }, instant(scan)); let expected = agg(vec![], AggIntent::Sum { col: None }, inner); assert_eq!( lower(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#), @@ -236,33 +228,26 @@ fn q53_outer_group_key_absent_from_nested_aggregate() { // parser's default `ignoring([])` match modifier. #[test] fn q52_outer_name_label_over_binary_op() { - let side = |metric: &str, env: &str| QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), // env - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(env.into()))), - }))], - schema: metric_schema(&["env", "__name__"]), - }), + let side = |metric: &str, env: &str| { + instant(scan( + metric, + vec![eq_pred(2, env)], // env + &["env", "__name__"], + )) }; let expected = agg( vec![3], // __name__ AggIntent::Sum { col: None }, - QueryExpr::BinaryOp { - op: BinaryOpKind::Set(PromQLVectorSetOpKind::Or), - lhs: Rc::new(side("metric_a", "1")), - rhs: Rc::new(side("metric_b", "2")), - vector_match: Some(VectorMatch { + binary( + BinaryOpKind::Set(PromQLVectorSetOpKind::Or), + Some(VectorMatch { kind: VectorMatchKind::Ignoring, labels: vec![], grouping: None, }), - }, + side("metric_a", "1"), + side("metric_b", "2"), + ), ); assert_eq!( lower(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#), @@ -276,28 +261,18 @@ fn q52_outer_name_label_over_binary_op() { // the inner rate is label-preserving. Scan schema [ts(0), value(1), instance(2)]. #[test] fn q39_sum_without_instance_over_rate() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![], - schema: metric_schema(&["instance"]), - }; let inner_rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + range(300, scan("http_requests_total", vec![], &["instance"])), ); - let expected = QueryExpr::Aggregate { + let expected = node(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(vec![2])), // exclude `instance` measures: vec![AggIntent::Sum { col: None }], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(inner_rate), - }; + child: inner_rate, + }); assert_eq!( lower("sum without (instance) (rate(http_requests_total[5m]))"), expected, @@ -310,35 +285,25 @@ fn q39_sum_without_instance_over_rate() { #[test] fn q40_week_over_week_offset() { let rate_over = |shift: Option| { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: metric_schema(&[]), - }; + let scan = scan("m", vec![], &[]); let ranged = match shift { - Some(ms) => QueryExpr::TimeShift { + Some(ms) => node(NonASAPOp::TimeShift { shift: TimeShift { offset_ms: ms, at: None, }, - child: Rc::new(scan), - }, + child: scan, + }), None => scan, }; - agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(ranged), - }, - ) - }; - let expected = QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), - lhs: Rc::new(rate_over(None)), - rhs: Rc::new(rate_over(Some(604_800_000))), // 1w - vector_match: None, + agg_per_entity(AggIntent::Rate, range(300, ranged)) }; + let expected = binary( + BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + None, + rate_over(None), + rate_over(Some(604_800_000)), // 1w + ); assert_eq!(lower("rate(m[5m]) - rate(m[5m] offset 1w)"), expected,); } @@ -346,22 +311,13 @@ fn q40_week_over_week_offset() { // (seconds → ms); a bare selector wrapped in a `TimeShift` carrying the anchor. #[test] fn q40_at_modifier_absolute() { - let expected = QueryExpr::TimeRange { - range: Duration::from_secs(1), - child: Rc::new(QueryExpr::TimeShift { - shift: TimeShift { - offset_ms: 0, - at: Some(AtModifier::Timestamp(1_609_746_000_000)), - }, - child: Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: "up".into(), - }, - predicates: vec![], - schema: metric_schema(&[]), - }), - }), - }; + let expected = instant(node(NonASAPOp::TimeShift { + shift: TimeShift { + offset_ms: 0, + at: Some(AtModifier::Timestamp(1_609_746_000_000)), + }, + child: scan("up", vec![], &[]), + })); assert_eq!(lower("up @ 1609746000"), expected); } @@ -370,24 +326,12 @@ fn q40_at_modifier_absolute() { // so outer sum by job still finds job at col 2 #[test] fn q24_sum_by_job_over_rate_over_filtered_scan() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "http_requests_total".into(), - }, - predicates: vec![Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("200".into()))), - }))], - schema: metric_schema(&["job", "status"]), - }; - let inner_rate = agg_per_entity( - AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }, + let scan = scan( + "http_requests_total", + vec![eq_pred(3, "200")], + &["job", "status"], ); + let inner_rate = agg_per_entity(AggIntent::Rate, range(300, scan)); let expected = agg(vec![2], AggIntent::Sum { col: None }, inner_rate); assert_eq!( lower(r#"sum by (job) (rate(http_requests_total{status="200"}[5m]))"#), @@ -403,35 +347,25 @@ fn q24_sum_by_job_over_rate_over_filtered_scan() { // the whole spine survives verbatim and the schema stays label-preserving. #[test] fn q27_nested_subquery_prometheus_docs_example() { - let scan = QueryExpr::Scan { - source: Source::TimeSeries { - metric: "distance_covered_total".into(), - }, - predicates: vec![], - schema: metric_schema(&[]), - }; let rate = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { - range: Duration::from_secs(5), - child: Rc::new(scan), - }, + range(5, scan("distance_covered_total", vec![], &[])), ); let deriv = agg_per_entity( AggIntent::Deriv, - QueryExpr::PromqlSubquery { + node(NonASAPOp::PromqlSubquery { range: Duration::from_secs(30), resolution: Some(Duration::from_secs(5)), - child: Rc::new(rate), - }, + child: rate, + }), ); let expected = agg_per_entity( AggIntent::Max { col: None }, - QueryExpr::PromqlSubquery { + node(NonASAPOp::PromqlSubquery { range: Duration::from_secs(600), resolution: None, - child: Rc::new(deriv), - }, + child: deriv, + }), ); assert_eq!( lower("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"), diff --git a/crates/integration-tests/tests/operator_design_examples.rs b/crates/integration-tests/tests/operator_design_examples.rs new file mode 100644 index 000000000..755dc61f2 --- /dev/null +++ b/crates/integration-tests/tests/operator_design_examples.rs @@ -0,0 +1,396 @@ +//! #511 examples: source text → unified dag → summary rewrite → flat export. +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::properties::summary_coverage::{CoverageRegion, SummaryCoverage}; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{ + ExactKind, ExactParams, FieldDataType, GroupingStrategy, SummaryUpdate, +}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; +use asap_types::types::AccuracyTarget; +use std::rc::Rc; +mod physical_common; + +fn catalog() -> SqlCatalog { + SqlCatalog::new() + .with_table( + "requests", + Schema::new(vec![ + Field::plain("bytes", DataType::Int64, true), + Field::plain("status", DataType::Int64, false), + ]), + ) + .with_table( + "lineitem", + Schema::new(vec![Field::plain("l_quantity", DataType::Int64, false)]), + ) +} + +/// The SQL scalar example keeps column scopes and a Boolean row predicate. +#[tokio::test] +async fn sql_filter_projection_example() { + let root = lower_sql( + "SELECT l_quantity * 2 AS q2 FROM lineitem WHERE l_quantity > 10", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert_eq!( + root.schema.fields[0], + Field::plain("q2", DataType::Int64, false) + ); + assert!( + matches!(root.expect_non_asap(),NonASAPOp::Project { cols,.. } if matches!(cols[0].expr,ScalarExpr::Arithmetic { .. })) + ); + let wire = physical_common::compile_physical_asap_dag(&root).unwrap(); + wire.validate().unwrap(); + assert_eq!(wire.nodes.len(), OperatorNode::reachable(&root).len()); +} + +/// SUM's evaluation preserves integer type and SQL NULL behavior across the rewrite. +#[tokio::test] +async fn sql_sum_projection_before_and_after_summary_rewrite() { + let root = lower_sql( + "SELECT SUM(bytes) + 1 AS total_bytes FROM requests WHERE status = 200", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert_eq!( + root.schema.fields[0], + Field::plain("total_bytes", DataType::Int64, true) + ); + fn rewrite(node: &Rc) -> Rc { + if let Some(NonASAPOp::Aggregate { + child, + reduction, + measures, + .. + }) = node.non_asap() + { + let [AggIntent::Sum { col: Some(column) }] = measures.as_slice() else { + panic!() + }; + let state = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(child), + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + input: SummaryUpdate::column(ColumnRef::Named( + child.schema.fields[*column].name.clone(), + )), + reduction: reduction.clone(), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap() + // Whole-source coverage, as the planner declares it today (#570). + .with_coverage(SummaryCoverage { + source: asap_types::ir::operator::Source::Table { + table_ref: "requests".into(), + }, + regions: vec![CoverageRegion { + time_ms: None, + population: Default::default(), + }], + }) + .unwrap(), + ); + let finalize = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: state, + }), + node.schema.clone(), + ) + .with_guarantee(None), + ); + return finalize; + } + Rc::new(node.map_children(rewrite).unwrap()) + } + let rewritten = rewrite(&root); + rewritten.validate_structure().unwrap(); + assert_eq!(rewritten.schema, root.schema); + let dag = OperatorNode::reachable(&rewritten); + assert!(dag + .iter() + .any(|n| matches!(n.asap(), Some(ASAPOp::SummaryAgg { .. })))); + assert!(dag + .iter() + .any(|n| matches!(n.asap(), Some(ASAPOp::FinalizeExactAccumulator { .. })))); + let wire = physical_common::compile_physical_asap_dag(&rewritten).unwrap(); + wire.validate().unwrap(); + assert_eq!(wire.nodes.len(), dag.len()); + let json = serde_json::to_string(&wire).unwrap(); + assert!(!json.contains("KeepPreAsap") && !json.contains("ScalarBridge")); +} + +/// Scalar subqueries survive normalization with shared, visible producers. +#[tokio::test] +async fn sql_scalar_subquery_retains_its_cardinality_contract() { + for query in [ + "SELECT (SELECT bytes FROM requests) AS v FROM lineitem", + "SELECT l_quantity NOT IN (SELECT bytes FROM requests) AS present FROM lineitem", + ] { + let root = lower_sql(query, &catalog(), AccuracyTarget::Exact) + .await + .unwrap(); + root.validate_structure().unwrap(); + assert!(root.children().len() > 1); + let wire = physical_common::compile_physical_asap_dag(&root).unwrap(); + assert!(wire + .edges + .iter() + .any(|e| e.role == asap_types::ir::export::EdgeRole::ScalarRef)); + } +} + +/// Execute the SQL SUM example for nonempty, empty and all-NULL populations. +#[tokio::test] +async fn sql_sum_example_executes_with_sql_null_semantics() { + use asap_executor::{ + physical_planner::{compile, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + sources::{DataSources, MemorySource}, + values::{Batch, Value}, + }; + use futures::StreamExt; + use std::{collections::BTreeMap, sync::Arc}; + let root = lower_sql( + "SELECT SUM(bytes) + 1 AS total_bytes FROM requests WHERE status = 200", + &catalog(), + AccuracyTarget::Exact, + ) + .await + .unwrap(); + let logical_scan = OperatorNode::reachable(&root) + .into_iter() + .find(|node| matches!(node.non_asap(), Some(NonASAPOp::Scan { .. }))) + .unwrap(); + let NonASAPOp::Scan { source, .. } = logical_scan.expect_non_asap() else { + panic!() + }; + let wire = physical_common::compile_physical_asap_dag(&root).unwrap(); + let scan = wire + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + asap_types::ir::export::PhysicalASAPOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::Scan { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(scan.output_schema.clone()); + let plan = compile( + &wire, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(wire.roots[0].0)], + ) + .unwrap(); + for (rows, expected) in [ + ( + vec![ + vec![Value::Int64(10), Value::Int64(200)], + vec![Value::Int64(20), Value::Int64(500)], + vec![Value::Null, Value::Int64(200)], + ], + Value::Int64(11), + ), + (vec![], Value::Null), + (vec![vec![Value::Null, Value::Int64(200)]], Value::Null), + ] { + let mut sources = DataSources::default(); + sources + .register( + source.clone(), + Arc::new( + MemorySource::new( + schema.clone(), + vec![Batch::try_new(schema.clone(), rows).unwrap()], + ) + .unwrap(), + ), + ) + .unwrap(); + let bound = plan + .instantiate(BTreeMap::from([( + u64::from(scan.id.0), + Box::new(sources.bind(&logical_scan).unwrap()) as Source<'_>, + )])) + .unwrap(); + let mut stream = bound + .execute( + plan.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].len(), 1); + match (&rows[0][0], expected) { + (Value::Null, Value::Null) => {} + (Value::Int64(actual), Value::Int64(expected)) => assert_eq!(*actual, expected), + other => panic!("{other:?}"), + } + } +} + +/// Empty window frames and filtered groups can yield NULL even on non-NULL input. +#[tokio::test] +async fn sql_window_and_filtered_aggregate_types() { + for query in [ + "SELECT SUM(l_quantity) OVER (ORDER BY l_quantity ROWS BETWEEN 2 PRECEDING AND 1 PRECEDING) AS s FROM lineitem", + "SELECT MIN(l_quantity) OVER (ORDER BY l_quantity ROWS BETWEEN 2 PRECEDING AND 1 PRECEDING) AS s FROM lineitem", + "SELECT SUM(l_quantity) FILTER (WHERE l_quantity < 0) AS s FROM lineitem GROUP BY l_quantity", + ] { + let root = lower_sql(query, &catalog(), AccuracyTarget::Exact).await.unwrap(); + root.validate_structure().unwrap(); + assert_eq!(root.schema.fields[0], Field::plain("s", DataType::Int64, true), "{query}"); + } +} + +/// A real query batch retains two result roots and executes both selected +/// plans. Stage 3 prices a query-time SUM state above the raw SUM it would +/// replace, so each plan aggregates its scan directly. No replacement dag is +/// constructed by the test. +#[tokio::test] +async fn batch_planning_selects_and_executes_each_plan() { + use asap_executor::{ + physical_planner::{compile, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_plan_selection::PlanningModels; + use asap_planner::{e2e_plan, FrontendInput, UserInput}; + use asap_types::workload::*; + use std::{collections::BTreeMap, sync::Arc}; + let queries = [ + "SELECT SUM(bytes) + 1 AS result FROM requests", + "SELECT SUM(bytes) * 2 AS result FROM requests", + ]; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::SQL(SqlDialect::DataFusionSQL), + query_batch: Some( + queries + .iter() + .map(|query| BatchEntry { + query: Query((*query).into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 2, + execute_at: None, + time_selection: TimeSelection::default(), + }) + .collect(), + ), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + arrival: DataArrival::AtRest, + ..Default::default() + }), + }; + let catalog = SqlCatalog::new().with_table( + "requests", + Schema::new(vec![Field::plain("bytes", DataType::Float64, false)]), + ); + let output = e2e_plan(UserInput::new( + &workload, + FrontendInput::Sql { catalog: &catalog }, + PlanningModels::builtin(), + )) + .await + .unwrap(); + assert_eq!(output.entry_indices(), [0, 1]); + assert_eq!(output.roots().len(), 2); + let states: Vec<_> = output + .operators() + .into_iter() + .filter(|n| matches!(n.asap(), Some(ASAPOp::SummaryAgg { .. }))) + .collect(); + assert!(states.is_empty(), "Stage 3 selects the raw SUM"); + for (plan, expected) in output.plans.iter().zip([31.0, 60.0]) { + let root = &plan.root; + root.validate_structure().unwrap(); + let wire = physical_common::compile_physical_asap_dag(root).unwrap(); + let scan = wire + .nodes + .iter() + .find(|n| { + matches!( + n.payload, + asap_types::ir::export::PhysicalASAPOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::Scan { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(scan.output_schema.clone()); + let program = compile( + &wire, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(wire.roots[0].0)], + ) + .unwrap(); + let result = physical_common::execute( + &program, + BTreeMap::from([( + u64::from(scan.id.0), + Batch::try_new( + schema, + vec![vec![Value::Float64(10.0)], vec![Value::Float64(20.0)]], + ) + .unwrap(), + )]), + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + ); + let rows: Vec<_> = result[0].iter().flat_map(|batch| batch.rows()).collect(); + assert_eq!(rows.len(), 1); + assert!( + matches!(rows[0][0], Value::Float64(v) if v == expected), + "{:?}", + rows + ); + } + // The batch exports as one physical DAG: a root per query, no state. + let workload_dag = output.execution_timed_dag().unwrap(); + assert_eq!(workload_dag.roots.len(), 2); + assert_ne!(workload_dag.roots[0], workload_dag.roots[1]); + assert_eq!( + workload_dag + .nodes + .iter() + .filter(|n| matches!( + n.payload, + asap_types::ir::export::PhysicalASAPOperatorPayload::SummaryAgg { .. } + )) + .count(), + 0 + ); +} diff --git a/crates/integration-tests/tests/operator_sharing.rs b/crates/integration-tests/tests/operator_sharing.rs new file mode 100644 index 000000000..790c28b67 --- /dev/null +++ b/crates/integration-tests/tests/operator_sharing.rs @@ -0,0 +1,197 @@ +//! Acceptance tests for operator sharing (issue #468): one operator IR +//! before and after ASAP optimization, so non-ASAP operators sit both above +//! and below summary operators, can share inputs with them, and can carry +//! summaries below set operators. +//! +//! Each test drives SQL text through `lower_sql` → `search_workload` → +//! global selection → `assemble_selected_dag`, the pipeline +//! `sql_to_post_asap.rs` uses. + +use std::rc::Rc; + +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_logical_optimizer::search_workload; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::DefaultCostModel; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; +use asap_types::types::AccuracyTarget; + +fn col(name: &str, dtype: DataType) -> Field { + Field::plain(name, dtype, false) +} + +/// TPC-H `lineitem`, as `frontend-sql/tests/data_quality_check/tpch_deequ.rs` +/// declares it (DECIMAL columns as `Float64`, no time index, no keys). +fn catalog() -> SqlCatalog { + SqlCatalog::new().with_table( + "lineitem", + Schema::new(vec![ + col("l_orderkey", DataType::Int64), + col("l_partkey", DataType::Int64), + col("l_suppkey", DataType::Int64), + col("l_linenumber", DataType::Int64), + col("l_quantity", DataType::Float64), + col("l_extendedprice", DataType::Float64), + col("l_discount", DataType::Float64), + col("l_tax", DataType::Float64), + col("l_returnflag", DataType::Utf8), + col("l_linestatus", DataType::Utf8), + col("l_shipdate", DataType::Date), + col("l_commitdate", DataType::Date), + col("l_receiptdate", DataType::Date), + col("l_shipinstruct", DataType::Utf8), + col("l_shipmode", DataType::Utf8), + col("l_comment", DataType::Utf8), + ]), + ) +} + +/// Lower `sql`, search, select with the default cost model and assemble the +/// selected post-ASAP DAG. +async fn plan(sql: &str, accuracy: AccuracyTarget) -> Rc { + let pre = lower_sql(sql, &catalog(), accuracy) + .await + .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")); + let space = search_workload(vec![("query", pre)]); + let selection = global_selection(&space, &DefaultCostModel); + selection + .assemble_selected_dag(&space.roots[0].1) + .expect("materialization failed") + .expect("root must be discovered") +} + +/// Every unique node reachable from `root` whose operator matches `pred`. +fn find_all( + root: &Rc, + pred: impl Fn(&OperatorNode) -> bool, +) -> Vec> { + OperatorNode::reachable(root) + .into_iter() + .filter(|node| pred(node)) + .collect() +} + +fn is_summary_evaluation(node: &OperatorNode) -> bool { + matches!( + node.operator, + Operator::ASAP( + ASAPOp::SummaryEstimate { .. } + | ASAPOp::FinalizeExactAccumulator { .. } + | ASAPOp::EvaluatePopulation { .. } + ) + ) +} + +fn is_scan(node: &OperatorNode) -> bool { + matches!(node.non_asap(), Some(NonASAPOp::Scan { .. })) +} + +/// The first node reached through single-input non-ASAP operators below +/// `node` (inclusive) that is not one: where an operator chain meets a +/// summary or a multi-input operator. +fn through_unary_non_asap(node: &Rc) -> &Rc { + match node.non_asap().map(|op| op.children()) { + Some(children) if children.len() == 1 => through_unary_non_asap(children[0]), + _ => node, + } +} + +// #468 problem 1: the Project above the summary evaluation and the Scan below +// it are both plain NonASAP nodes (no post-ASAP-only wrapper variant). +#[ignore = "planner chooses no summary here: Avg has no summary realization, so the Aggregate stays a logical pass-through"] +#[tokio::test] +async fn project_above_and_scan_below_a_summary_are_both_non_asap_nodes() { + let root = plan( + "WITH metric AS (SELECT avg(CASE WHEN l_quantity BETWEEN 1 AND 50 THEN 1.0 ELSE 0.0 END) \ + AS in_range FROM lineitem) SELECT in_range, in_range = 1.0 AS ok FROM metric", + AccuracyTarget::Exact, + ) + .await; + // The root is the outer SELECT list: a NonASAP Project. + assert!( + matches!(root.operator, Operator::NonASAP(NonASAPOp::Project { .. })), + "root must be the outer Project, got {:?}", + root.operator + ); + // A summary evaluation sits below the Project chain. + let evaluation = through_unary_non_asap(&root); + assert!( + is_summary_evaluation(evaluation), + "the Project chain must read a summary, got {:?}", + evaluation.operator + ); + // Below the summary the Scan is the same NonASAP operator a front end emits. + let scans = find_all(evaluation, is_scan); + assert_eq!(scans.len(), 1, "one lineitem Scan below the summary"); + assert!(!scans[0].is_asap()); + // The flat plan exports (time first, wire 6). + post_asap_dag(&root); +} + +// #468 problem 2: the exact aggregate and the sketch read one shared Scan +// (`Rc::ptr_eq`), not two copies. +#[ignore = "waits for the binding rule splitting multi-measure aggregates"] +#[tokio::test] +async fn exact_aggregate_and_sketch_share_one_scan() { + let root = plan( + "SELECT avg(l_extendedprice), approx_percentile_cont(l_discount, 0.99) FROM lineitem", + AccuracyTarget::Epsilon(0.01), + ) + .await; + let exact = find_all(&root, |node| { + matches!(node.non_asap(), Some(NonASAPOp::Aggregate { .. })) + }); + let sketch = find_all(&root, |node| { + matches!(node.asap(), Some(ASAPOp::SummaryAgg { .. })) + }); + assert_eq!(exact.len(), 1, "one exact Aggregate for avg: {root:?}"); + assert_eq!( + sketch.len(), + 1, + "one sketch SummaryAgg for the percentile: {root:?}" + ); + let scan_under = |node: &Rc| { + let scans = find_all(node, is_scan); + assert_eq!(scans.len(), 1, "one Scan under {:?}", node.operator); + Rc::clone(&scans[0]) + }; + assert!( + Rc::ptr_eq(&scan_under(&exact[0]), &scan_under(&sketch[0])), + "the exact aggregate and the sketch must read one shared Scan" + ); +} + +// #468 problem 3: a summary can sit below a set operator — each side of the +// UNION ALL holds its own SummaryEstimate. +#[tokio::test] +async fn each_side_of_union_all_holds_a_summary_estimate() { + let root = plan( + "SELECT approx_distinct(l_partkey) FROM lineitem \ + UNION ALL SELECT approx_distinct(l_suppkey) FROM lineitem", + AccuracyTarget::Epsilon(0.01), + ) + .await; + // The SQL front end lowers UNION ALL to `SetOp { all: true }`. + let Some(NonASAPOp::SetOp { + all: true, + left, + right, + .. + }) = root.non_asap() + else { + panic!("root must be the UNION ALL SetOp, got {:?}", root.operator) + }; + for side in [left, right] { + let estimates = find_all(side, |node| { + matches!(node.asap(), Some(ASAPOp::SummaryEstimate { .. })) + }); + assert!( + !estimates.is_empty(), + "UNION ALL side has no SummaryEstimate: {:?}", + side.operator + ); + } + post_asap_dag(&root); +} diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs index 93bebd338..3403d27e2 100644 --- a/crates/integration-tests/tests/physical_common/mod.rs +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -1,4 +1,4 @@ -use asap_physical_operators::{ +use asap_executor::{ operators::Operator, physical_planner::{CompiledPhysicalDAG, Source}, runtime::{Limits, RunContext, Scope}, @@ -7,6 +7,7 @@ use asap_physical_operators::{ use futures::{executor::block_on, StreamExt}; use std::collections::BTreeMap; +#[allow(dead_code)] pub fn execute( plan: &CompiledPhysicalDAG, inputs: BTreeMap, @@ -40,3 +41,34 @@ pub fn execute( .await }) } + +#[allow(dead_code)] +pub fn compile_physical_asap_dag( + root: &std::rc::Rc, +) -> Result> { + compile_with( + root, + &asap_types::ir::MaterializationAssignment::all_query_time(), + ) +} + +/// As [`compile_physical_asap_dag`], with every summary maintained at +/// ingestion time, the placement precompute compilation requires. +#[allow(dead_code)] // Not every test binary sharing this module compiles precompute. +pub fn compile_maintained_physical_asap_dag( + root: &std::rc::Rc, +) -> Result> { + compile_with( + root, + &asap_types::ir::MaterializationAssignment::all_ingestion_time(), + ) +} + +fn compile_with( + root: &std::rc::Rc, + assignment: &asap_types::ir::MaterializationAssignment, +) -> Result> { + let root = + asap_types::ir::apply_materialization_timings(root, assignment, &mut Default::default())?; + Ok(asap_types::ir::export::compile_physical_asap_dag(&root)?) +} diff --git a/crates/integration-tests/tests/planner_layering_common/mod.rs b/crates/integration-tests/tests/planner_layering_common/mod.rs new file mode 100644 index 000000000..43bce875e --- /dev/null +++ b/crates/integration-tests/tests/planner_layering_common/mod.rs @@ -0,0 +1,712 @@ +//! Shared helpers for the #509 Examples 2–4 acceptance tests +//! (`planner_layering_example{2,3,4}.rs`). Specs: +//! `docs/design_docs/proposals/planner-layering-example{2,3,4}-acceptance.md`. +//! +//! [`run_stages`] runs the real library pipeline (`plan_stages`: Stage 1 → +//! 2 → 3). The functions under "Pending adapters" read properties the IR +//! cannot express yet (window summaries, materialization, retention). They +//! return today's only possible answer; the implementer of each feature +//! replaces the body with a read of the new IR, as Example 1's stubs were +//! replaced. Tests that depend on them are `#[ignore]`d with the feature. +#![allow(dead_code)] + +use std::collections::{BTreeMap, BTreeSet, HashSet}; + +use asap_physical_optimizer::implementation::physical_candidates::PhysicalCandidate; +use asap_plan_selection::{plan_stages, PlanningModels, Selection}; +use asap_types::ir::export::{ + compile_logical_asap_workload, LogicalASAPDAG, LogicalASAPNodeId, LogicalASAPOperatorPayload, + LogicalASAPQueryRoot, PhysicalASAPDAG, +}; +use asap_types::ir::properties::ExecutionTiming; +use asap_types::ir::schema::{FieldDataType, SketchAlgorithm, SketchParams, SketchStatistic}; +use asap_types::ir::QueryRoot; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, BatchEntry, DataArrival, DataDistribution, DataWorkload, DurationMs, + Evidence, EvidenceSource, LatencyRequirement, PlanningWorkload, Predictability, Query, + QueryLanguage, QueryRecurrence, QueryRequirements, QueryTimeScope, QueryWorkload, Rate, + RepeatedDemand, RepeatingEntry, RepetitionInterval, TimeSelection, TimestampMs, +}; + +/// Every enumerated candidate is built and displayed; the largest example +/// (Example 3, Pattern A) has 486 today. +pub const MAX_CANDIDATES: usize = 4096; + +pub fn declared(value: T) -> Evidence { + Evidence { + value: Some(value), + source: EvidenceSource::Declared, + ..Default::default() + } +} + +// ── Workloads shared by Examples 3 and 4 ──────────────────────────────── + +pub const MINUTE_MS: u64 = 60_000; +/// PromQL's `y` is 365 days. +pub const YEAR_MS: u64 = 365 * 24 * 60 * MINUTE_MS; +/// Pattern A's batch execution time T (2026-01-01T00:00:00Z). +pub const T_MS: u64 = 1_767_225_600_000; + +/// The #509 shared data workload with `arrival`. +pub fn shared_data_workload(arrival: DataArrival) -> DataWorkload { + DataWorkload { + data_ingestion_interval: declared(DurationMs(15_000)), + ingestion_volume: Evidence::default(), + // Data at rest has no ingestion rate (`validate` rejects one). + ingestion_rate: match arrival { + DataArrival::AtRest => Evidence::default(), + _ => declared(Rate(1_000_000.0 / 15.0)), + }, + input_cardinality: declared(1_000_000), + distribution: declared(DataDistribution::Zipf), + arrival, + } +} + +/// Pattern A's five queries: (PromQL, `lookback`, `as_of` − T). +pub const PATTERN_A: [(&str, u64, u64); 5] = [ + ("quantile_over_time(0.99, latency_ms[5y])", 5 * YEAR_MS, 0), + ("quantile_over_time(0.99, latency_ms[1y])", YEAR_MS, 0), + ( + "quantile_over_time(0.99, latency_ms[1y] offset 1y)", + YEAR_MS, + YEAR_MS, + ), + ( + "quantile_over_time(0.99, latency_ms[1y] offset 2y)", + YEAR_MS, + 2 * YEAR_MS, + ), + ( + "quantile_over_time(0.99, latency_ms[3y] offset 2y)", + 3 * YEAR_MS, + 2 * YEAR_MS, + ), +]; + +/// How Pattern A's batch recurs (Example 4 varies it). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum PatternARecurrence { + /// As given: one `ad_hoc` batch run once at T. + OnceAdHoc, + /// Example 4: repeated monthly and `Predictable { known_at }`; each run + /// reads the intervals ending at its own evaluation time. + MonthlyPredictable, +} + +fn pattern_a_requirements() -> QueryRequirements { + QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::EpsilonDelta { + epsilon: 0.005, + delta: 0.01, + }), + response_latency: LatencyRequirement::Unspecified, + } +} + +/// Example 3, Pattern A: p99 over five historical intervals. +pub fn pattern_a(recurrence: PatternARecurrence, arrival: DataArrival) -> PlanningWorkload { + let (batch, repeating) = match recurrence { + PatternARecurrence::OnceAdHoc => ( + Some( + PATTERN_A + .iter() + .map(|&(query, lookback, before_t)| BatchEntry { + query: Query(query.into()), + requirements: pattern_a_requirements(), + predictability: Predictability::AdHoc, + invocations: 1, + execute_at: Some(TimestampMs(T_MS)), + time_selection: TimeSelection { + scope: QueryTimeScope::Longitudinal, + lookback: Some(DurationMs(lookback)), + as_of: Some(TimestampMs(T_MS - before_t)), + }, + }) + .collect(), + ), + None, + ), + PatternARecurrence::MonthlyPredictable => ( + None, + Some( + PATTERN_A + .iter() + .map(|&(query, lookback, _)| RepeatingEntry { + query: Query(query.into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval( + (30 * 24 * 60 * MINUTE_MS) as u32, + )), + requirements: pattern_a_requirements(), + predictability: Predictability::Predictable { + known_at: Some(TimestampMs(T_MS)), + }, + time_selection: TimeSelection { + scope: QueryTimeScope::Longitudinal, + lookback: Some(DurationMs(lookback)), + as_of: None, + }, + }) + .collect(), + ), + ), + }; + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: batch, + repeating_queries: repeating, + }, + data_workload: Some(shared_data_workload(arrival)), + } +} + +pub const PATTERN_B: &str = "quantile_over_time(0.99, latency_ms[5m])"; +pub const PATTERN_B_WINDOW_MS: u64 = 5 * MINUTE_MS; +pub const PATTERN_B_INTERVAL_MS: u64 = MINUTE_MS; + +/// Example 3, Pattern B: a p99 panel over the last 5 min, every minute. +pub fn pattern_b() -> PlanningWorkload { + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: None, + repeating_queries: Some(vec![RepeatingEntry { + query: Query(PATTERN_B.into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval( + PATTERN_B_INTERVAL_MS as u32, + )), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }), + response_latency: LatencyRequirement::ExplicitMaxMs(200.0), + }, + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection { + scope: QueryTimeScope::RealTime, + lookback: Some(DurationMs(PATTERN_B_WINDOW_MS)), + as_of: None, + }, + }]), + }, + data_workload: Some(shared_data_workload(DataArrival::ContinuouslyIngesting)), + } +} + +/// Lower and plan a PromQL workload. +pub fn run_promql(workload: &PlanningWorkload) -> Run { + run_stages(workload, lower_promql(workload)) +} + +/// Whether `form` answers a query window by merging several summaries. +pub fn needs_merge(form: WindowForm, query_window_ms: u64) -> bool { + match form { + WindowForm::None => false, + WindowForm::Sliding { length_ms, .. } => length_ms < query_window_ms, + WindowForm::Tumbling { .. } | WindowForm::ExponentialHistogram { .. } => true, + } +} + +// ── Lowering ───────────────────────────────────────────────────────────── + +/// PromQL workload roots, each series' identity as a column (the row +/// representation per-series state needs, as in Example 1). +pub fn lower_promql(workload: &PlanningWorkload) -> Vec { + asap_frontend_promql::lower_promql_query_workload(workload, 0) + .expect("workload lowers") + .into_iter() + .map(|root| match root { + QueryRoot::Operator(node) => QueryRoot::Operator( + asap_types::ir::schema_support::with_promql_series_identity(&node) + .expect("series identity"), + ), + scalar => scalar, + }) + .collect() +} + +/// The relational operator's wire `kind`, or the aggregate's first measure +/// as `aggregate:`, for each node of one query's Stage 0 DAG, sorted +/// (node ids are post-order; compare as a set of operations). +pub fn stage0_operations(root: &QueryRoot) -> Vec { + let dag = asap_types::ir::export::compile_logical_asap_query(root).expect("compiles"); + let json = serde_json::to_value(&dag).expect("serializes"); + let mut ops: Vec = json["nodes"] + .as_array() + .unwrap() + .iter() + .map(|n| { + let op = &n["payload"]["operator"]; + match op["measures"][0]["kind"].as_str() { + Some(measure) => format!("aggregate:{measure}"), + None => op["kind"].as_str().unwrap_or("?").to_owned(), + } + }) + .collect(); + ops.sort(); + ops +} + +// ── Pipeline ───────────────────────────────────────────────────────────── + +/// One Stage 1 candidate. `query_roots` has one root per workload entry, in +/// `QueryWorkload::entries()` order. +#[derive(Debug, Clone)] +pub struct Logical { + pub id: String, + /// From Pass 2's identical-expression variant. + pub shared_input: bool, + pub dag: LogicalASAPDAG, + pub query_roots: Vec, +} + +/// One Stage 2 candidate, derived from the Stage 1 candidate `from_logical`. +#[derive(Debug, Clone)] +pub struct Physical { + pub id: String, + pub from_logical: String, + pub dag: PhysicalASAPDAG, + pub query_roots: Vec, + pub stage2: PhysicalCandidate, +} + +#[derive(Debug, Clone)] +pub struct Run { + pub stage0: Logical, + pub logical: Vec, + pub physical: Vec, + pub selection: Selection, +} + +impl Run { + pub fn invalid(&self) -> BTreeMap<&str, &str> { + self.selection + .rejected + .iter() + .filter(|r| !r.valid) + .map(|r| (r.id.as_str(), r.reason.as_str())) + .collect() + } + + pub fn cost(&self, id: &str) -> Option { + self.selection.costs.get(id).map(|c| c.total) + } + + pub fn physical(&self, id: &str) -> &Physical { + self.physical.iter().find(|p| p.id == id).expect("id") + } + + pub fn physical_of<'a>(&'a self, logical: &'a Logical) -> impl Iterator { + self.physical + .iter() + .filter(move |p| p.from_logical == logical.id) + } +} + +fn export(roots: &[QueryRoot]) -> (LogicalASAPDAG, Vec) { + let dag = compile_logical_asap_workload(roots).expect("logical export"); + let query_roots = dag + .roots + .iter() + .map(|root| match root { + LogicalASAPQueryRoot::Operator(id) => *id, + LogicalASAPQueryRoot::Scalar(_) => panic!("Examples 2–4 have operator roots"), + }) + .collect(); + (dag, query_roots) +} + +/// Stage 0 → 3 over `roots`, the lowered entries of `workload`, through the +/// library's `plan_stages`. `plan_stages` reads only the accuracy targets +/// and the data workload; recurrence and predictability (which Stage 2 +/// materialization needs, Example 4) are not passed yet. +pub fn run_stages(workload: &PlanningWorkload, roots: Vec) -> Run { + let (dag, query_roots) = export(&roots); + let stage0 = Logical { + id: "S0".into(), + shared_input: false, + dag, + query_roots, + }; + let targets: Vec<_> = workload + .query_workload + .entries() + .map(|entry| Some(entry.requirements.accuracy.target())) + .collect(); + let run = plan_stages( + roots.into_iter().enumerate().collect(), + &targets, + workload.data_workload.as_ref().expect("data workload"), + PlanningModels::builtin(), + MAX_CANDIDATES, + ) + .expect("plans"); + let enumeration = run.enumeration.expect("enumerated"); + assert_eq!( + enumeration.candidates.len(), + enumeration.combinations, + "every candidate is displayed" + ); + let mut logical = Vec::new(); + let mut physical = Vec::new(); + for c in enumeration.candidates { + let p = c.physical.expect("every candidate builds"); + let roots: Vec<_> = c + .logical + .expect("composes") + .into_iter() + .map(|(_, r)| r) + .collect(); + let (dag, query_roots) = export(&roots); + logical.push(Logical { + id: p.from_logical.clone(), + shared_input: c.shared, + dag, + query_roots, + }); + physical.push(Physical { + id: p.id.clone(), + from_logical: p.from_logical.clone(), + dag: p.dag.clone(), + query_roots: p.dag.roots.clone(), + stage2: p, + }); + } + Run { + stage0, + logical, + physical, + selection: enumeration.selection, + } +} + +// ── DAG helpers ────────────────────────────────────────────────────────── + +/// The logical and physical exports share node ids, payloads and edge +/// endpoints; the helpers below read only those. +pub trait ExportedDag { + fn producers(&self, consumer: LogicalASAPNodeId) -> Vec; + fn consumers(&self, producer: LogicalASAPNodeId) -> Vec; + fn node_ids(&self) -> Vec; + fn payload(&self, id: LogicalASAPNodeId) -> &LogicalASAPOperatorPayload; +} + +macro_rules! exported_dag { + ($t:ty) => { + impl ExportedDag for $t { + fn producers(&self, consumer: LogicalASAPNodeId) -> Vec { + let edges = self.edges.iter().filter(|e| e.consumer == consumer); + edges.map(|e| e.producer).collect() + } + fn consumers(&self, producer: LogicalASAPNodeId) -> Vec { + let edges = self.edges.iter().filter(|e| e.producer == producer); + edges.map(|e| e.consumer).collect() + } + fn node_ids(&self) -> Vec { + self.nodes.iter().map(|n| n.id).collect() + } + fn payload(&self, id: LogicalASAPNodeId) -> &LogicalASAPOperatorPayload { + &self + .nodes + .iter() + .find(|n| n.id == id) + .expect("node") + .payload + } + } + }; +} +exported_dag!(LogicalASAPDAG); +exported_dag!(PhysicalASAPDAG); + +/// Every node `root` depends on, including itself. +pub fn closure(dag: &impl ExportedDag, root: LogicalASAPNodeId) -> HashSet { + let mut seen = HashSet::from([root]); + let mut stack = vec![root]; + while let Some(node) = stack.pop() { + for producer in dag.producers(node) { + if seen.insert(producer) { + stack.push(producer); + } + } + } + seen +} + +/// The queries (indexes into `query_roots`) whose result depends on `node`. +pub fn readers( + dag: &impl ExportedDag, + query_roots: &[LogicalASAPNodeId], + node: LogicalASAPNodeId, +) -> BTreeSet { + query_roots + .iter() + .enumerate() + .filter(|(_, &root)| closure(dag, root).contains(&node)) + .map(|(q, _)| q) + .collect() +} + +/// The relational operator's wire `kind` (`"scan"`, `"time_range"`, …). +pub fn relational(payload: &LogicalASAPOperatorPayload) -> Option { + match payload { + LogicalASAPOperatorPayload::Relational { .. } => { + let json = serde_json::to_value(payload).expect("payload serializes"); + json["operator"]["kind"].as_str().map(str::to_owned) + } + _ => None, + } +} + +pub fn is_summary(payload: &LogicalASAPOperatorPayload) -> bool { + matches!( + payload, + LogicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(..), + .. + } | LogicalASAPOperatorPayload::SummaryEstimate { .. } + | LogicalASAPOperatorPayload::SummaryMerge + ) +} + +/// The sketch build nodes (`SummaryAgg` over a sketch family) of `dag`, +/// with their algorithm and parameters. +pub fn sketch_builds( + dag: &impl ExportedDag, +) -> Vec<(LogicalASAPNodeId, SketchAlgorithm, SketchParams)> { + dag.node_ids() + .into_iter() + .filter_map(|id| match dag.payload(id) { + LogicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + } => Some((id, kind.algorithm().clone(), kind.params().clone())), + _ => None, + }) + .collect() +} + +/// The estimation nodes that read `build`, directly or through merges. +pub fn estimates_of( + dag: &impl ExportedDag, + build: LogicalASAPNodeId, +) -> Vec<(LogicalASAPNodeId, SketchStatistic)> { + let mut out = Vec::new(); + let mut stack = vec![build]; + let mut seen = HashSet::new(); + while let Some(node) = stack.pop() { + for consumer in dag.consumers(node) { + if !seen.insert(consumer) { + continue; + } + match dag.payload(consumer) { + LogicalASAPOperatorPayload::SummaryEstimate { query } => { + out.push((consumer, query.clone())) + } + LogicalASAPOperatorPayload::SummaryMerge => stack.push(consumer), + _ => {} + } + } + } + out +} + +/// One query's local option: the sorted sketch algorithms built in its +/// closure, or `"exact"` when it builds none. +pub fn query_option( + dag: &impl ExportedDag, + query_roots: &[LogicalASAPNodeId], + query: usize, +) -> String { + let reach = closure(dag, query_roots[query]); + let mut names: Vec<_> = sketch_builds(dag) + .into_iter() + .filter(|(id, ..)| reach.contains(id)) + .map(|(_, algorithm, _)| format!("{algorithm:?}")) + .collect(); + names.sort(); + if names.is_empty() { + "exact".into() + } else { + names.join("+") + } +} + +/// Nodes reachable from more than one query root. +pub fn cross_query_nodes( + dag: &impl ExportedDag, + query_roots: &[LogicalASAPNodeId], +) -> BTreeSet { + dag.node_ids() + .into_iter() + .filter(|&id| readers(dag, query_roots, id).len() > 1) + .collect() +} + +pub fn assert_valid_and_uniquely_named(run: &Run) { + let ids: BTreeSet<_> = run.logical.iter().map(|c| c.id.as_str()).collect(); + assert_eq!(ids.len(), run.logical.len(), "unique logical ids"); + for c in &run.logical { + c.dag.validate().unwrap_or_else(|e| panic!("{}: {e}", c.id)); + } +} + +/// Stage 2 maps the logical candidates one-to-one onto physical ones. +pub fn assert_stage2_bijection(run: &Run) { + let sources: BTreeSet<_> = run + .physical + .iter() + .map(|p| p.from_logical.as_str()) + .collect(); + let logical: BTreeSet<_> = run.logical.iter().map(|c| c.id.as_str()).collect(); + assert_eq!(sources, logical); + assert_eq!(run.physical.len(), run.logical.len()); +} + +/// One selected id; every other candidate rejected once, with a reason. +pub fn assert_selects_one_and_explains_the_rest(run: &Run) { + let all: BTreeSet<_> = run.physical.iter().map(|p| p.id.clone()).collect(); + let mut accounted: BTreeSet<_> = run + .selection + .rejected + .iter() + .map(|r| r.id.clone()) + .collect(); + assert_eq!( + accounted.len(), + run.selection.rejected.len(), + "rejected once each" + ); + assert!(run.selection.rejected.iter().all(|r| !r.reason.is_empty())); + assert!(accounted.insert(run.selection.selected.clone())); + assert_eq!(accounted, all); +} + +/// The selected plan costs no more than any valid candidate. +pub fn assert_selects_cheapest_valid(run: &Run) { + let invalid = run.invalid(); + let best = run + .cost(&run.selection.selected) + .expect("selected is priced"); + for p in run + .physical + .iter() + .filter(|p| !invalid.contains_key(p.id.as_str())) + { + assert!(best <= run.cost(&p.id).unwrap(), "{} is cheaper", p.id); + } +} + +/// Each priced candidate's cost has one entry per DAG node and `total` is +/// their sum, so a node read by several queries is charged once. +pub fn assert_each_node_charged_once(run: &Run) { + let invalid = run.invalid(); + for p in &run.physical { + if invalid.contains_key(p.id.as_str()) { + assert!(run.cost(&p.id).is_none(), "{} is priced", p.id); + continue; + } + let cost = &run.selection.costs[&p.id]; + let nodes: BTreeSet<_> = p.dag.nodes.iter().map(|n| n.id).collect(); + let charged: BTreeSet<_> = cost.per_node.keys().copied().collect(); + assert_eq!(charged, nodes, "{}", p.id); + let sum: f64 = cost.per_node.values().map(|c| c.cost).sum(); + assert!( + (cost.total - sum).abs() <= 1e-9 * sum.abs().max(1.0), + "{}", + p.id + ); + } +} + +/// The evaluation interval of entry `query`, for a fixed-interval repeating +/// entry. +pub fn evaluation_interval_ms(workload: &PlanningWorkload, query: usize) -> Option { + match &workload.query_workload.entries().nth(query)?.recurrence { + QueryRecurrence::Repeated(RepeatedDemand::FixedInterval(i)) => Some(u64::from(i.0)), + _ => None, + } +} + +pub fn lookback_ms(workload: &PlanningWorkload, query: usize) -> u64 { + let entry = workload.query_workload.entries().nth(query).expect("entry"); + let DurationMs(ms) = entry.time_selection.lookback.expect("lookback"); + ms +} + +// ── Pending adapters ───────────────────────────────────────────────────── + +/// The window summary a summary build node maintains (#509 Pass 2, +/// window-composition rule). +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum WindowForm { + /// Rebuilt from the query window's raw samples (no window summary). + None, + /// Windows of `length_ms` starting every `slide_ms`. + Sliding { length_ms: u64, slide_ms: u64 }, + /// Back-to-back windows of `length_ms`. + Tumbling { length_ms: u64 }, + /// EH buckets covering `horizon_ms` of history. + ExponentialHistogram { horizon_ms: u64 }, +} + +/// Pending (needs Pass 2 window composition): the IR has no window summary +/// yet, so every build is rebuilt from its query window. +pub fn window_form(_dag: &impl ExportedDag, _build: LogicalASAPNodeId) -> WindowForm { + WindowForm::None +} + +/// Stage 2's materialization choice for one node (#509 "Materialization"). +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum Materialization { + /// Runs as data arrives; output stored before any query asks. + IngestionTime, + /// Runs when a query first needs it; output kept for later executions + /// and other queries of the batch. + QueryTimeKept, + /// Runs at query time for each execution; output discarded. + NotMaterialized, +} + +/// Pending (needs Stage 2 materialization): the export records only the +/// execution timing, so a query-time node is never kept. +pub fn materialization(p: &Physical, node: LogicalASAPNodeId) -> Materialization { + let n = p.dag.nodes.iter().find(|n| n.id == node).expect("node"); + match n.output_state.timing { + ExecutionTiming::IngestionTime => Materialization::IngestionTime, + ExecutionTiming::QueryTime => Materialization::NotMaterialized, + } +} + +/// Pending (needs Stage 2 materialization): how long a materialized node's +/// output is kept, in event time. `None` until Stage 2 records retention. +pub fn retention_ms(_p: &Physical, _node: LogicalASAPNodeId) -> Option { + None +} + +pub fn runs_at_ingestion(p: &Physical, node: LogicalASAPNodeId) -> bool { + materialization(p, node) == Materialization::IngestionTime +} + +/// Every node upstream of an ingestion-time node also runs at ingestion time. +pub fn assert_ingestion_upstream_is_ingestion(run: &Run) { + for p in &run.physical { + for n in &p.dag.nodes { + if !runs_at_ingestion(p, n.id) { + continue; + } + for up in closure(&p.dag, n.id) { + assert!( + runs_at_ingestion(p, up), + "{}: {up:?} feeds ingestion-time {:?} at query time", + p.id, + n.id + ); + } + } + } +} diff --git a/crates/integration-tests/tests/planner_layering_example1.rs b/crates/integration-tests/tests/planner_layering_example1.rs new file mode 100644 index 000000000..0d9811b79 --- /dev/null +++ b/crates/integration-tests/tests/planner_layering_example1.rs @@ -0,0 +1,1129 @@ +//! Acceptance tests for #509 "Example 1: Aggregation over dimensions", MVP scope. +//! +//! Spec: `docs/design_docs/proposals/planner-layering-example1-acceptance.md`. +//! Written by the test designer before the Phase C stage APIs existed; the +//! implementer replaced the stubs in [`stages`] with adapters over the real +//! stages. Tests that still fail because the implementation differs from the +//! spec stay `#[ignore]`d with the difference as the reason. +//! +//! MVP scope: Stage 1 = Pass 1 + the identical-expression rule only (no +//! window-composition variants); Stage 2 = physical operator implementation +//! only (no materialization). Counts follow the planner's output (user +//! decision): 1 → 64 → 64 → 1, because Pass 1 also offers exact accumulators +//! and whole-expression top-k sketches (32 combinations) and Pass 2 adds a +//! shared-input variant of each. The doc's 1 → 54 → 156 → 1 needs window +//! composition and materialization. + +use std::collections::{BTreeMap, BTreeSet, HashSet}; + +use asap_plan_selection::PlanningModels; +use asap_types::ir::export::{LogicalASAPDAG, LogicalASAPNodeId, LogicalASAPOperatorPayload}; +use asap_types::ir::schema::state_type::{GroupingStrategy, HydraKind, SketchAlgorithm}; +use asap_types::ir::schema::FieldDataType; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, DataArrival, DataDistribution, DataWorkload, DurationMs, Evidence, + EvidenceSource, LatencyRequirement, PlanningWorkload, Predictability, Query, QueryLanguage, + QueryRequirements, QueryTimeScope, QueryWorkload, Rate, RepeatedDemand, RepeatingEntry, + RepetitionInterval, TimeSelection, +}; + +/// Adapters from the Phase C stage APIs to the shapes these tests were written +/// against. They only convert; every decision is the real stage's. +#[allow(dead_code)] +mod stages { + use std::rc::Rc; + + use super::*; + use asap_plan_selection::{plan_stages, MAX_ENUMERATED_CANDIDATES}; + use asap_types::ir::export::{compile_logical_asap_workload, LogicalASAPQueryRoot}; + use asap_types::ir::{OperatorNode, QueryRoot}; + + /// One whole-workload candidate. `query_roots` holds one root per + /// workload entry, in `QueryWorkload::entries()` order (`[q1, q2]`). + /// `roots` are the in-memory roots the next stage consumes. + #[derive(Debug, Clone)] + pub struct LogicalCandidate { + pub id: String, + pub label: String, + pub dag: LogicalASAPDAG, + pub query_roots: Vec, + pub roots: Vec>, + } + + pub type PhysicalASAPDAG = asap_types::ir::export::PhysicalASAPDAG; + + /// One Stage 2 candidate, derived from exactly one Stage 1 candidate. + #[derive(Debug, Clone)] + pub struct PhysicalCandidate { + pub id: String, + pub from_logical: String, + pub label: String, + pub dag: PhysicalASAPDAG, + pub query_roots: Vec, + pub stage2: asap_physical_optimizer::implementation::physical_candidates::PhysicalCandidate, + } + + /// Whole-workload cost of one physical candidate; `per_node` has one + /// entry per DAG node, so a shared node is charged once. + #[derive(Debug, Clone)] + pub struct CandidateCost { + pub total: f64, + pub per_node: BTreeMap, + } + + /// A candidate Stage 3 did not select. `valid == false` means it failed + /// an accuracy, latency or capability check; `true` means it lost on cost. + #[derive(Debug, Clone)] + pub struct Rejection { + pub id: String, + pub valid: bool, + pub reason: String, + } + + /// Stage 3 output. Costs are keyed by physical candidate id. + #[derive(Debug, Clone)] + pub struct Selection { + pub selected: String, + pub costs: BTreeMap, + pub rejected: Vec, + } + + /// The frontend DAG with each PromQL series' full identity as a column, + /// the row representation per-series state needs at runtime. + fn lower(workload: &PlanningWorkload) -> Vec { + asap_frontend_promql::lower_promql_query_workload(workload, 0) + .expect("Example 1 lowers") + .into_iter() + .map(|root| match root { + QueryRoot::Operator(node) => QueryRoot::Operator( + asap_types::ir::schema_support::with_promql_series_identity(&node) + .expect("series identity"), + ), + QueryRoot::Scalar(_) => panic!("Example 1 has operator roots"), + }) + .collect() + } + + fn candidate(id: String, label: String, roots: Vec) -> LogicalCandidate { + let dag = compile_logical_asap_workload(&roots).expect("logical export"); + let query_roots = dag + .roots + .iter() + .map(|root| match root { + LogicalASAPQueryRoot::Operator(id) => *id, + LogicalASAPQueryRoot::Scalar(_) => panic!("Example 1 has operator roots"), + }) + .collect(); + let roots = roots + .into_iter() + .map(|root| match root { + QueryRoot::Operator(node) => node, + QueryRoot::Scalar(_) => panic!("Example 1 has operator roots"), + }) + .collect(); + LogicalCandidate { + id, + label, + dag, + query_roots, + roots, + } + } + + /// Stage 0: frontends lower every query into one summary-free workload DAG. + pub fn stage0_logical(workload: &PlanningWorkload) -> LogicalCandidate { + candidate("S0".into(), "frontend".into(), lower(workload)) + } + + /// Stage 1: every combination of Pass 1 local alternatives, independent + /// and with the shared input (Pass 2), as the library's stage pipeline + /// enumerates them. Lowers `workload` again: Pass 1 reads the in-memory + /// DAG, not the Stage 0 export. + pub fn stage1_logical_asap( + workload: &PlanningWorkload, + _logical: &LogicalCandidate, + ) -> Vec { + let targets: Vec<_> = workload + .query_workload + .entries() + .map(|entry| Some(entry.requirements.accuracy.target())) + .collect(); + let run = plan_stages( + lower(workload).into_iter().enumerate().collect(), + &targets, + workload.data_workload.as_ref().expect("data workload"), + PlanningModels::builtin(), + MAX_ENUMERATED_CANDIDATES, + ) + .expect("plans"); + let enumeration = run.enumeration.expect("enumerated"); + assert_eq!(enumeration.candidates.len(), enumeration.combinations); + enumeration + .candidates + .into_iter() + .map(|c| { + let physical = c.physical.expect("every Example 1 candidate builds"); + let label = format!("{:?}{}", c.choice, if c.shared { " shared" } else { "" }); + let roots = c.logical.expect("composes"); + candidate( + physical.from_logical, + label, + roots.into_iter().map(|(_, root)| root).collect(), + ) + }) + .collect() + } + + /// Stage 2: physical operator implementation of every logical candidate + /// (no materialization in the MVP). + pub fn stage2_physical( + _workload: &PlanningWorkload, + logical: &[LogicalCandidate], + ) -> Vec { + logical + .iter() + .enumerate() + .map(|(index, l)| { + let mut stage2 = + asap_physical_optimizer::implementation::physical_candidates::stage2_physical( + &l.id, &l.roots, + ) + .unwrap_or_else(|e| panic!("{}: {e}", l.id)); + stage2.id = format!("P{}", index + 1); + stage2.label = l.label.clone(); + PhysicalCandidate { + id: stage2.id.clone(), + from_logical: stage2.from_logical.clone(), + label: stage2.label.clone(), + dag: stage2.dag.clone(), + query_roots: stage2.dag.roots.clone(), + stage2, + } + }) + .collect() + } + + /// Stage 3: reject invalid candidates, cost the rest for the whole + /// workload, select the cheapest. + pub fn stage3_select( + workload: &PlanningWorkload, + physical: &[PhysicalCandidate], + models: PlanningModels<'_>, + ) -> Selection { + let targets: Vec<_> = workload + .query_workload + .entries() + .map(|entry| Some(entry.requirements.accuracy.target())) + .collect(); + let candidates: Vec<_> = physical.iter().map(|p| p.stage2.clone()).collect(); + let selection = asap_plan_selection::stage3_select( + &candidates, + &targets, + workload.data_workload.as_ref().expect("data workload"), + models, + ) + .expect("Stage 3 selects"); + Selection { + selected: selection.selected, + costs: selection + .costs + .into_iter() + .map(|(id, cost)| { + let per_node = cost + .per_node + .into_iter() + .map(|(node, c)| (node, c.cost)) + .collect(); + ( + id, + CandidateCost { + total: cost.total, + per_node, + }, + ) + }) + .collect(), + rejected: selection + .rejected + .into_iter() + .map(|r| Rejection { + id: r.id, + valid: r.valid, + reason: r.reason, + }) + .collect(), + } + } + + pub fn runs_at_ingestion(candidate: &PhysicalCandidate, node: LogicalASAPNodeId) -> bool { + candidate + .dag + .nodes + .iter() + .find(|n| n.id == node) + .expect("node") + .output_state + .timing + == asap_types::ir::properties::ExecutionTiming::IngestionTime + } +} + +use stages::*; + +// ── Example 1 workload ─────────────────────────────────────────────────── + +const Q1: &str = "sum by (job) (rate(http_requests_total[1m]))"; +const Q2: &str = "topk by (job) (10, sum_over_time(http_requests_total[1m]))"; + +fn declared(value: T) -> Evidence { + Evidence { + value: Some(value), + source: EvidenceSource::Declared, + ..Default::default() + } +} + +fn dashboard_panel(query: &str, requirements: QueryRequirements) -> RepeatingEntry { + RepeatingEntry { + query: Query(query.into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval(10_000)), + requirements, + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection { + scope: QueryTimeScope::RealTime, + lookback: Some(DurationMs(60_000)), + as_of: None, + }, + } +} + +/// Example 1 queries over the shared data workload of #509. +fn example1_workload() -> PlanningWorkload { + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: None, + repeating_queries: Some(vec![ + dashboard_panel( + Q1, + QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + response_latency: LatencyRequirement::Unspecified, + }, + ), + dashboard_panel( + Q2, + QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }), + response_latency: LatencyRequirement::ExplicitMaxMs(100.0), + }, + ), + ]), + }, + data_workload: Some(DataWorkload { + arrival: DataArrival::ContinuouslyIngesting, + data_ingestion_interval: declared(DurationMs(15_000)), + ingestion_volume: Evidence::default(), + ingestion_rate: declared(Rate(1_000_000.0 / 15.0)), + input_cardinality: declared(1_000_000), + distribution: declared(DataDistribution::Zipf), + }), + } +} + +// ── DAG helpers ────────────────────────────────────────────────────────── + +/// Q2's local option, read off its summary build node. A whole-expression +/// sketch reads the raw samples of the range, absorbing `sum_over_time`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +enum Q2Option { + Exact, + CountMinHeapPerJob, + CountSketchHeapPerJob, + WholeCountMinHeapPerJob, + WholeCountSketchHeapPerJob, + Hydra, +} + +/// The logical and physical exports share node ids, payloads and edge +/// endpoints; the helpers below read only those. +trait ExportedDag { + fn producers(&self, consumer: LogicalASAPNodeId) -> Vec; + fn node_payload(&self, id: LogicalASAPNodeId) -> &LogicalASAPOperatorPayload; +} + +impl ExportedDag for LogicalASAPDAG { + fn producers(&self, consumer: LogicalASAPNodeId) -> Vec { + let edges = self.edges.iter().filter(|e| e.consumer == consumer); + edges.map(|e| e.producer).collect() + } + fn node_payload(&self, id: LogicalASAPNodeId) -> &LogicalASAPOperatorPayload { + &self + .nodes + .iter() + .find(|n| n.id == id) + .expect("node") + .payload + } +} + +impl ExportedDag for PhysicalASAPDAG { + fn producers(&self, consumer: LogicalASAPNodeId) -> Vec { + let edges = self.edges.iter().filter(|e| e.consumer == consumer); + edges.map(|e| e.producer).collect() + } + fn node_payload(&self, id: LogicalASAPNodeId) -> &LogicalASAPOperatorPayload { + &self + .nodes + .iter() + .find(|n| n.id == id) + .expect("node") + .payload + } +} + +/// Every node `root` depends on, including itself. +fn closure(dag: &impl ExportedDag, root: LogicalASAPNodeId) -> HashSet { + let mut seen = HashSet::from([root]); + let mut stack = vec![root]; + while let Some(node) = stack.pop() { + for producer in dag.producers(node) { + if seen.insert(producer) { + stack.push(producer); + } + } + } + seen +} + +fn payload(dag: &impl ExportedDag, id: LogicalASAPNodeId) -> &LogicalASAPOperatorPayload { + dag.node_payload(id) +} + +fn is_summary(payload: &LogicalASAPOperatorPayload) -> bool { + matches!( + payload, + LogicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(..), + .. + } | LogicalASAPOperatorPayload::SummaryEstimate { .. } + | LogicalASAPOperatorPayload::SummaryMerge + ) +} + +/// Sketch families built in `nodes`, as Example 1's Q2 options. +fn sketch_options( + dag: &impl ExportedDag, + nodes: &HashSet, +) -> BTreeSet { + nodes + .iter() + .filter_map(|&id| match payload(dag, id) { + LogicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(kind, grouping), + .. + } => { + let whole = dag + .producers(id) + .iter() + .any(|&p| relational(payload(dag, p)).as_deref() == Some("time_range")); + let per_job = matches!(grouping, GroupingStrategy::PerSubpopulationInstance); + Some(match (grouping, kind.algorithm(), whole) { + ( + GroupingStrategy::SharedMultiSubpopulation { + kind: HydraKind::HydraCms, + .. + }, + _, + _, + ) => Q2Option::Hydra, + (_, SketchAlgorithm::CmsWithHeap, false) if per_job => { + Q2Option::CountMinHeapPerJob + } + (_, SketchAlgorithm::CmsWithHeap, true) if per_job => { + Q2Option::WholeCountMinHeapPerJob + } + (_, SketchAlgorithm::CountSketchWithHeap, false) if per_job => { + Q2Option::CountSketchHeapPerJob + } + (_, SketchAlgorithm::CountSketchWithHeap, true) if per_job => { + Q2Option::WholeCountSketchHeapPerJob + } + other => panic!("summary family outside Example 1: {other:?}"), + }) + } + _ => None, + }) + .collect() +} + +/// Exact accumulator kinds built in `nodes` (e.g. `["Rate", "Sum"]`). +fn exact_accumulators(dag: &impl ExportedDag, nodes: &HashSet) -> Vec { + let mut kinds: Vec<_> = nodes + .iter() + .filter_map(|&id| match payload(dag, id) { + LogicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::ExactAggregate(kind, _), + .. + } => Some(format!("{kind:?}")), + _ => None, + }) + .collect(); + kinds.sort(); + kinds +} + +fn roots(query_roots: &[LogicalASAPNodeId]) -> (LogicalASAPNodeId, LogicalASAPNodeId) { + assert_eq!(query_roots.len(), 2, "every candidate covers Q1 and Q2"); + (query_roots[0], query_roots[1]) +} + +/// Q2's option and whether Q1 and Q2 share any node, for one candidate. +fn classify(dag: &impl ExportedDag, query_roots: &[LogicalASAPNodeId]) -> (Q2Option, bool) { + let (q1, q2) = roots(query_roots); + let (c1, c2) = (closure(dag, q1), closure(dag, q2)); + let options = sketch_options(dag, &c2); + assert!(options.len() <= 1, "Q2 uses one local option: {options:?}"); + let option = options.into_iter().next().unwrap_or(Q2Option::Exact); + (option, !c1.is_disjoint(&c2)) +} + +/// The relational operator's wire `kind` (`"scan"`, `"sort"`, …); the +/// operator enum itself is not public outside `asap-types`. +fn relational(payload: &LogicalASAPOperatorPayload) -> Option { + match payload { + LogicalASAPOperatorPayload::Relational { .. } => { + let json = serde_json::to_value(payload).expect("payload serializes"); + json["operator"]["kind"].as_str().map(str::to_owned) + } + _ => None, + } +} + +fn pipeline() -> ( + PlanningWorkload, + Vec, + Vec, +) { + let workload = example1_workload(); + let logical = stage1_logical_asap(&workload, &stage0_logical(&workload)); + let physical = stage2_physical(&workload, &logical); + (workload, logical, physical) +} + +/// One candidate's local choices: Q1's exact accumulators, Q2's top-k option +/// and Q2's exact accumulators. +type Choices = (Vec, Q2Option, Vec); + +fn choices(dag: &impl ExportedDag, query_roots: &[LogicalASAPNodeId]) -> Choices { + let (q1, q2) = roots(query_roots); + let (c1, c2) = (closure(dag, q1), closure(dag, q2)); + let (option, _) = classify(dag, query_roots); + ( + exact_accumulators(dag, &c1), + option, + exact_accumulators(dag, &c2), + ) +} + +/// The 32 Pass 1 combinations: Q1's rate and sum each raw or an exact +/// accumulator (4) × (Q2's top-k exact, Count-Min + heap or CountSketch + +/// heap (3) × Q2's sum_over_time raw or an exact accumulator (2), or a +/// whole-expression Count-Min + heap or CountSketch + heap that absorbs +/// sum_over_time (2)). +fn expected_choices() -> BTreeSet { + let q1 = [vec![], vec!["Rate"], vec!["Sum"], vec!["Rate", "Sum"]]; + let q2 = [ + Q2Option::Exact, + Q2Option::CountMinHeapPerJob, + Q2Option::CountSketchHeapPerJob, + ]; + let owned = |kinds: &[&str]| kinds.iter().map(|k| k.to_string()).collect::>(); + let mut all = BTreeSet::new(); + for a in &q1 { + for &option in &q2 { + for b in [vec![], vec!["Sum"]] { + all.insert((owned(a), option, owned(&b))); + } + } + for option in [ + Q2Option::WholeCountMinHeapPerJob, + Q2Option::WholeCountSketchHeapPerJob, + ] { + all.insert((owned(a), option, vec![])); + } + } + all +} + +// ── Workload ───────────────────────────────────────────────────────────── + +/// The encoded workload is valid and normalizes to Q1 then Q2. +#[test] +fn workload_encodes_example1() { + let workload = example1_workload(); + workload.validate().expect("valid workload"); + let entries: Vec<_> = workload.query_workload.entries().collect(); + assert_eq!(entries.len(), 2); + assert_eq!(entries[0].query.0, Q1); + assert_eq!(entries[1].query.0, Q2); +} + +// ── Stage 0 ────────────────────────────────────────────────────────────── + +/// Today's PromQL frontend already lowers each query to the doc's Stage 0 chain. +#[test] +fn stage0_frontend_lowers_each_query_to_doc_chain() { + let roots = asap_frontend_promql::lower_promql_query_workload(&example1_workload(), 0) + .expect("Example 1 lowers"); + let chains: Vec> = roots + .iter() + .map(|root| { + let dag = asap_types::ir::export::compile_logical_asap_query(root).expect("compiles"); + let json = serde_json::to_value(&dag).expect("serializes"); + let mut ops: Vec = json["nodes"] + .as_array() + .unwrap() + .iter() + .map(|n| { + let op = &n["payload"]["operator"]; + match op["measures"][0]["kind"].as_str() { + Some(measure) => format!("aggregate:{measure}"), + None => op["kind"].as_str().unwrap_or("?").to_owned(), + } + }) + .collect(); + ops.sort(); // node ids are assigned in post-order; compare as a set of operations + ops + }) + .collect(); + assert_eq!( + chains, + [ + ["aggregate:rate", "aggregate:sum", "scan", "time_range"], + ["aggregate:sum", "aggregate:top_k", "scan", "time_range"], + ] + ); +} + +/// Stage 0 yields one workload DAG with one root per query and no summaries. +#[test] +fn stage0_one_summary_free_workload_dag() { + let stage0 = stage0_logical(&example1_workload()); + stage0.dag.validate().expect("valid DAG"); + roots(&stage0.query_roots); + assert!(stage0.dag.nodes.iter().all(|n| !is_summary(&n.payload))); +} + +/// Stage 0 keeps Q1 and Q2 separate; sharing is a Stage 1 decision. +#[test] +fn stage0_queries_do_not_share_nodes() { + let stage0 = stage0_logical(&example1_workload()); + let (q1, q2) = roots(&stage0.query_roots); + assert!(closure(&stage0.dag, q1).is_disjoint(&closure(&stage0.dag, q2))); +} + +// ── Stage 1 ────────────────────────────────────────────────────────────── + +/// Stage 1 outputs the 32 Pass 1 combinations twice: L1–L32 with separate +/// inputs, then L33–L64 with the shared input (Pass 2). +#[test] +fn stage1_has_64_candidates_covering_every_combination_twice() { + let (_, logical, _) = pipeline(); + assert_eq!(logical.len(), 64); + for (half, shared) in [(&logical[..32], false), (&logical[32..], true)] { + let found: BTreeSet<_> = half + .iter() + .map(|c| choices(&c.dag, &c.query_roots)) + .collect(); + assert_eq!(found, expected_choices(), "each combination exactly once"); + for c in half { + assert_eq!(classify(&c.dag, &c.query_roots).1, shared, "{}", c.id); + } + } +} + +/// Q1 is exact in every Stage 1 candidate: no summary is reachable from its root. +#[test] +fn stage1_q1_is_always_exact() { + let (_, logical, _) = pipeline(); + for c in &logical { + let (q1, _) = roots(&c.query_roots); + let reach = closure(&c.dag, q1); + assert!( + reach.iter().all(|&id| !is_summary(payload(&c.dag, id))), + "{}: Q1 reaches a summary", + c.id + ); + } +} + +/// Q2's summary families are exactly Count-Min + heap and CountSketch + heap +/// per job, and Hydra over all jobs. +#[test] +#[ignore = "missing feature: Pass 1 has no Hydra alternative"] +fn stage1_q2_summary_families_are_heap_sketches_and_hydra() { + let (_, logical, _) = pipeline(); + let families: BTreeSet<_> = logical + .iter() + .flat_map(|c| { + let (_, q2) = roots(&c.query_roots); + sketch_options(&c.dag, &closure(&c.dag, q2)) + }) + .collect(); + assert_eq!( + families, + BTreeSet::from([ + Q2Option::CountMinHeapPerJob, + Q2Option::CountSketchHeapPerJob, + Q2Option::WholeCountMinHeapPerJob, + Q2Option::WholeCountSketchHeapPerJob, + Q2Option::Hydra + ]) + ); +} + +/// Sharing adds a variant and keeps the independent one, for every Q2 option +/// Pass 1 offers (Hydra: see `stage1_q2_summary_families_are_heap_sketches_and_hydra`). +#[test] +fn stage1_keeps_independent_and_shared_variants() { + let (_, logical, _) = pipeline(); + let found: Vec<_> = logical + .iter() + .map(|c| classify(&c.dag, &c.query_roots)) + .collect(); + for option in [ + Q2Option::Exact, + Q2Option::CountMinHeapPerJob, + Q2Option::CountSketchHeapPerJob, + Q2Option::WholeCountMinHeapPerJob, + Q2Option::WholeCountSketchHeapPerJob, + ] { + assert!( + found.contains(&(option, false)), + "{option:?} independent missing" + ); + assert!(found.contains(&(option, true)), "{option:?} shared missing"); + } +} + +/// Only the raw input is shared between Q1 and Q2; no summary is shared. +#[test] +fn stage1_shares_input_but_never_a_summary() { + let (_, logical, _) = pipeline(); + for c in &logical { + let (q1, q2) = roots(&c.query_roots); + let shared = &closure(&c.dag, q1) & &closure(&c.dag, q2); + for id in shared { + let p = payload(&c.dag, id); + assert!( + matches!(relational(p).as_deref(), Some("scan" | "time_range")), + "{}: shared node {id:?} is not the range selector input: {p:?}", + c.id + ); + } + } +} + +/// Stage 1 candidates are valid DAGs with unique ids. +#[test] +fn stage1_candidates_are_valid_and_uniquely_named() { + let (_, logical, _) = pipeline(); + let ids: BTreeSet<_> = logical.iter().map(|c| c.id.as_str()).collect(); + assert_eq!(ids.len(), logical.len()); + for c in &logical { + c.dag.validate().unwrap_or_else(|e| panic!("{}: {e}", c.id)); + } +} + +// ── Stage 2 ────────────────────────────────────────────────────────────── + +/// No candidate is discarded before Stage 3: Stage 2 maps the 64 logical candidates one-to-one. +#[test] +fn stage2_keeps_every_logical_candidate() { + let (_, logical, physical) = pipeline(); + assert_eq!(physical.len(), 64); + let sources: BTreeSet<_> = physical.iter().map(|p| p.from_logical.as_str()).collect(); + let logical_ids: BTreeSet<_> = logical.iter().map(|c| c.id.as_str()).collect(); + assert_eq!(sources, logical_ids); + let ids: BTreeSet<_> = physical.iter().map(|p| p.id.as_str()).collect(); + assert_eq!(ids.len(), physical.len()); +} + +/// Stage 2 preserves each logical candidate's Q2 option and input sharing. +#[test] +fn stage2_preserves_logical_choices() { + let (_, logical, physical) = pipeline(); + for p in &physical { + let source = logical.iter().find(|c| c.id == p.from_logical).unwrap(); + assert_eq!( + classify(&p.dag, &p.query_roots), + classify(&source.dag, &source.query_roots), + "{} vs {}", + p.id, + source.id + ); + } +} + +/// Exact TopK is implemented as a sort followed by a limit. +#[test] +fn stage2_exact_topk_is_sort_then_limit() { + let (_, _, physical) = pipeline(); + let exact: Vec<_> = physical + .iter() + .filter(|p| classify(&p.dag, &p.query_roots).0 == Q2Option::Exact) + .collect(); + assert_eq!(exact.len(), 16); + for p in exact { + let sort_then_limit = p.dag.edges.iter().any(|e| { + relational(payload(&p.dag, e.producer)).as_deref() == Some("sort") + && relational(payload(&p.dag, e.consumer)).as_deref() == Some("limit") + }); + assert!(sort_then_limit, "{}: no sort → limit", p.id); + } +} + +/// A summary Q2 is a build node feeding a top-10 estimation node, with no merge. +#[test] +fn stage2_summary_topk_is_build_then_estimate() { + let (_, _, physical) = pipeline(); + for p in &physical { + if classify(&p.dag, &p.query_roots).0 == Q2Option::Exact { + continue; + } + let kinds: Vec<_> = p.dag.nodes.iter().map(|n| &n.payload).collect(); + assert!( + !kinds + .iter() + .any(|k| matches!(k, LogicalASAPOperatorPayload::SummaryMerge)), + "{}: no window summaries in the MVP, so no merge", + p.id + ); + let build_to_estimate = p.dag.edges.iter().any(|e| { + matches!( + payload(&p.dag, e.producer), + LogicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(..), + .. + } + ) && matches!( + payload(&p.dag, e.consumer), + LogicalASAPOperatorPayload::SummaryEstimate { .. } + ) + }); + assert!(build_to_estimate, "{}: no build → estimate", p.id); + } +} + +/// With no materialization in the MVP, every node runs at query time. +#[test] +fn stage2_everything_runs_at_query_time() { + let (_, _, physical) = pipeline(); + for p in &physical { + for n in &p.dag.nodes { + assert!( + !runs_at_ingestion(p, n.id), + "{}: {:?} at ingestion", + p.id, + n.id + ); + } + } +} + +/// Compile `p` in the physical planner (the runtime capability check). Inputs +/// are what a deployment supplies: raw series for each sub-DAG the planner +/// runs as a retained PromQL expression, and the samples of each time range a +/// native operator reads. +fn compile_in_runtime(p: &PhysicalCandidate) -> Result<(), String> { + use asap_executor::physical_planner::{compile, promql_fallback, InputContract}; + use asap_types::ir::export::compile_physical_asap_workload_with_node_ids; + use std::sync::Arc; + let ids = compile_physical_asap_workload_with_node_ids(&p.stage2.roots) + .expect("re-export") + .node_ids; + let mut inputs = BTreeMap::new(); + let mut pending = p.dag.roots.clone(); + let mut seen = HashSet::new(); + while let Some(id) = pending.pop() { + if !seen.insert(id) { + continue; + } + let node = ids.operator_node(id).expect("node"); + let time_range = relational(payload(&p.dag, id)).as_deref() == Some("time_range"); + // Raw samples a summary reads are an input, not a retained expression. + let summary_input = time_range + && p.dag.edges.iter().any(|e| { + e.producer == id + && matches!( + payload(&p.dag, e.consumer), + LogicalASAPOperatorPayload::SummaryAgg { .. } + ) + }); + let fallback = (!node.contains_asap() && !summary_input) + .then(|| promql_fallback::raw_series(node).ok()) + .flatten(); + if let Some(selectors) = fallback { + for (i, (_, schema)) in selectors.into_iter().enumerate() { + let slot = promql_fallback::raw_series_input(u64::from(id.0), i); + inputs.insert(slot, InputContract::bounded(schema)); + } + } else if time_range { + let schema = p.dag.nodes.iter().find(|n| n.id == id).unwrap(); + let schema = Arc::new(schema.output_schema.clone()); + inputs.insert(u64::from(id.0), InputContract::bounded(schema)); + } else { + pending.extend(p.dag.producers(id)); + } + } + let roots: Vec = p.dag.roots.iter().map(|id| u64::from(id.0)).collect(); + compile(&p.dag, inputs, &roots) + .map(|_| ()) + .map_err(|e| format!("{} ({}): {e}", p.id, p.label)) +} + +/// Runtime capability check (added by the implementer, not part of the +/// spec): the physical planner compiles every candidate Stage 3 finds valid, +/// and rejects the invalid ones (Count-Min over weights not proven +/// non-negative) for the same reason Stage 3 gives. +#[test] +fn stage2_runtime_compiles_exactly_the_candidates_stage3_finds_valid() { + let (workload, _, physical) = pipeline(); + let selection = stage3_select(&workload, &physical, PlanningModels::builtin()); + let invalid: BTreeMap<_, _> = selection + .rejected + .iter() + .filter(|r| !r.valid) + .map(|r| (r.id.as_str(), r.reason.as_str())) + .collect(); + assert_eq!(invalid.len(), 24, "the Count-Min + heap candidates"); + for p in &physical { + let compiled = compile_in_runtime(p); + match invalid.get(p.id.as_str()) { + None => compiled.unwrap(), + Some(reason) => { + assert!( + reason.contains("CmsWithHeap needs non-negative update weights"), + "{}: {reason}", + p.id + ); + let error = compiled.expect_err(&p.id); + assert!( + error.contains("CMS requires a nonnegative weight contract"), + "{error}" + ); + } + } + } +} + +/// A CountSketch+heap top-k readout compiles: the IR's derived readout schema +/// is the ranked-rows shape the runtime's keyed evaluation produces. +#[test] +fn stage2_count_sketch_heap_topk_compiles_in_the_physical_planner() { + let (_, _, physical) = pipeline(); + let count_sketch: Vec<_> = physical + .iter() + .filter(|p| { + p.dag.nodes.iter().any(|n| { + matches!(&n.payload, LogicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(kind, _), .. + } if *kind.algorithm() == SketchAlgorithm::CountSketchWithHeap) + }) + }) + .collect(); + assert!(!count_sketch.is_empty()); + for p in count_sketch { + compile_in_runtime(p).unwrap(); + } +} + +/// The plan Stage 3 selects compiles in the physical planner (added by the +/// implementer, not part of the spec). +#[test] +fn stage3_selected_plan_compiles_in_the_physical_planner() { + let (workload, _, physical) = pipeline(); + let selection = stage3_select(&workload, &physical, PlanningModels::builtin()); + let selected = physical + .iter() + .find(|p| p.id == selection.selected) + .unwrap(); + compile_in_runtime(selected).unwrap(); +} + +// ── Stage 3 ────────────────────────────────────────────────────────────── + +/// Stage 3 selects one candidate and gives every other one a reason. +#[test] +fn stage3_selects_one_and_explains_the_rest() { + let (workload, _, physical) = pipeline(); + let selection = stage3_select(&workload, &physical, PlanningModels::builtin()); + let all: BTreeSet<_> = physical.iter().map(|p| p.id.clone()).collect(); + let mut accounted: BTreeSet<_> = selection.rejected.iter().map(|r| r.id.clone()).collect(); + assert_eq!( + accounted.len(), + selection.rejected.len(), + "rejected once each" + ); + assert!(selection.rejected.iter().all(|r| !r.reason.is_empty())); + assert!( + accounted.insert(selection.selected.clone()), + "selected is not rejected" + ); + assert_eq!(accounted, all); +} + +/// The selected plan is the cheapest valid candidate for the whole workload. +#[test] +fn stage3_selects_cheapest_valid() { + let (workload, _, physical) = pipeline(); + let selection = stage3_select(&workload, &physical, PlanningModels::builtin()); + let invalid: BTreeSet<_> = selection + .rejected + .iter() + .filter(|r| !r.valid) + .map(|r| r.id.as_str()) + .collect(); + let best = selection.costs[&selection.selected].total; + for p in physical.iter().filter(|p| !invalid.contains(p.id.as_str())) { + assert!(best <= selection.costs[&p.id].total, "{} is cheaper", p.id); + } +} + +/// Every node is charged exactly once, so a shared input is costed once for +/// both queries. Stage 3 prices valid candidates only (user decision). +#[test] +fn stage3_charges_each_node_once() { + let (workload, _, physical) = pipeline(); + let selection = stage3_select(&workload, &physical, PlanningModels::builtin()); + let invalid: BTreeSet<_> = selection + .rejected + .iter() + .filter(|r| !r.valid) + .map(|r| r.id.as_str()) + .collect(); + for p in &physical { + if invalid.contains(p.id.as_str()) { + assert!(!selection.costs.contains_key(&p.id), "{} is priced", p.id); + continue; + } + let cost = &selection.costs[&p.id]; + let nodes: BTreeSet<_> = p.dag.nodes.iter().map(|n| n.id).collect(); + let charged: BTreeSet<_> = cost.per_node.keys().copied().collect(); + assert_eq!( + charged, nodes, + "{}: per-node costs cover each node once", + p.id + ); + let sum: f64 = cost.per_node.values().sum(); + assert!( + (cost.total - sum).abs() <= 1e-9 * sum.abs().max(1.0), + "{}", + p.id + ); + } +} + +/// Sharing the input never costs more than reading it separately, for the +/// same local choices. Count-Min + heap is invalid here and has no cost +/// (Hydra: see `stage1_q2_summary_families_are_heap_sketches_and_hydra`). +#[test] +fn stage3_shared_input_is_not_costlier() { + let (workload, _, physical) = pipeline(); + let selection = stage3_select(&workload, &physical, PlanningModels::builtin()); + let by_combo: BTreeMap<_, _> = physical + .iter() + .filter_map(|p| { + let cost = selection.costs.get(&p.id)?.total; + let shared = classify(&p.dag, &p.query_roots).1; + Some(((choices(&p.dag, &p.query_roots), shared), cost)) + }) + .collect(); + for option in [ + Q2Option::Exact, + Q2Option::CountSketchHeapPerJob, + Q2Option::WholeCountSketchHeapPerJob, + ] { + assert!( + by_combo.keys().any(|((_, o, _), _)| *o == option), + "{option:?} has no priced candidate" + ); + } + for ((choice, shared), cost) in &by_combo { + if *shared { + let separate = by_combo[&(choice.clone(), false)]; + assert!(*cost <= separate, "{choice:?}"); + } + } + assert!( + by_combo.keys().any(|(_, shared)| *shared), + "no shared variant" + ); +} + +/// The selected plan shares the input, the doc's "Raw with a shared input" +/// winner; its saving over the same choices read separately is exactly one +/// scan and one range node, priced once instead of twice. +#[test] +fn stage3_selects_a_shared_input_plan() { + let (workload, _, physical) = pipeline(); + let selection = stage3_select(&workload, &physical, PlanningModels::builtin()); + let selected = physical + .iter() + .find(|p| p.id == selection.selected) + .unwrap(); + assert!(classify(&selected.dag, &selected.query_roots).1); + let separate = physical + .iter() + .find(|p| { + !classify(&p.dag, &p.query_roots).1 + && choices(&p.dag, &p.query_roots) == choices(&selected.dag, &selected.query_roots) + }) + .unwrap(); + let input_cost: f64 = selected + .dag + .nodes + .iter() + .filter(|n| { + matches!( + relational(&n.payload).as_deref(), + Some("scan" | "time_range") + ) + }) + .map(|n| selection.costs[&selected.id].per_node[&n.id]) + .sum(); + let saving = selection.costs[&separate.id].total - selection.costs[&selected.id].total; + assert!( + (saving - input_cost).abs() < 1e-9, + "{saving} vs {input_cost}" + ); +} + +/// Every Q2 realization, exact or sketch, whole-expression or not, returns +/// the same selected-rows schema (#579) in Stage 1, so consumers see one +/// shape. (Stage 2 exports exact top-k as sort → limit over the per-series +/// rows, whose schema still carries `ts`.) +#[test] +fn q2_roots_keep_one_schema_across_realizations() { + let (_, logical, _) = pipeline(); + let schemas: Vec<_> = logical + .iter() + .map(|c| { + ( + classify(&c.dag, &c.query_roots).0, + c.roots[1].schema.clone(), + ) + }) + .collect(); + assert!(schemas + .iter() + .any(|(o, _)| *o == Q2Option::WholeCountSketchHeapPerJob)); + for (option, schema) in &schemas { + assert_eq!(schema, &schemas[0].1, "{option:?}"); + } +} diff --git a/crates/integration-tests/tests/planner_layering_example2.rs b/crates/integration-tests/tests/planner_layering_example2.rs new file mode 100644 index 000000000..6aa59e72c --- /dev/null +++ b/crates/integration-tests/tests/planner_layering_example2.rs @@ -0,0 +1,547 @@ +//! Acceptance tests for #509 "Example 2: One summary for several +//! computations — the summary-capability rule in Pass 2". +//! +//! Spec: `docs/design_docs/proposals/planner-layering-example2-acceptance.md`. +//! Written by a test designer who does not implement the stages. Tests that +//! need an unimplemented feature are `#[ignore]`d, naming it. +//! +//! The doc's workload is SQL. The SQL frontend lowers Q1 to a distinct count +//! but not Q2 and Q3 to `Entropy` and `L2` (the doc's own TODO), so Stages +//! 1–3 run on a PromQL stand-in with the same structure: three statistics +//! of one input over the same 1-min window, through the frontend's +//! `distinct_over_time`, `entropy_over_time` and `l2_over_time`. + +mod planner_layering_common; + +use std::collections::BTreeSet; + +use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; +use asap_types::ir::schema::{ + DataType, Field, Schema, SketchAlgorithm, SketchParams, SketchStatistic, +}; +use asap_types::ir::QueryRoot; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, DataArrival, DataDistribution, DataWorkload, DurationMs, + LatencyRequirement, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, + QueryTimeScope, QueryWorkload, Rate, RepeatedDemand, RepeatingEntry, RepetitionInterval, + SqlDialect, TimeSelection, +}; +use planner_layering_common::*; + +// ── Example 2 workload ─────────────────────────────────────────────────── + +const SQL_Q1: &str = + "SELECT COUNT(DISTINCT src_ip) FROM flows WHERE ts >= now() - INTERVAL '1 minute'"; +const SQL_Q2: &str = "SELECT -SUM(p * LN(p)) FROM (\ + SELECT COUNT(*) * 1.0 / SUM(COUNT(*)) OVER () AS p FROM flows \ + WHERE ts >= now() - INTERVAL '1 minute' GROUP BY src_ip)"; +const SQL_Q3: &str = "SELECT SQRT(SUM(c * c)) FROM (\ + SELECT src_ip, COUNT(*) AS c FROM flows \ + WHERE ts >= now() - INTERVAL '1 minute' GROUP BY src_ip)"; + +/// PromQL stand-in for Q1–Q3 (see the module docs). +const PROMQL_Q1: &str = "distinct_over_time(flows_src_ip[1m])"; +const PROMQL_Q2: &str = "entropy_over_time(flows_src_ip[1m])"; +const PROMQL_Q3: &str = "l2_over_time(flows_src_ip[1m])"; + +/// (ε, δ) of Q1, Q2, Q3. Q3 is the strictest. +const ACCURACY: [(f64, f64); 3] = [(0.02, 0.01), (0.05, 0.01), (0.01, 0.01)]; + +fn panel(query: &str, (epsilon, delta): (f64, f64)) -> RepeatingEntry { + RepeatingEntry { + query: Query(query.into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval(10_000)), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::EpsilonDelta { + epsilon, + delta, + }), + response_latency: LatencyRequirement::Unspecified, + }, + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection { + scope: QueryTimeScope::RealTime, + lookback: Some(DurationMs(60_000)), + as_of: None, + }, + } +} + +/// The shared data workload with 10,000,000 distinct source IPs. +fn data_workload(ingestion_interval: bool) -> DataWorkload { + DataWorkload { + arrival: DataArrival::ContinuouslyIngesting, + data_ingestion_interval: if ingestion_interval { + declared(DurationMs(15_000)) + } else { + Default::default() + }, + ingestion_volume: Default::default(), + ingestion_rate: declared(Rate(1_000_000.0 / 15.0)), + input_cardinality: declared(10_000_000), + distribution: declared(DataDistribution::Zipf), + } +} + +fn workload(language: QueryLanguage, queries: [&str; 3]) -> PlanningWorkload { + let sql = matches!(language, QueryLanguage::SQL(_)); + PlanningWorkload { + query_workload: QueryWorkload { + language, + query_batch: None, + repeating_queries: Some( + queries + .iter() + .zip(ACCURACY) + .map(|(q, accuracy)| panel(q, accuracy)) + .collect(), + ), + }, + // `data_ingestion_interval` is not needed for SQL; PromQL needs it. + data_workload: Some(data_workload(!sql)), + } +} + +/// Example 2 as the doc writes it. +fn sql_workload() -> PlanningWorkload { + workload( + QueryLanguage::SQL(SqlDialect::DataFusionSQL), + [SQL_Q1, SQL_Q2, SQL_Q3], + ) +} + +fn standin_workload() -> PlanningWorkload { + workload(QueryLanguage::PromQL, [PROMQL_Q1, PROMQL_Q2, PROMQL_Q3]) +} + +/// `flows(ts, src_ip)`: the doc names only these two columns. +fn catalog() -> SqlCatalog { + SqlCatalog::new().with_table( + "flows", + Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("src_ip", DataType::Utf8, false), + ], + 0, + vec![], + ), + ) +} + +async fn lower_sql(workload: &PlanningWorkload) -> Vec { + let mut roots = Vec::new(); + for entry in workload.query_workload.entries() { + let root = lower_sql_dialect( + &entry.query.0, + &catalog(), + SqlDialect::DataFusionSQL, + entry.requirements.accuracy.target(), + ) + .await + .unwrap_or_else(|e| panic!("{}: {e}", entry.query.0)); + roots.push(QueryRoot::Operator(root)); + } + roots +} + +fn pipeline() -> Run { + let workload = standin_workload(); + run_stages(&workload, lower_promql(&workload)) +} + +// ── Helpers ────────────────────────────────────────────────────────────── + +/// UnivMon build nodes of `c` with the queries that read each. +fn univmons(c: &Logical) -> Vec<(BTreeSet, SketchParams, BTreeSet)> { + sketch_builds(&c.dag) + .into_iter() + .filter(|(_, algorithm, _)| *algorithm == SketchAlgorithm::UnivMon) + .map(|(id, _, params)| { + let statistics = estimates_of(&c.dag, id) + .into_iter() + .map(|(_, s)| format!("{s:?}")) + .collect(); + (readers(&c.dag, &c.query_roots, id), params, statistics) + }) + .collect() +} + +/// The UnivMon read by every query in `queries` and by no other. +fn shared_univmon(c: &Logical, queries: &[usize]) -> Option { + let want: BTreeSet<_> = queries.iter().copied().collect(); + univmons(c) + .into_iter() + .find(|(readers, ..)| *readers == want) + .map(|(_, params, _)| params) +} + +/// Whether some summary build is read by more than one query. +fn shares_a_summary(c: &Logical) -> bool { + sketch_builds(&c.dag) + .iter() + .any(|(id, ..)| readers(&c.dag, &c.query_roots, *id).len() > 1) +} + +fn options(run: &Run, query: usize) -> BTreeSet { + run.logical + .iter() + .filter(|c| !shares_a_summary(c)) + .map(|c| query_option(&c.dag, &c.query_roots, query)) + .collect() +} + +/// UnivMon state size: heap entries plus counters over every layer. +fn footprint(params: &SketchParams) -> u64 { + let SketchParams::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + } = params + else { + panic!("UnivMon params: {params:?}"); + }; + u64::from(*heap_size) + u64::from(*sketch_rows) * u64::from(*sketch_cols) * u64::from(*layers) +} + +/// Q`query`'s own UnivMon parameters, from an independent candidate. +fn own_univmon(run: &Run, query: usize) -> SketchParams { + run.logical + .iter() + .filter(|c| !shares_a_summary(c)) + .find_map(|c| shared_univmon(c, &[query])) + .expect("Pass 1 offers UnivMon") +} + +// ── Workload and Stage 0 ───────────────────────────────────────────────── + +/// The encoded SQL workload is valid and normalizes to Q1, Q2, Q3. +#[test] +fn workload_encodes_example2() { + let workload = sql_workload(); + workload.validate().expect("valid workload"); + let queries: Vec<_> = workload + .query_workload + .entries() + .map(|e| e.query.0) + .collect(); + assert_eq!(queries, [SQL_Q1, SQL_Q2, SQL_Q3]); + standin_workload().validate().expect("valid stand-in"); +} + +/// The SQL frontend lowers Q1 to a distinct count over `flows`. +#[tokio::test] +async fn stage0_sql_q1_lowers_to_a_distinct_count() { + let roots = lower_sql(&sql_workload()).await; + let ops = stage0_operations(&roots[0]); + assert!(ops.contains(&"scan".to_string()), "{ops:?}"); + assert!( + ops.contains(&"aggregate:cardinality".to_string()), + "{ops:?}" + ); +} + +/// The SQL frontend recognizes Q2 as `Entropy(src_ip)` and Q3 as `L2(src_ip)`. +#[tokio::test] +#[ignore = "needs SQL frontend recognition of the Entropy and L2 forms (doc TODO)"] +async fn stage0_sql_q2_q3_lower_to_entropy_and_l2() { + let roots = lower_sql(&sql_workload()).await; + assert!(stage0_operations(&roots[1]).contains(&"aggregate:frequency_entropy".to_string())); + assert!(stage0_operations(&roots[2]).contains(&"aggregate:frequency_l2".to_string())); +} + +/// Once the SQL forms are recognized, the SQL workload has the stand-in's Pass 1 options. +#[tokio::test] +#[ignore = "needs SQL frontend recognition of the Entropy and L2 forms (doc TODO)"] +async fn stage1_sql_has_the_standin_options() { + let workload = sql_workload(); + let sql = run_stages(&workload, lower_sql(&workload).await); + let standin = pipeline(); + for q in 0..3 { + assert_eq!(options(&sql, q), options(&standin, q), "Q{}", q + 1); + } +} + +/// The stand-in lowers to the three statistics, each over a 1-min range of one input. +#[test] +fn stage0_standin_lowers_to_three_statistics_of_one_input() { + let roots = lower_promql(&standin_workload()); + let ops: Vec<_> = roots.iter().map(stage0_operations).collect(); + for (ops, statistic) in ops.iter().zip([ + "aggregate:cardinality", + "aggregate:frequency_entropy", + "aggregate:frequency_l2", + ]) { + let mut want = vec!["scan".to_string(), statistic.into(), "time_range".into()]; + want.sort(); + assert_eq!(ops, &want); + } +} + +/// Stage 0 is one summary-free workload DAG whose queries share no node. +#[test] +fn stage0_one_summary_free_dag_without_sharing() { + let run = pipeline(); + let s0 = &run.stage0; + s0.dag.validate().expect("valid DAG"); + assert_eq!(s0.query_roots.len(), 3); + assert!(s0.dag.nodes.iter().all(|n| !is_summary(&n.payload))); + assert!(cross_query_nodes(&s0.dag, &s0.query_roots).is_empty()); +} + +// ── Stage 1 ────────────────────────────────────────────────────────────── + +/// Pass 1 offers an exact and a UnivMon option for each of the three statistics. +#[test] +fn stage1_pass1_offers_exact_and_univmon_for_each_statistic() { + let run = pipeline(); + for q in 0..3 { + let found = options(&run, q); + assert!(found.contains("exact"), "Q{}: {found:?}", q + 1); + assert!(found.contains("UnivMon"), "Q{}: {found:?}", q + 1); + } +} + +/// Pass 1 offers a specialized distinct-count summary for Q1. +#[test] +fn stage1_pass1_offers_a_distinct_count_summary() { + let found = options(&pipeline(), 0); + assert!( + found + .iter() + .any(|o| ["Hll", "Theta", "Kmv"].contains(&o.as_str())), + "{found:?}" + ); +} + +/// Pass 1 offers a specialized entropy summary for Q2 and a norm summary for Q3. +#[test] +#[ignore = "needs Pass 1 specialized entropy and L2 summary families"] +fn stage1_pass1_offers_specialized_entropy_and_l2_summaries() { + let run = pipeline(); + for q in [1, 2] { + let found = options(&run, q); + assert!( + found.iter().any(|o| o != "exact" && o != "UnivMon"), + "Q{}: {found:?}", + q + 1 + ); + } +} + +/// Every combination of the three queries' own options is kept as an independent candidate. +#[test] +fn stage1_keeps_every_independent_combination() { + let run = pipeline(); + let found: BTreeSet<_> = run + .logical + .iter() + .filter(|c| !shares_a_summary(c)) + .map(|c| { + (0..3) + .map(|q| query_option(&c.dag, &c.query_roots, q)) + .collect::>() + }) + .collect(); + let (o1, o2, o3) = (options(&run, 0), options(&run, 1), options(&run, 2)); + assert_eq!(found.len(), o1.len() * o2.len() * o3.len()); +} + +/// The summary-capability rule adds one UnivMon build feeding a distinct-count, an entropy and an L2 estimate. +/// (Passes today through the identical-expression rule: Pass 1 gives every +/// UnivMon the same parameters, so the shared-input variant merges them.) +#[test] +fn stage1_summary_capability_adds_one_univmon_for_all_three() { + let run = pipeline(); + let shared = run.logical.iter().find(|c| { + let ums = univmons(c); + ums.len() == 1 && ums[0].0 == BTreeSet::from([0, 1, 2]) + }); + let c = shared.expect("a candidate with one UnivMon read by Q1, Q2 and Q3"); + let statistics = &univmons(c)[0].2; + for s in [ + SketchStatistic::Cardinality, + SketchStatistic::FrequencyEntropy, + SketchStatistic::FrequencyL2, + ] { + assert!(statistics.contains(&format!("{s:?}")), "{statistics:?}"); + } +} + +/// For each pair, a candidate shares one UnivMon between the two while the third keeps each of its own options. +#[test] +#[ignore = "needs Pass 2 summary-capability rule"] +fn stage1_summary_capability_adds_pairwise_shared_univmons() { + let run = pipeline(); + for (pair, third) in [([0, 1], 2), ([0, 2], 1), ([1, 2], 0)] { + let thirds: BTreeSet<_> = run + .logical + .iter() + .filter(|c| shared_univmon(c, &pair).is_some()) + .map(|c| query_option(&c.dag, &c.query_roots, third)) + .collect(); + assert_eq!(thirds, options(&run, third), "pair {pair:?}"); + } +} + +/// A UnivMon shared by several statistics is sized for the strictest consumer. +/// (Holds today only because UnivMon parameters do not depend on ε.) +#[test] +fn stage1_shared_univmon_is_sized_for_the_strictest_consumer() { + let run = pipeline(); + let strictest = |queries: &[usize]| { + *queries + .iter() + .min_by(|a, b| ACCURACY[**a].0.total_cmp(&ACCURACY[**b].0)) + .unwrap() + }; + let mut seen = 0; + for queries in [vec![0, 1, 2], vec![0, 1], vec![0, 2], vec![1, 2]] { + for c in &run.logical { + if let Some(params) = shared_univmon(c, &queries) { + seen += 1; + assert_eq!(params, own_univmon(&run, strictest(&queries)), "{}", c.id); + } + } + } + assert!(seen > 0, "no shared UnivMon"); +} + +/// Pass 1 sizes UnivMon for its query's target: ε = 0.01 (Q3) needs a larger state than ε = 0.05 (Q2). +#[test] +#[ignore = "needs UnivMon sizing for an accuracy target (Pass 1 uses fixed parameters)"] +fn stage1_univmon_is_sized_per_accuracy_target() { + let run = pipeline(); + assert!(footprint(&own_univmon(&run, 2)) > footprint(&own_univmon(&run, 1))); +} + +/// Only a UnivMon, never a specialized summary, is shared across the statistics. +#[test] +fn stage1_only_univmon_is_shared_across_statistics() { + let run = pipeline(); + for c in &run.logical { + for (id, algorithm, _) in sketch_builds(&c.dag) { + if readers(&c.dag, &c.query_roots, id).len() > 1 { + assert_eq!(algorithm, SketchAlgorithm::UnivMon, "{}", c.id); + } + } + } +} + +/// Stage 1 candidates are valid DAGs with unique ids. +#[test] +fn stage1_candidates_are_valid_and_uniquely_named() { + assert_valid_and_uniquely_named(&pipeline()); +} + +// ── Stage 2 ────────────────────────────────────────────────────────────── + +/// No candidate is discarded before Stage 3: Stage 2 maps logical candidates one-to-one. +#[test] +fn stage2_keeps_every_logical_candidate() { + assert_stage2_bijection(&pipeline()); +} + +/// Stage 2 keeps the shared UnivMon as one build node read by all three queries. +#[test] +fn stage2_keeps_the_shared_univmon_as_one_build() { + let run = pipeline(); + let mut seen = 0; + for c in run + .logical + .iter() + .filter(|c| shared_univmon(c, &[0, 1, 2]).is_some()) + { + for p in run.physical_of(c) { + seen += 1; + let builds: Vec<_> = sketch_builds(&p.dag) + .into_iter() + .filter(|(_, a, _)| *a == SketchAlgorithm::UnivMon) + .collect(); + assert_eq!(builds.len(), 1, "{}", p.id); + assert_eq!(readers(&p.dag, &p.query_roots, builds[0].0).len(), 3); + } + } + assert!(seen > 0); +} + +// ── Stage 3 ────────────────────────────────────────────────────────────── + +/// Stage 3 selects one candidate and gives every other one a reason. +#[test] +fn stage3_selects_one_and_explains_the_rest() { + assert_selects_one_and_explains_the_rest(&pipeline()); +} + +/// The selected plan is the cheapest valid candidate for the whole workload. +#[test] +fn stage3_selects_cheapest_valid() { + assert_selects_cheapest_valid(&pipeline()); +} + +/// Every node is charged once, so a summary read by several queries is costed once. +#[test] +fn stage3_charges_each_node_once() { + assert_each_node_charged_once(&pipeline()); +} + +/// UnivMon candidates are judged by an accuracy model instead of being rejected for lacking one. +#[test] +#[ignore = "needs a UnivMon accuracy model in Stage 3"] +fn stage3_judges_univmon_with_an_accuracy_model() { + let run = pipeline(); + for (id, reason) in run.invalid() { + assert!( + !reason.contains("no accuracy model for UnivMon"), + "{id}: {reason}" + ); + } +} + +/// One UnivMon for all three costs no more than three separate UnivMons. +#[test] +#[ignore = "needs a UnivMon accuracy model in Stage 3"] +fn stage3_shared_univmon_costs_no_more_than_three() { + let run = pipeline(); + let cost = |c: &Logical| run.cost(&run.physical_of(c).next().unwrap().id); + let separate: Vec<_> = run + .logical + .iter() + .filter(|c| { + !shares_a_summary(c) + && (0..3).all(|q| query_option(&c.dag, &c.query_roots, q) == "UnivMon") + }) + .map(|c| cost(c).unwrap_or_else(|| panic!("{} is not priced", c.id))) + .collect(); + assert!(!separate.is_empty(), "independent all-UnivMon candidate"); + let mut compared = 0; + for shared in run + .logical + .iter() + .filter(|c| shared_univmon(c, &[0, 1, 2]).is_some()) + { + let shared_cost = cost(shared).unwrap_or_else(|| panic!("{} is not priced", shared.id)); + for separate_cost in &separate { + assert!( + shared_cost <= *separate_cost, + "{shared_cost} vs {separate_cost}" + ); + } + compared += 1; + } + assert!(compared > 0); +} + +/// The selected plan updates at most one UnivMon per flow record: separate UnivMons are dominated by the shared one. +#[test] +fn stage3_selected_plan_has_at_most_one_univmon() { + let run = pipeline(); + let selected = run.physical(&run.selection.selected); + let count = sketch_builds(&selected.dag) + .iter() + .filter(|(_, a, _)| *a == SketchAlgorithm::UnivMon) + .count(); + assert!(count <= 1, "{}: {count} UnivMons", selected.id); +} diff --git a/crates/integration-tests/tests/planner_layering_example3.rs b/crates/integration-tests/tests/planner_layering_example3.rs new file mode 100644 index 000000000..5a21555e3 --- /dev/null +++ b/crates/integration-tests/tests/planner_layering_example3.rs @@ -0,0 +1,442 @@ +//! Acceptance tests for #509 "Example 3: Aggregation over windows — the +//! window-composition rule in Pass 2". +//! +//! Spec: `docs/design_docs/proposals/planner-layering-example3-acceptance.md`. +//! Written by a test designer who does not implement the stages. Tests that +//! need an unimplemented feature are `#[ignore]`d, naming it. Window +//! summaries are read through `planner_layering_common::window_form`, which +//! the window-composition implementer fills in. + +mod planner_layering_common; + +use std::collections::BTreeSet; + +use asap_types::ir::export::{LogicalASAPNodeId, LogicalASAPOperatorPayload}; +use asap_types::ir::schema::{SketchAlgorithm, SketchParams, SketchStatistic}; +use asap_types::workload::DataArrival; +use planner_layering_common::*; + +fn run_a() -> Run { + run_promql(&pattern_a( + PatternARecurrence::OnceAdHoc, + DataArrival::Mixed, + )) +} + +fn run_b() -> Run { + run_promql(&pattern_b()) +} + +/// The local options of `query` across all candidates. +fn options(run: &Run, query: usize) -> BTreeSet { + run.logical + .iter() + .map(|c| query_option(&c.dag, &c.query_roots, query)) + .collect() +} + +fn op(name: &str) -> String { + name.to_string() +} + +/// The summary builds of `c`: (build, algorithm, window form, readers). +fn windowed_builds( + c: &Logical, +) -> Vec<( + LogicalASAPNodeId, + SketchAlgorithm, + WindowForm, + BTreeSet, +)> { + sketch_builds(&c.dag) + .into_iter() + .map(|(id, algorithm, _)| { + let form = window_form(&c.dag, id); + (id, algorithm, form, readers(&c.dag, &c.query_roots, id)) + }) + .collect() +} + +// ── Workloads ──────────────────────────────────────────────────────────── + +/// Pattern A is one ad hoc batch of five queries run once at T; Pattern B one repeating query. +#[test] +fn workload_encodes_example3() { + let a = pattern_a(PatternARecurrence::OnceAdHoc, DataArrival::Mixed); + a.validate().expect("valid Pattern A"); + let queries: Vec<_> = a.query_workload.entries().map(|e| e.query.0).collect(); + assert_eq!(queries, PATTERN_A.map(|(q, ..)| q)); + let b = pattern_b(); + b.validate().expect("valid Pattern B"); + assert_eq!(evaluation_interval_ms(&b, 0), Some(PATTERN_B_INTERVAL_MS)); + assert_eq!(lookback_ms(&b, 0), PATTERN_B_WINDOW_MS); +} + +// ── Pattern A: Stage 0 ─────────────────────────────────────────────────── + +/// Each Pattern A query lowers to scan → (time shift) → range → quantile. +#[test] +fn stage0_a_lowers_each_query_to_its_interval() { + let roots = lower_promql(&pattern_a( + PatternARecurrence::OnceAdHoc, + DataArrival::Mixed, + )); + let ops: Vec<_> = roots.iter().map(stage0_operations).collect(); + let plain = vec![op("aggregate:quantile"), op("scan"), op("time_range")]; + let shifted = vec![ + op("aggregate:quantile"), + op("scan"), + op("time_range"), + op("time_shift"), + ]; + assert_eq!( + ops, + [ + plain.clone(), + plain, + shifted.clone(), + shifted.clone(), + shifted + ] + ); +} + +/// Stage 0 is one summary-free DAG whose five queries share no node. +#[test] +fn stage0_a_one_summary_free_dag_without_sharing() { + let run = run_a(); + let s0 = &run.stage0; + s0.dag.validate().expect("valid DAG"); + assert_eq!(s0.query_roots.len(), 5); + assert!(s0.dag.nodes.iter().all(|n| !is_summary(&n.payload))); + assert!(cross_query_nodes(&s0.dag, &s0.query_roots).is_empty()); +} + +// ── Pattern A: Stage 1 ─────────────────────────────────────────────────── + +/// Pass 1 offers each query an exact and a KLL candidate over its own interval. +#[test] +fn stage1_a_pass1_offers_exact_and_kll_per_query() { + let run = run_a(); + for q in 0..5 { + let found = options(&run, q); + assert!( + found.contains("exact") && found.contains("Kll"), + "q{}: {found:?}", + q + 1 + ); + } +} + +/// The five independent KLLs: one per query, each read by its query only, all sized for ε = 0.005. +#[test] +fn stage1_a_keeps_five_independent_klls() { + let run = run_a(); + let all_kll = run + .logical + .iter() + .find(|c| { + !c.shared_input && (0..5).all(|q| query_option(&c.dag, &c.query_roots, q) == "Kll") + }) + .expect("independent all-KLL candidate"); + let builds = sketch_builds(&all_kll.dag); + assert_eq!(builds.len(), 5); + let readers: BTreeSet<_> = builds + .iter() + .map(|(id, ..)| readers(&all_kll.dag, &all_kll.query_roots, *id)) + .collect(); + assert_eq!(readers, (0..5).map(|q| BTreeSet::from([q])).collect()); + let sizes: BTreeSet<_> = builds + .iter() + .map(|(_, _, params)| match params { + SketchParams::Kll { k } => *k, + other => panic!("{other:?}"), + }) + .collect(); + assert_eq!(sizes.len(), 1, "one ε, one KLL size: {sizes:?}"); +} + +/// The identical-expression rule shares only raw input (scan, range, shift), never a quantile or summary, and keeps the unshared variant. +#[test] +fn stage1_a_identical_expression_rule_shares_only_raw_input() { + let run = run_a(); + assert!(run.logical.iter().any(|c| c.shared_input)); + assert!(run.logical.iter().any(|c| !c.shared_input)); + for c in &run.logical { + for id in cross_query_nodes(&c.dag, &c.query_roots) { + let kind = relational(c.dag.payload(id)); + assert!( + matches!(kind.as_deref(), Some("scan" | "time_range" | "time_shift")), + "{}: shared {kind:?}", + c.id + ); + } + } +} + +/// One Exponential Histogram of KLLs over [T − 5y, T] serves all five queries, each through its own merge and p99 estimate. +#[test] +#[ignore = "needs Pass 2 window composition (#580): Exponential Histogram"] +fn stage1_a_window_composition_adds_one_eh_for_all_five() { + let run = run_a(); + let found = run.logical.iter().find(|c| { + windowed_builds(c).iter().any(|(_, a, form, readers)| { + *a == SketchAlgorithm::Kll + && matches!(form, WindowForm::ExponentialHistogram { horizon_ms } if *horizon_ms >= 5 * YEAR_MS) + && readers.len() == 5 + }) + }); + let c = found.expect("a candidate with one EH of KLLs read by all five queries"); + let (eh, ..) = windowed_builds(c) + .into_iter() + .find(|(_, _, form, _)| matches!(form, WindowForm::ExponentialHistogram { .. })) + .unwrap(); + let estimates = estimates_of(&c.dag, eh); + assert_eq!(estimates.len(), 5, "one estimate per query"); + for (estimate, statistic) in estimates { + assert_eq!(statistic, SketchStatistic::Quantile { q: 0.99 }); + let merged = c + .dag + .producers(estimate) + .into_iter() + .any(|p| matches!(c.dag.payload(p), LogicalASAPOperatorPayload::SummaryMerge)); + assert!( + merged, + "{}: estimate {estimate:?} does not read a merge", + c.id + ); + } +} + +/// The shared EH candidate is added next to the independent KLL candidates, not instead of them. +#[test] +#[ignore = "needs Pass 2 window composition (#580): Exponential Histogram"] +fn stage1_a_keeps_independent_and_shared_window_summaries() { + let run = run_a(); + let shared = run.logical.iter().any(|c| { + windowed_builds(c).iter().any(|(_, _, f, r)| { + matches!(f, WindowForm::ExponentialHistogram { .. }) && r.len() == 5 + }) + }); + let independent = run.logical.iter().any(|c| { + let builds = windowed_builds(c); + builds.len() == 5 + && builds.iter().all(|(_, a, f, r)| { + *a == SketchAlgorithm::Kll && *f == WindowForm::None && r.len() == 1 + }) + }); + assert!(shared && independent); +} + +/// Pass 2 adds a candidate for each way of grouping two queries onto one shared window summary. +#[test] +#[ignore = "needs Pass 2 window composition (#580): partial groupings"] +fn stage1_a_window_composition_groups_every_pair() { + let run = run_a(); + let groups: BTreeSet<_> = run + .logical + .iter() + .flat_map(windowed_builds) + .filter(|(_, _, f, r)| *f != WindowForm::None && r.len() > 1) + .map(|(.., r)| r) + .collect(); + for a in 0..5 { + for b in a + 1..5 { + assert!( + groups.contains(&BTreeSet::from([a, b])), + "pair q{} q{}", + a + 1, + b + 1 + ); + } + } +} + +// ── Pattern A: Stage 3 ─────────────────────────────────────────────────── + +/// Stage 3 selects one Pattern A candidate, explains the rest, and picks the cheapest valid one. +#[test] +fn stage3_a_selects_cheapest_valid() { + let run = run_a(); + assert_selects_one_and_explains_the_rest(&run); + assert_selects_cheapest_valid(&run); +} + +/// Every node is charged once, so a node read by several queries is costed once. +#[test] +fn stage3_a_charges_each_node_once() { + assert_each_node_charged_once(&run_a()); +} + +/// Sharing the scan never costs more than scanning separately, for the same local choices. +#[test] +#[ignore = "needs Stage 3 row estimates for time-shifted scans and for a time range narrower than its input"] +fn stage3_a_shared_scan_is_not_costlier() { + let run = run_a(); + let key = |c: &Logical| { + (0..5) + .map(|q| query_option(&c.dag, &c.query_roots, q)) + .collect::>() + }; + let cost = |c: &Logical| run.cost(&run.physical_of(c).next().unwrap().id); + let mut compared = 0; + for shared in run.logical.iter().filter(|c| c.shared_input) { + let separate = run + .logical + .iter() + .find(|c| !c.shared_input && key(c) == key(shared)) + .expect("independent counterpart"); + if let (Some(s), Some(i)) = (cost(shared), cost(separate)) { + assert!(s <= i, "{} {s} vs {} {i}", shared.id, separate.id); + compared += 1; + } + } + assert!(compared > 0); +} + +// ── Pattern B: Stage 0 and Pass 1 ──────────────────────────────────────── + +/// Pattern B lowers to scan → range 5m → quantile, with no summary. +#[test] +fn stage0_b_lowers_to_one_range_quantile() { + let run = run_b(); + let roots = lower_promql(&pattern_b()); + assert_eq!( + stage0_operations(&roots[0]), + [op("aggregate:quantile"), op("scan"), op("time_range")] + ); + assert!(run.stage0.dag.nodes.iter().all(|n| !is_summary(&n.payload))); +} + +/// Pass 1 offers an exact and a KLL candidate for the 5-min window. +#[test] +fn stage1_b_pass1_offers_exact_and_kll() { + let found = options(&run_b(), 0); + assert!( + found.contains("exact") && found.contains("Kll"), + "{found:?}" + ); +} + +// ── Pattern B: window composition ──────────────────────────────────────── + +/// Each Pass 1 option gets three window forms: none, a 5-min sliding window with a 1-min slide, and 1-min tumbling windows. +#[test] +#[ignore = "needs Pass 2 window composition (#580): sliding and tumbling windows"] +fn stage1_b_window_composition_adds_sliding_and_tumbling_per_option() { + let run = run_b(); + let summaries: BTreeSet<_> = options(&run, 0) + .into_iter() + .filter(|o| o != "exact") + .collect(); + let found: BTreeSet<_> = run + .logical + .iter() + .flat_map(windowed_builds) + .map(|(_, a, f, _)| (format!("{a:?}"), f)) + .collect(); + let forms = [ + WindowForm::None, + WindowForm::Sliding { + length_ms: PATTERN_B_WINDOW_MS, + slide_ms: PATTERN_B_INTERVAL_MS, + }, + WindowForm::Tumbling { + length_ms: PATTERN_B_INTERVAL_MS, + }, + ]; + for option in &summaries { + for form in forms { + assert!(found.contains(&(option.clone(), form)), "{option} {form:?}"); + } + } +} + +/// Window parameters are legal: L divides W, s divides L and the evaluation interval, a tumbling length divides both. +#[test] +fn stage1_b_window_parameters_are_legal() { + let (w, every) = (PATTERN_B_WINDOW_MS, PATTERN_B_INTERVAL_MS); + for c in &run_b().logical { + for (_, _, form, _) in windowed_builds(c) { + match form { + WindowForm::Sliding { + length_ms, + slide_ms, + } => { + assert_eq!(w % length_ms, 0, "{}: L | W", c.id); + assert_eq!(length_ms % slide_ms, 0, "{}: s | L", c.id); + assert_eq!(every % slide_ms, 0, "{}: s | interval", c.id); + } + WindowForm::Tumbling { length_ms } => { + assert_eq!(w % length_ms, 0, "{}: length | W", c.id); + assert_eq!(every % length_ms, 0, "{}: length | interval", c.id); + } + WindowForm::ExponentialHistogram { .. } => { + panic!("{}: Pattern B has no EH", c.id) + } + WindowForm::None => {} + } + } + } +} + +/// A sliding window with L = W reads one completed window (no merge); tumbling windows merge W / length of them. +#[test] +#[ignore = "needs Pass 2 window composition (#580): sliding and tumbling windows"] +fn stage1_b_merge_only_where_the_window_form_needs_it() { + let run = run_b(); + let mut seen = BTreeSet::new(); + for c in &run.logical { + for (build, _, form, _) in windowed_builds(c) { + let merged = estimates_of(&c.dag, build).iter().any(|(e, _)| { + c.dag + .producers(*e) + .into_iter() + .any(|p| matches!(c.dag.payload(p), LogicalASAPOperatorPayload::SummaryMerge)) + }); + assert_eq!( + merged, + needs_merge(form, PATTERN_B_WINDOW_MS), + "{}: {form:?}", + c.id + ); + if form != WindowForm::None { + seen.insert(merged); + } + } + } + assert_eq!( + seen, + BTreeSet::from([false, true]), + "both a sliding and a tumbling candidate" + ); +} + +/// A window form that merges is used only with a summary whose states merge (KLL, DDSketch). +#[test] +fn stage1_window_merges_use_mergeable_summaries() { + for (run, window) in [(run_a(), 5 * YEAR_MS), (run_b(), PATTERN_B_WINDOW_MS)] { + for c in &run.logical { + for (_, algorithm, form, _) in windowed_builds(c) { + if needs_merge(form, window) { + assert!( + matches!(algorithm, SketchAlgorithm::Kll | SketchAlgorithm::DDSketch), + "{}: {algorithm:?} {form:?}", + c.id + ); + } + } + } + } +} + +// ── Pattern B: Stage 3 ─────────────────────────────────────────────────── + +/// Stage 3 selects one Pattern B candidate, explains the rest, and picks the cheapest valid one. +#[test] +fn stage3_b_selects_cheapest_valid() { + let run = run_b(); + assert_selects_one_and_explains_the_rest(&run); + assert_selects_cheapest_valid(&run); + assert_each_node_charged_once(&run); +} diff --git a/crates/integration-tests/tests/planner_layering_example4.rs b/crates/integration-tests/tests/planner_layering_example4.rs new file mode 100644 index 000000000..4d88724ca --- /dev/null +++ b/crates/integration-tests/tests/planner_layering_example4.rs @@ -0,0 +1,340 @@ +//! Acceptance tests for #509 "Example 4: Materialization of window summaries +//! in physical planning". +//! +//! Spec: `docs/design_docs/proposals/planner-layering-example4-acceptance.md`. +//! Written by a test designer who does not implement the stages. Tests that +//! need an unimplemented feature are `#[ignore]`d, naming it. Window +//! summaries, materialization and retention are read through the pending +//! adapters in `planner_layering_common`, which the implementers fill in. +//! +//! The workloads are Example 3's, with Pattern A's recurrence and data +//! arrival varied as the doc does. + +mod planner_layering_common; + +use std::collections::{BTreeMap, BTreeSet}; + +use asap_types::ir::export::{LogicalASAPNodeId, LogicalASAPOperatorPayload}; +use asap_types::ir::schema::SketchAlgorithm; +use asap_types::workload::{DataArrival, PlanningWorkload}; +use planner_layering_common::*; + +/// Pattern A as given: one ad hoc batch at T over mixed data. +fn once() -> PlanningWorkload { + pattern_a(PatternARecurrence::OnceAdHoc, DataArrival::Mixed) +} + +fn monthly() -> PlanningWorkload { + pattern_a(PatternARecurrence::MonthlyPredictable, DataArrival::Mixed) +} + +fn at_rest() -> PlanningWorkload { + pattern_a(PatternARecurrence::OnceAdHoc, DataArrival::AtRest) +} + +/// The window-summary build of `c` with form `want` (`Some(form)` matches +/// exactly; `None` matches any EH) read by `readers` queries. +fn window_build( + c: &Logical, + want: Option, + readers_count: usize, +) -> Option { + sketch_builds(&c.dag) + .into_iter() + .find_map(|(id, algorithm, _)| { + let form = window_form(&c.dag, id); + let matches = match want { + Some(want) => form == want, + None => matches!(form, WindowForm::ExponentialHistogram { .. }), + }; + (algorithm == SketchAlgorithm::Kll + && matches + && readers(&c.dag, &c.query_roots, id).len() == readers_count) + .then_some(id) + }) +} + +/// The physical candidates of the logical candidate with a KLL window +/// summary of `form` read by `readers_count` queries, keyed by the +/// materialization of that build. Several logical candidates may match (for +/// example with and without the shared input); the first is used. +fn options_of( + run: &Run, + form: Option, + readers_count: usize, +) -> BTreeMap { + let logical = run + .logical + .iter() + .find(|c| window_build(c, form, readers_count).is_some()) + .unwrap_or_else(|| { + let form = form.map_or("EH".to_string(), |f| format!("{f:?}")); + panic!("no logical candidate with a {form} KLL window summary") + }); + let build = window_build(logical, form, readers_count).unwrap(); + // Stage 2 keeps node ids from the logical export only where it does not + // rewrite; find the build again in each physical DAG. + run.physical_of(logical) + .map(|p| { + let id = sketch_builds(&p.dag) + .into_iter() + .find(|(id, a, _)| { + *a == SketchAlgorithm::Kll + && window_form(&p.dag, *id) == window_form(&logical.dag, build) + }) + .map(|(id, ..)| id) + .expect("the window summary survives Stage 2"); + (materialization(p, id), (p.id.clone(), id)) + }) + .collect() +} + +fn eh_options(run: &Run) -> BTreeMap { + options_of(run, None, 5) +} + +fn tumbling() -> Option { + Some(WindowForm::Tumbling { + length_ms: PATTERN_B_INTERVAL_MS, + }) +} + +fn sliding() -> Option { + Some(WindowForm::Sliding { + length_ms: PATTERN_B_WINDOW_MS, + slide_ms: PATTERN_B_INTERVAL_MS, + }) +} + +fn cost(run: &Run, id: &str) -> f64 { + run.cost(id).unwrap_or_else(|| panic!("{id} is not priced")) +} + +use Materialization::{IngestionTime, NotMaterialized, QueryTimeKept}; + +// ── Constraints over every candidate ───────────────────────────────────── + +/// Every node upstream of an ingestion-time node also runs at ingestion time, in every workload variant. +#[test] +fn stage2_ingestion_time_upstream_is_ingestion_time() { + for workload in [once(), monthly(), at_rest(), pattern_b()] { + assert_ingestion_upstream_is_ingestion(&run_promql(&workload)); + } +} + +/// Data at rest has no ingestion, so no candidate runs anything at ingestion time. +#[test] +fn stage2_at_rest_runs_nothing_at_ingestion_time() { + let run = run_promql(&at_rest()); + for p in &run.physical { + for n in &p.dag.nodes { + assert!(!runs_at_ingestion(p, n.id), "{}: {:?}", p.id, n.id); + } + } +} + +/// A materialized output is kept for as long as its consumers read: it covers every reader's lookback and offset. +#[test] +fn stage2_materialized_output_covers_its_consumers() { + let offsets: Vec = PATTERN_A.iter().map(|(_, _, before_t)| *before_t).collect(); + for (workload, offsets) in [ + (once(), offsets.clone()), + (monthly(), offsets.clone()), + (at_rest(), offsets), + (pattern_b(), vec![0]), + ] { + let run = run_promql(&workload); + for p in &run.physical { + for n in &p.dag.nodes { + if materialization(p, n.id) == NotMaterialized { + continue; + } + let need = readers(&p.dag, &p.query_roots, n.id) + .into_iter() + .map(|q| lookback_ms(&workload, q) + offsets[q]) + .max() + .unwrap_or(0); + let kept = retention_ms(p, n.id).unwrap_or_else(|| { + panic!("{}: {:?} is materialized without a retention", p.id, n.id) + }); + assert!(kept >= need, "{}: {:?} kept {kept} < {need}", p.id, n.id); + } + } + } +} + +// ── Pattern A ──────────────────────────────────────────────────────────── + +/// The shared EH candidate yields A1 (query time, kept), A2 (ingestion time) and A3 (not materialized). +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage2_a_shared_eh_has_three_materialization_options() { + let found: BTreeSet<_> = eh_options(&run_promql(&once())).into_keys().collect(); + assert_eq!( + found, + BTreeSet::from([IngestionTime, QueryTimeKept, NotMaterialized]) + ); +} + +/// With data at rest, A2 is not generated; A1 and A3 remain. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage2_a_at_rest_drops_the_ingestion_time_option() { + let found: BTreeSet<_> = eh_options(&run_promql(&at_rest())).into_keys().collect(); + assert_eq!(found, BTreeSet::from([QueryTimeKept, NotMaterialized])); +} + +/// A materialized shared EH is one build node read by all five queries, charged once. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage2_a_materialized_eh_is_built_once_for_all_consumers() { + let run = run_promql(&once()); + for (m, (id, build)) in eh_options(&run) { + if m == NotMaterialized { + continue; + } + let p = run.physical(&id); + let ehs = sketch_builds(&p.dag) + .into_iter() + .filter(|(b, ..)| { + matches!( + window_form(&p.dag, *b), + WindowForm::ExponentialHistogram { .. } + ) + }) + .count(); + assert_eq!(ehs, 1, "{id}"); + assert_eq!(readers(&p.dag, &p.query_roots, build).len(), 5, "{id}"); + assert!( + run.selection.costs[&id].per_node.contains_key(&build), + "{id}" + ); + } +} + +/// A3 rebuilds the EH for each of the five queries, so it costs more than A1, which builds it once. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage3_a_rebuilding_per_query_costs_more_than_building_once() { + let run = run_promql(&once()); + let options = eh_options(&run); + let (a1, a3) = (&options[&QueryTimeKept].0, &options[&NotMaterialized].0); + assert!( + cost(&run, a3) > cost(&run, a1), + "A3 {} vs A1 {}", + cost(&run, a3), + cost(&run, a1) + ); +} + +/// As given (run once, ad hoc), A1 is the cheapest of the three: A2 maintains years of history for one batch. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage3_a_once_adhoc_prefers_the_query_time_eh() { + let run = run_promql(&once()); + let options = eh_options(&run); + let a1 = cost(&run, &options[&QueryTimeKept].0); + for (m, (id, _)) in &options { + assert!(a1 <= cost(&run, id), "A1 {a1} vs {m:?} {}", cost(&run, id)); + } +} + +/// Repeated monthly and predictable, A2's maintenance is shared by many batches, so A2 gains on A1. +#[test] +#[ignore = "needs Pass 2 window composition (#580), Stage 2 materialization and recurrence in plan_stages"] +fn stage3_a_monthly_amortizes_ingestion_time_maintenance() { + let ratio = |workload: PlanningWorkload| { + let run = run_promql(&workload); + let options = eh_options(&run); + cost(&run, &options[&IngestionTime].0) / cost(&run, &options[&QueryTimeKept].0) + }; + let (once, monthly) = (ratio(once()), ratio(monthly())); + assert!(monthly < once, "A2/A1: monthly {monthly} vs once {once}"); +} + +// ── Pattern B ──────────────────────────────────────────────────────────── + +/// The 1-min tumbling KLL candidate yields B1 (ingestion time), B2 (not materialized) and B3 (query time, kept). +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage2_b_tumbling_kll_has_three_materialization_options() { + let found: BTreeSet<_> = options_of(&run_promql(&pattern_b()), tumbling(), 1) + .into_keys() + .collect(); + assert_eq!( + found, + BTreeSet::from([IngestionTime, QueryTimeKept, NotMaterialized]) + ); +} + +/// B1 builds the tumbling KLLs at ingestion time and merges and estimates at query time. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage2_b_b1_builds_at_ingestion_and_merges_at_query_time() { + let run = run_promql(&pattern_b()); + let (id, build) = options_of(&run, tumbling(), 1)[&IngestionTime].clone(); + let p = run.physical(&id); + assert!(runs_at_ingestion(p, build)); + let estimates = estimates_of(&p.dag, build); + assert!(!estimates.is_empty()); + for (estimate, _) in estimates { + assert!( + !runs_at_ingestion(p, estimate), + "{id}: estimate at ingestion" + ); + let merge = p + .dag + .producers(estimate) + .into_iter() + .find(|&m| matches!(p.dag.payload(m), LogicalASAPOperatorPayload::SummaryMerge)); + let merge = merge.unwrap_or_else(|| panic!("{id}: no merge before the estimate")); + assert!(!runs_at_ingestion(p, merge), "{id}: merge at ingestion"); + } +} + +/// B3 runs nothing at ingestion time; it keeps the tumbling KLLs it builds at query time. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage2_b_b3_keeps_query_time_windows() { + let run = run_promql(&pattern_b()); + let (id, _) = options_of(&run, tumbling(), 1)[&QueryTimeKept].clone(); + let p = run.physical(&id); + assert!( + p.dag.nodes.iter().all(|n| !runs_at_ingestion(p, n.id)), + "{id}" + ); +} + +/// The sliding-window KLL is kept from ingestion time or from query time; not materializing it is the no-window plan. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage2_b_sliding_kll_has_two_materialization_options() { + let found: BTreeSet<_> = options_of(&run_promql(&pattern_b()), sliding(), 1) + .into_keys() + .collect(); + assert_eq!(found, BTreeSet::from([IngestionTime, QueryTimeKept])); +} + +/// B2 rebuilds all five tumbling KLLs at every evaluation, so it costs at least B1 and B3. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage3_b_rebuilding_every_window_costs_most() { + let run = run_promql(&pattern_b()); + let options = options_of(&run, tumbling(), 1); + let b2 = cost(&run, &options[&NotMaterialized].0); + for m in [IngestionTime, QueryTimeKept] { + assert!(cost(&run, &options[&m].0) <= b2, "{m:?} vs B2 {b2}"); + } +} + +/// Repeating over arriving data, the built-in models pick B1 among the tumbling options. +#[test] +#[ignore = "needs Pass 2 window composition (#580) and Stage 2 materialization"] +fn stage3_b_prefers_ingestion_time_tumbling_windows() { + let run = run_promql(&pattern_b()); + let options = options_of(&run, tumbling(), 1); + let b1 = cost(&run, &options[&IngestionTime].0); + for (m, (id, _)) in &options { + assert!(b1 <= cost(&run, id), "B1 {b1} vs {m:?} {}", cost(&run, id)); + } +} diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index 9937469cc..d51358cb3 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -1,14 +1,12 @@ //! Planner-selected summaries over raw samples compile as precompute DAGs //! and produce the same estimates as feeding their kernel sample by sample. +mod physical_common; +use asap_types::ir::export::{PhysicalASAPDAG, PhysicalASAPOperatorPayload}; +use asap_types::ir::OperatorNode; +use physical_common::compile_maintained_physical_asap_dag; use std::{collections::BTreeMap, collections::BTreeSet, rc::Rc, sync::Arc}; -use asap_aware_mapping::cost_model::DefaultCostModel; -use asap_aware_mapping::{ - search_workload, Replacement, ReplacementStrategy, ReplacementSubDAG, SketchAlgorithmStrategy, - TargetSubDAG, -}; -use asap_integration_tests::fixtures::lower_promql; -use asap_physical_operators::{ +use asap_executor::{ factory::create_planner_accumulator, operators::Operator, physical_planner::{precompute, Source}, @@ -17,12 +15,19 @@ use asap_physical_operators::{ values::{Batch, Value}, AggregateCore, KeyByLabelValues, Statistic, }; -use asap_types::post_asap::{ - compile_post_asap_dag, EntityIdentity, ExactKind, FieldDataType, PostAsapDAG, - PostAsapOperatorPayload, SketchAlgorithm, SketchStatistic, SummaryInputExpr, SummaryNode, +use asap_integration_tests::fixtures::lower_promql; +use asap_logical_optimizer::{ + search_workload, ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, + TargetSubDAG, +}; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::cost::cost_model::DefaultCostModel; +use asap_types::ir::operator::Reduction; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{ + EntityIdentity, ExactKind, FieldDataType, SketchAlgorithm, SketchStatistic, SummaryInputExpr, SummaryUpdate, }; -use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; use asap_types::types::AccuracyTarget; use futures::{executor::block_on, StreamExt}; @@ -51,23 +56,22 @@ fn canonical(labels: &Series) -> Series { /// Every Planner candidate for `query`: the searched selection plus each /// summary replacement of the root. -fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { - let root = Rc::new(lower_promql(query, accuracy).expect("lowering failed")); - let mut result = SketchAlgorithmStrategy::default_cost_model() +fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { + let root = lower_promql(query, accuracy).expect("lowering failed"); + let mut result = ASAPStrategies::default() .replacements(&TargetSubDAG::new(&root)) .into_iter() .filter_map(|candidate| match candidate { ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. } => Some(node), _ => None, }) .collect::>(); let space = search_workload(vec![("query", root)]); - if let Ok(Some(selected)) = space - .global_selection(&DefaultCostModel) - .assemble_selected_dag(&space.roots[0].1) + if let Ok(Some(selected)) = + global_selection(&space, &DefaultCostModel).assemble_selected_dag(&space.roots[0].1) { result.push(selected); } @@ -75,10 +79,10 @@ fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { } /// Raw-input summary nodes: `(dag, raw source id, summary id)`. -fn raw_summaries(dag: &PostAsapDAG) -> Vec<(u64, u64)> { +fn raw_summaries(dag: &PhysicalASAPDAG) -> Vec<(u64, u64)> { dag.nodes .iter() - .filter(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .filter(|node| matches!(node.payload, PhysicalASAPOperatorPayload::SummaryAgg { .. })) .filter_map(|node| { let inputs = dag .edges @@ -89,8 +93,13 @@ fn raw_summaries(dag: &PostAsapDAG) -> Vec<(u64, u64)> { return None; }; let source = dag.nodes.iter().find(|n| n.id == edge.producer)?; - matches!(source.payload, PostAsapOperatorPayload::Fallback { .. }) - .then_some((u64::from(source.id.0), u64::from(node.id.0))) + matches!( + source.payload, + PhysicalASAPOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::TimeRange { .. } + } + ) + .then_some((u64::from(source.id.0), u64::from(node.id.0))) }) .collect() } @@ -114,7 +123,7 @@ fn samples() -> Vec<(Series, i64, f64)> { } fn execute( - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, source: u64, root: u64, rows: &[(Series, i64, f64)], @@ -128,9 +137,9 @@ fn execute( .map(|n| (&n.output_schema, &n.payload)) ); }); - let program = serde_json::from_slice::< - asap_physical_operators::physical_planner::CompiledPhysicalDAG, - >(&serde_json::to_vec(&program).unwrap()) + let program = serde_json::from_slice::( + &serde_json::to_vec(&program).unwrap(), + ) .unwrap(); let schema = precompute::raw_sample_schema(); let batch = Batch::try_new( @@ -240,7 +249,7 @@ fn weight(update: &SummaryUpdate, value: f64) -> f64 { } /// Estimates that identify a state's content for comparison. -fn readouts(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { +fn evaluations(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { if let Some(exact) = state.as_any().downcast_ref::() { let FieldDataType::ExactAggregate(kind, _) = family else { unreachable!() @@ -255,7 +264,7 @@ fn readouts(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { other => panic!("unexpected exact kind {other:?}"), }; return vec![exact - .readout(statistic, None, None::<&KeyByLabelValues>) + .evaluation(statistic, None, None::<&KeyByLabelValues>) .unwrap() .unwrap()]; } @@ -277,7 +286,7 @@ fn readouts(state: &dyn AggregateCore, family: &FieldDataType) -> Vec { /// or the family when it has no native state. fn check( query: &str, - dag: &PostAsapDAG, + dag: &PhysicalASAPDAG, source: u64, root: u64, rows: &[(Series, i64, f64)], @@ -287,7 +296,7 @@ fn check( .iter() .find(|n| u64::from(n.id.0) == root) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let PhysicalASAPOperatorPayload::SummaryAgg { family, input, reduction, @@ -311,8 +320,8 @@ fn check( Reduction::PerEntity => vec![], }; let stored_only = matches!(family, FieldDataType::Sketch(kind, _) - if kind.algorithm() == &asap_types::post_asap::SketchAlgorithm::Cms); - if stored_only || asap_physical_operators::capability::validate_native_family(family).is_err() { + if kind.algorithm() == &asap_types::ir::schema::SketchAlgorithm::Cms); + if stored_only || asap_executor::capability::validate_native_family(family).is_err() { // Families without a native state (e.g. UnivMon), or with native // stored state only (plain CMS), are outside precompute execution; // their compile must fail. @@ -358,7 +367,7 @@ fn check( } } let mut expected = - BTreeMap::>::new(); + BTreeMap::>::new(); for (labels, time, value) in rows { let updater = expected .entry(population(reduction, &keys, labels)) @@ -370,8 +379,8 @@ fn check( for (labels, state) in actual { let reference = expected[&labels].snapshot_accumulator(); assert_eq!( - readouts(state.as_ref(), family), - readouts(reference.as_ref(), family), + evaluations(state.as_ref(), family), + evaluations(reference.as_ref(), family), "{query}: {labels:?}" ); } @@ -413,7 +422,7 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { let mut checked = BTreeMap::new(); for (query, accuracy) in queries { for candidate in candidates(query, accuracy.clone()) { - let dag = compile_post_asap_dag(&candidate).unwrap(); + let dag = compile_maintained_physical_asap_dag(&candidate).unwrap(); for (source, root) in raw_summaries(&dag) { match check(query, &dag, source, root, &rows) { Ok(family) => { @@ -459,21 +468,21 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { /// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with /// another update, keeping its raw input and reduction. -fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PostAsapDAG, u64, u64) { +fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PhysicalASAPDAG, u64, u64) { let candidate = candidates( "sum by (service) (sum_over_time(m[5m]))", AccuracyTarget::Exact, ) .pop() .unwrap(); - let mut dag = compile_post_asap_dag(&candidate).unwrap(); + let mut dag = compile_maintained_physical_asap_dag(&candidate).unwrap(); let (source, root) = raw_summaries(&dag)[0]; let node = dag .nodes .iter_mut() .find(|n| u64::from(n.id.0) == root) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { + let PhysicalASAPOperatorPayload::SummaryAgg { family: old, input: update, .. @@ -503,10 +512,8 @@ fn grouped_raw_summary(family: FieldDataType, input: SummaryUpdate) -> (PostAsap // estimate each item's exact total; invalid weight contracts do not compile. #[test] fn raw_sample_heaps_resolve_items_from_labels() { - use asap_types::post_asap::{ - GroupingStrategy::PerSubpopulationInstance, NonNegativeWeightProof, SketchKind, - SketchParams, WeightDomain, - }; + use asap_types::ir::schema::GroupingStrategy::PerSubpopulationInstance; + use asap_types::ir::schema::{NonNegativeWeightProof, SketchKind, SketchParams, WeightDomain}; let heap = |algorithm, params| { FieldDataType::Sketch(SketchKind::new(algorithm, params), PerSubpopulationInstance) }; @@ -587,9 +594,9 @@ fn raw_sample_heaps_resolve_items_from_labels() { // `without` grouping over raw samples drops the listed labels and `__name__`. #[test] fn raw_sample_without_grouping_drops_labels_and_name() { - use asap_types::pre_asap::query_expr::GroupKeys; + use asap_types::ir::operator::GroupKeys; let family = - FieldDataType::ExactAggregate(ExactKind::Sum, asap_types::post_asap::ExactParams::Sum); + FieldDataType::ExactAggregate(ExactKind::Sum, asap_types::ir::schema::ExactParams::Sum); let (mut dag, source, root) = grouped_raw_summary(family, SummaryUpdate::column(ColumnRef::SampleValue)); let service = dag @@ -607,7 +614,7 @@ fn raw_sample_without_grouping_drops_labels_and_name() { .iter_mut() .find(|n| u64::from(n.id.0) == root) .unwrap(); - let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = &mut node.payload else { + let PhysicalASAPOperatorPayload::SummaryAgg { reduction, .. } = &mut node.payload else { unreachable!() }; *reduction = Reduction::Reduce(GroupKeys::without(vec![service])); diff --git a/crates/integration-tests/tests/promql_numeric_regressions.rs b/crates/integration-tests/tests/promql_numeric_regressions.rs index 692fe9c7a..1d1aac8ab 100644 --- a/crates/integration-tests/tests/promql_numeric_regressions.rs +++ b/crates/integration-tests/tests/promql_numeric_regressions.rs @@ -1,36 +1,42 @@ -//! Numeric regression fixtures: actual PromQL lowering plus numeric update/readout checks. +//! Numeric regression fixtures: actual PromQL lowering plus numeric update/evaluation checks. //! The count/sum interpreter below verifies planner update semantics, not a deployed backend. -use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; use asap_integration_tests::fixtures::lower_promql; -use asap_types::post_asap::{ - compile_post_asap_dag, ExactKind, FieldDataType, SummaryExpr, SummaryInputExpr, SummaryNode, - SummaryUpdate, -}; -use asap_types::pre_asap::{ColumnRef, Reduction}; +use asap_integration_tests::post_asap::post_asap_dag; +use asap_logical_optimizer::pass1::replacement::is_logical_rewrite; +use asap_logical_optimizer::{ASAPStrategies, Replacement, ReplacementStrategy, TargetSubDAG}; +use asap_types::ir::operator::Reduction; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{ExactKind, FieldDataType, SummaryInputExpr, SummaryUpdate}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; use asap_types::types::AccuracyTarget; use std::rc::Rc; -fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { - let pre = Rc::new(lower_promql(query, accuracy).unwrap()); - SketchAlgorithmStrategy::default_cost_model() +fn plan(query: &str, accuracy: AccuracyTarget) -> Rc { + let pre = lower_promql(query, accuracy).unwrap(); + ASAPStrategies::default() .replacements(&TargetSubDAG::new(&pre)) .into_iter() .find_map(|r| match r.replacement { - Replacement::Summary(n) => Some(n), + // A bound decision: a summary DAG or a kept (exact) sub-DAG. + Replacement::SubDAG(n) if !is_logical_rewrite(&n) => Some(n), _ => None, }) - .unwrap_or_else(|| asap_aware_mapping::replacement::keep_pre_asap(&pre).unwrap()) + .unwrap_or_else(|| asap_logical_optimizer::pass1::replacement::retain_exact(&pre).unwrap()) } -fn aggregate(node: &SummaryNode) -> (&FieldDataType, &SummaryUpdate, &Reduction) { - match &node.expr { - SummaryExpr::SummaryAgg { +fn aggregate(node: &OperatorNode) -> (&FieldDataType, &SummaryUpdate, &Reduction) { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, .. - } => (family, input, reduction), - SummaryExpr::SummaryEstimate { summary_input, .. } => aggregate(summary_input), - SummaryExpr::ValueOperation { child, .. } => aggregate(child), + }) => (family, input, reduction), + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => aggregate(summary_input), + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => aggregate(child), + // A value operation (Project/Filter/Sort/Limit/...) over the state. + Operator::NonASAP(op) if op.children().len() == 1 && node.contains_asap() => { + aggregate(op.children()[0]) + } other => panic!("not a maintained accumulator: {other:?}"), } } @@ -63,7 +69,7 @@ fn count_up_counts_targets_even_when_values_repeat_or_change_sign() { 3. ); } - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } /// Ten samples give count ten, whereas sum retains the signed sample values. @@ -81,7 +87,7 @@ fn window_counts_and_sums_distinguish_one_zero_three_and_negative_values() { let got: f64 = (0..10).map(|_| contribution(family, update, value)).sum(); assert_eq!(got, if is_count { 10. } else { value * 10. }); } - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -97,7 +103,7 @@ fn sum_rate_and_increase_have_real_exact_accumulator_nodes() { let (family, _, _) = aggregate(&node); assert!(matches!(family, FieldDataType::ExactAggregate(k, _) if *k == kind)); assert!(node.guarantee.as_ref().unwrap().is_exact()); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } @@ -109,11 +115,12 @@ fn checked_ratio_must_not_certify_cross_zero_interpolation() { AccuracyTarget::Epsilon(0.01), ); assert!( - matches!(node.expr, SummaryExpr::BinaryOp { .. }), + matches!(node.operator, Operator::NonASAP(NonASAPOp::BinaryOp { .. })) + && node.contains_asap(), "direct quantile ratio should remain an available candidate" ); assert!(node.guarantee.is_none()); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); // Keep the actual signed-sketch counterexample: division guards alone pass // even though the quantile interpolation does not preserve relative error. let alpha = (0.01 - 8.0 * f64::EPSILON) / 2.01; @@ -148,37 +155,37 @@ fn quantile_over_temporal_average_keeps_a_legal_candidate() { ] { let node = plan(query, AccuracyTarget::Epsilon(0.01)); assert!( - !matches!(node.expr, SummaryExpr::KeepPreAsap(_)), + node.contains_asap(), "outer sketch candidate must survive: {query}" ); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { - panic!("outer sketch readout") + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator else { + panic!("outer sketch evaluation") }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { panic!("outer sketch state") }; assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + !child.contains_asap(), "guarded expression must retain native maintenance input" ); - compile_post_asap_dag(&node).unwrap(); + post_asap_dag(&node); } } struct OneKeyTopKEvidence; -impl asap_aware_mapping::accuracy::AccuracyEvidenceProvider for OneKeyTopKEvidence { +impl asap_logical_optimizer::accuracy::AccuracyEvidenceProvider for OneKeyTopKEvidence { fn propagation_stats( &self, - op: &asap_types::post_asap::CompositionOperator, + op: &asap_types::ir::properties::CompositionOperator, _family: &FieldDataType, - _query: Option<&asap_types::post_asap::SketchStatistic>, - ) -> asap_aware_mapping::accuracy::PropagationStats { + _query: Option<&asap_types::ir::schema::SketchStatistic>, + ) -> asap_logical_optimizer::accuracy::PropagationStats { // Single-key fixture: no excluded keys; bounds cover every value below. if matches!( op, - asap_types::post_asap::CompositionOperator::TopKSelection + asap_types::ir::properties::CompositionOperator::TopKSelection ) { - asap_aware_mapping::accuracy::PropagationStats { + asap_logical_optimizer::accuracy::PropagationStats { topk_selected_lower_bound: Some(-1000.), topk_excluded_upper_bound: Some(-1001.), topk_interval_failure_probability: Some(0.001), @@ -192,11 +199,9 @@ impl asap_aware_mapping::accuracy::AccuracyEvidenceProvider for OneKeyTopKEviden #[test] fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { - use asap_aware_mapping::accuracy::{DefaultAccuracyModel, EqualSplitAllocator}; - use asap_aware_mapping::cost_model::DefaultCostModel; - use asap_types::post_asap::{NonNegativeWeightProof, SketchAlgorithm, WeightDomain}; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + use asap_logical_optimizer::accuracy::{DefaultAccuracyModel, EqualSplitAllocator}; + use asap_types::ir::schema::{NonNegativeWeightProof, SketchAlgorithm, WeightDomain}; + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &OneKeyTopKEvidence, @@ -207,7 +212,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { } else { "topk(1, sum_over_time(up[5m]))" }; - let pre = Rc::new(lower_promql(query, AccuracyTarget::Epsilon(0.01)).unwrap()); + let pre = lower_promql(query, AccuracyTarget::Epsilon(0.01)).unwrap(); let candidates = strategy.replacements(&TargetSubDAG::new(&pre)); let wanted = if is_count { SketchAlgorithm::CmsWithHeap @@ -217,7 +222,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { let node = candidates .iter() .find_map(|c| { - let Replacement::Summary(node) = &c.replacement else { + let Replacement::SubDAG(node) = &c.replacement else { return None; }; let (family, _, _) = aggregate(node); @@ -240,7 +245,7 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { SummaryInputExpr::Column(ColumnRef::SampleValue) ); for c in &candidates { - if let Replacement::Summary(n) = &c.replacement { + if let Replacement::SubDAG(n) = &c.replacement { assert!( !matches!(aggregate(n).0, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) ); @@ -271,6 +276,6 @@ fn sketch_counts_use_unit_weights_and_signed_sums_keep_value_weights() { }; assert_eq!(got, if is_count { 10. } else { 10. * value }); } - compile_post_asap_dag(node).unwrap(); + post_asap_dag(node); } } diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 97a08a0c4..86e859d99 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -1,81 +1,96 @@ //! End-to-end query-string → post-ASAP IR pin (issue #98). //! -//! Drives the full pipeline — PromQL text → pre-ASAP `QueryExpr` -//! (`lower_promql`) → post-ASAP `SummaryExpr` DAG (via -//! `SketchAlgorithmStrategy::replacements`, see [`realize`] below) — and pins +//! Drives the full pipeline — PromQL text → non-ASAP `OperatorNode` +//! (`lower_promql`) → post-ASAP `OperatorNode` DAG (via +//! `ASAPStrategies::replacements`, see [`realize`] below) — and pins //! the summary-bound shape node by node, including the family `(Kind, //! Params)` committed on each edge's schema. use std::rc::Rc; -use asap_aware_mapping::accuracy::{ +use asap_integration_tests::fixtures::lower_promql; +use asap_integration_tests::post_asap::{ + maintained, maintained_post_asap_dag, post_asap_dag, timed, +}; +use asap_logical_optimizer::accuracy::{ AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, QuantileInputDomain, }; -use asap_aware_mapping::cost_model::DefaultCostModel; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; -use asap_aware_mapping::{ - search_workload, search_workload_with_targets, AccuracyModel, Replacement, ReplacementStrategy, - ReplacementSubDAG, SketchAlgorithmStrategy, TargetSubDAG, +use asap_logical_optimizer::pass1::replacement::{ + is_logical_rewrite, retain_exact, RealizationError, }; -use asap_integration_tests::fixtures::lower_promql; -use asap_types::post_asap::{ - compile_post_asap_dag, CompositionOperator, EntityIdentity, ExactKind, ExactParams, - FieldDataType, GroupingStrategy, Schema, SketchAlgorithm, SketchKind, SketchParams, - SketchStatistic, SummaryExpr, SummaryInputExpr, SummaryNode, SummaryUpdate, ValueOperation, +use asap_logical_optimizer::{ + search_workload, search_workload_with_targets, ASAPStrategies, AccuracyModel, Replacement, + ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; -use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; -use asap_types::pre_asap::schema::DataType; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::cost::cost_model::DefaultCostModel; +use asap_types::ir::export::{NonASAPOpKind, PhysicalASAPOperatorPayload}; +use asap_types::ir::operator::operator_properties::Reduction; +use asap_types::ir::properties::CompositionOperator; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::DataType; +use asap_types::ir::schema::{ + EntityIdentity, ExactKind, ExactParams, FieldDataType, GroupingStrategy, Schema, + SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, SummaryInputExpr, SummaryUpdate, +}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; use asap_types::types::AccuracyTarget; -/// This crate has no "bind me one DAG" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// This crate has no "bind me one tree" public API any more — +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() +fn realize(root: &Rc) -> Result, RealizationError> { + let target = TargetSubDAG::new(root); + match ASAPStrategies::default() .replacements(&target) .into_iter() .next() { + // A bound decision (summary DAG or kept sub-DAG); a logical rewrite + // is not a binding, so it falls back to keeping the target. Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. - }) => Ok(node), - _ => keep_pre_asap(&root), + }) if !is_logical_rewrite(&node) => Ok(node), + _ => retain_exact(root), } + .inspect(|node| { + node.validate_structure() + .expect("planned dag satisfies the unified IR contract") + }) } #[test] -fn distinct_over_time_offers_hll_cardinality_readout() { +fn distinct_over_time_offers_hll_cardinality_evaluation() { // The real frontend must reach an existing HLL candidate without a // function-specific post-ASAP node or a sample-count rewrite. - let root = Rc::new( - lower_promql( - "distinct_over_time(cpu_usage{job=\"worker\"}[5m])", - AccuracyTarget::Epsilon(0.02), - ) - .unwrap(), - ); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql( + "distinct_over_time(cpu_usage{job=\"worker\"}[5m])", + AccuracyTarget::Epsilon(0.02), + ) + .unwrap(); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); + for candidate in &candidates { + if let Replacement::SubDAG(node) = &candidate.replacement { + node.validate_structure().unwrap(); + } + } assert!(candidates.iter().any(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { return false }; - let SummaryExpr::SummaryEstimate { summary_input, query, .. } = &node.expr else { return false }; + let Replacement::SubDAG(node) = &candidate.replacement else { return false }; + let Some(ASAPOp::SummaryEstimate { summary_input, query, .. }) = node.asap() else { return false }; matches!(query, SketchStatistic::Cardinality) - && matches!(&summary_input.expr, SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + && matches!(summary_input.asap(), Some(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::Hll) }), "no HLL cardinality candidate: {candidates:?}"); } -fn lower_search_and_materialize(query: &str) -> Rc { - let pre = Rc::new(lower_promql(query, AccuracyTarget::Exact).expect("lowering failed")); +fn lower_search_and_materialize(query: &str) -> Rc { + let pre = lower_promql(query, AccuracyTarget::Exact).expect("lowering failed"); let space = search_workload(vec![("query", pre)]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") @@ -88,36 +103,38 @@ fn value_ranked_topk_preserves_summary_children_in_post_asap_dag() { "topk(3, rate(cpu_seconds_total[5m]))", "topk by (job) (2, max_over_time(memory_bytes[6h]))", ] { - let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { n, offset, .. }, + let root = timed(&lower_search_and_materialize(query)); + let Some(NonASAPOp::Limit { + n, + offset, child: sort, .. - } = &root.expr + }) = root.non_asap() else { - panic!("expected query-time Limit for {query}, got {:?}", root.expr); + panic!( + "expected query-time Limit for {query}, got {:?}", + root.operator + ); }; - assert!(*n > 0 && *offset == 0); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Sort { .. }, - child, - .. - } = &sort.expr - else { + assert_eq!( + root.timing, + Some(asap_types::ir::properties::ExecutionTiming::QueryTime) + ); + assert!(n.is_some_and(|n| n > 0) && *offset == 0); + let Some(NonASAPOp::Sort { child, .. }) = sort.non_asap() else { panic!("expected query-time Sort under Limit for {query}"); }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - child: state, - .. - } = &child.expr - else { + assert_eq!( + sort.timing, + Some(asap_types::ir::properties::ExecutionTiming::QueryTime) + ); + let Some(ASAPOp::FinalizeExactAccumulator { child: state }) = child.asap() else { panic!( "Sort must consume finalized values for {query}: {:?}", - child.expr + child.operator ); }; - assert!(matches!(state.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(state.asap(), Some(ASAPOp::SummaryAgg { .. }))); assert!(child .schema .fields @@ -135,9 +152,9 @@ fn exact_counter_weighted_topk_fails_closed_without_membership_certificate() { ] { let root = lower_search_and_materialize(query); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + !root.contains_asap(), "exact target must not accept an uncertified membership sidecar for {query}: {:?}", - root.expr + root.operator ); } } @@ -146,23 +163,14 @@ fn exact_counter_weighted_topk_fails_closed_without_membership_certificate() { fn instant_topk_and_unsupported_child_remain_local_residuals() { for query in ["topk(3, memory_bytes)", "topk(3, deriv(memory_bytes[5m]))"] { let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { .. }, - child: sort, - .. - } = &root.expr - else { + let Some(NonASAPOp::Limit { child: sort, .. }) = root.non_asap() else { panic!("expected Limit for {query}"); }; - let SummaryExpr::ValueOperation { - child, operation, .. - } = &sort.expr - else { + let Some(NonASAPOp::Sort { child, .. }) = sort.non_asap() else { panic!("expected Sort for {query}"); }; - assert!(matches!(operation, ValueOperation::Sort { .. })); assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + !child.contains_asap(), "only the unsupported child should remain exact for {query}" ); } @@ -177,7 +185,7 @@ fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { .dtype } -fn lower_and_realize(query: &str) -> Rc { +fn lower_and_realize(query: &str) -> Rc { let pre = lower_promql(query, AccuracyTarget::Exact).expect("lowering failed"); realize(&pre).expect("binding failed") } @@ -186,19 +194,17 @@ fn lower_and_realize(query: &str) -> Rc { fn promql_binary_arithmetic_retains_two_summary_leaves() { for op in ["+", "-", "*", "/", "%", "^", "atan2"] { let root = lower_and_realize(&format!("rate(a[1m]) {op} rate(b[1m])")); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp for {op}, got {:?}", root.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp for {op}, got {:?}", root.operator); }; for operand in [lhs, rhs] { - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } = &operand.expr - else { - panic!("expected an explicit exact readout, got {:?}", operand.expr); + let Some(ASAPOp::FinalizeExactAccumulator { child }) = operand.asap() else { + panic!( + "expected an explicit exact evaluation, got {:?}", + operand.operator + ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(child.asap(), Some(ASAPOp::SummaryAgg { .. }))); } } } @@ -207,47 +213,36 @@ fn promql_binary_arithmetic_retains_two_summary_leaves() { fn value_ranked_topk_over_binary_ratio_finalizes_both_summary_operands() { let query = "topk(1, sum by(job)(increase(a[6h])) / sum by(job)(increase(b[6h])))"; let root = lower_search_and_materialize(query); - let SummaryExpr::ValueOperation { - operation: ValueOperation::Limit { - n: 1, offset: 0, .. - }, + let Some(NonASAPOp::Limit { + n: Some(1), + offset: 0, child: sort, .. - } = &root.expr + }) = root.non_asap() else { - panic!("expected Limit root, got {:?}", root.expr); + panic!("expected Limit root, got {:?}", root.operator); }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::Sort { .. }, - child: binary, - .. - } = &sort.expr - else { - panic!("expected Sort below Limit, got {:?}", sort.expr); + let Some(NonASAPOp::Sort { child: binary, .. }) = sort.non_asap() else { + panic!("expected Sort below Limit, got {:?}", sort.operator); }; - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &binary.expr else { - panic!("expected BinaryOp below Sort, got {:?}", binary.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = binary.non_asap() else { + panic!("expected BinaryOp below Sort, got {:?}", binary.operator); }; for operand in [lhs, rhs] { - let SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - child, - .. - } = &operand.expr - else { + let Some(ASAPOp::FinalizeExactAccumulator { child }) = operand.asap() else { panic!( "expected exact accumulator finalization, got {:?}", - operand.expr + operand.operator ); }; - assert!(matches!(child.expr, SummaryExpr::SummaryAgg { .. })); + assert!(matches!(child.asap(), Some(ASAPOp::SummaryAgg { .. }))); } } struct SeparatedTopK; impl AccuracyEvidenceProvider for SeparatedTopK { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &OperatorNode) -> Option { Some(1000) } @@ -271,15 +266,12 @@ impl AccuracyEvidenceProvider for SeparatedTopK { // Rate-weighted summaries must consume finalized rates, never raw counter deltas. #[test] fn grouped_rate_topk_consumes_finalized_rate_values() { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopK, @@ -288,24 +280,31 @@ fn grouped_rate_topk_consumes_finalized_rate_values() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), + Replacement::SubDAG(node) if candidate.rationale.contains("CmsWithHeap") => Some(node), _ => None, }) .expect("rate-weighted CMS plan"); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); assert!(!dag.nodes.iter().any(|node| matches!( node.payload, - asap_types::post_asap::PostAsapOperatorPayload::RelationalJoin { .. } + PhysicalASAPOperatorPayload::Relational { + operator: NonASAPOpKind::Join { .. } + } ))); - let node = dag.nodes.iter().find(|node| matches!(&node.payload, - asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } - if kind.algorithm() == &SketchAlgorithm::CmsWithHeap)).unwrap(); + let node = dag + .nodes + .iter() + .find(|node| { + matches!(&node.payload, + PhysicalASAPOperatorPayload::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + if kind.algorithm() == &SketchAlgorithm::CmsWithHeap) + }) + .unwrap(); assert_eq!( node.output_state.timing, - asap_types::post_asap::ExecutionTiming::QueryTime + asap_types::ir::properties::ExecutionTiming::QueryTime ); - let asap_types::post_asap::PostAsapOperatorPayload::SummaryAgg { input, .. } = &node.payload - else { + let PhysicalASAPOperatorPayload::SummaryAgg { input, .. } = &node.payload else { unreachable!() }; assert_eq!( @@ -332,15 +331,12 @@ fn weighted_topk_keeps_candidates_with_missing_population_evidence() { SeparatedTopK.propagation_stats(op, family, query) } } - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &NoPopulationBound, @@ -355,27 +351,24 @@ fn weighted_topk_keeps_candidates_with_missing_population_evidence() { // Unknown requirements must survive physical export for deployment to inspect. #[test] fn weighted_topk_exports_symbolic_evidence_requirements() { - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); let candidate = candidates .iter() .find(|candidate| candidate.rationale.contains("CmsWithHeap")) .unwrap(); assert!(candidate.has_missing_accuracy_evidence()); - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { panic!("summary candidate") }; - let dag = compile_post_asap_dag(node).unwrap(); + let dag = post_asap_dag(node); let exported = serde_json::to_string(&dag).unwrap(); assert!(exported.contains("topk_max_distinct_items")); assert!(exported.contains("topk_membership_margin")); @@ -387,19 +380,16 @@ fn weighted_topk_exports_symbolic_evidence_requirements() { fn weighted_topk_rejects_invalid_population_evidence() { struct InvalidPopulation; impl AccuracyEvidenceProvider for InvalidPopulation { - fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + fn topk_max_distinct_items(&self, _: &OperatorNode) -> Option { Some(0) } } - let root = Rc::new( - lower_promql( - "topk by(job)(2, sum by(service, job)(rate(m[1m])))", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let root = lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &InvalidPopulation, @@ -415,18 +405,15 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { "topk by(job)(2, sum by(service, job)(rate(m[1m])))", "topk(2, sum by(job)(increase(m[6h])))", ] { - let root = Rc::new( - lower_promql( - query, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let root = lower_promql( + query, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopK, @@ -435,60 +422,50 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { .replacements(&TargetSubDAG::new(&root)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => { + Replacement::SubDAG(node) if candidate.rationale.contains("CmsWithHeap") => { Some(node) } _ => None, }) .expect("weighted summary"); - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Limit { + n: Some(2), + offset: 0, + partition_by, child: sorted, - operation: - ValueOperation::Limit { - n: 2, - offset: 0, - partition_by, - }, - .. - } = &plan.expr + }) = plan.non_asap() else { panic!("grouped limit") }; - let SummaryExpr::ValueOperation { + let Some(NonASAPOp::Sort { + partition_by: sort_groups, child: projected, - operation: - ValueOperation::Sort { - partition_by: sort_groups, - .. - }, .. - } = &sorted.expr + }) = sorted.non_asap() else { panic!("grouped sort") }; assert_eq!(partition_by, sort_groups); assert_eq!(partition_by.len(), usize::from(query.contains("topk by"))); - let SummaryExpr::ValueOperation { - child: readout, - operation: ValueOperation::Project { .. }, - .. - } = &projected.expr + let Some(NonASAPOp::Project { + child: evaluation, .. + }) = projected.non_asap() else { panic!("logical output projection") }; - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, query: SketchStatistic::TopK { k }, - } = &readout.expr + }) = evaluation.asap() else { - panic!("heap readout") + panic!("heap evaluation") }; assert!(*k > 2, "candidate capacity is independent of output count"); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child: rates, input, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("weighted summary") }; @@ -497,16 +474,13 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { SummaryInputExpr::Column(ColumnRef::SampleValue) ); assert!(matches!( - rates.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } + rates.asap(), + Some(ASAPOp::FinalizeExactAccumulator { .. }) )); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = post_asap_dag(&plan); for phase in [ - asap_types::post_asap::ExecutionTiming::IngestionTime, - asap_types::post_asap::ExecutionTiming::QueryTime, + asap_types::ir::properties::ExecutionTiming::IngestionTime, + asap_types::ir::properties::ExecutionTiming::QueryTime, ] { let phases = dag.nodes.iter().map(|node| (node.id, phase)).collect(); let placed = dag.with_execution_phases(&phases).unwrap(); @@ -518,64 +492,48 @@ fn rate_and_increase_topk_use_summary_scores_and_grouped_limits() { let guarantee = plan.guarantee.as_ref().unwrap(); assert!(guarantee.failure_probability.evaluate().unwrap() <= 0.01); assert!(guarantee.provenance.iter().any(|source| matches!(source, - asap_types::post_asap::GuaranteeSource::ChildGuarantee { guarantee, .. } - if guarantee.metric == asap_types::post_asap::ErrorMetric::Frequency))); + asap_types::ir::properties::GuaranteeSource::ChildGuarantee { guarantee, .. } + if guarantee.metric == asap_types::ir::properties::ErrorMetric::Frequency))); } } #[test] fn promql_binary_arithmetic_preserves_both_scalar_operand_orders() { - fn is_exact_readout_or_scalar(node: &SummaryNode) -> bool { - matches!(node.expr, SummaryExpr::KeepPreAsap(_)) - || matches!( - node.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) - } - for query in ["rate(a[1m]) / 2", "2 / rate(a[1m])"] { + for (query, scalar_left) in [("rate(a[1m]) / 2", false), ("2 / rate(a[1m])", true)] { let root = lower_and_realize(query); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp for {query}, got {:?}", root.expr); + let Some(NonASAPOp::Project { cols, .. }) = root.non_asap() else { + panic!("expected Project") }; - assert!(is_exact_readout_or_scalar(lhs)); - assert!(is_exact_readout_or_scalar(rhs)); - assert!( - matches!( - lhs.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) || matches!( - rhs.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - ) - ); + let ScalarExpr::Arithmetic { left, right, .. } = &cols[1].expr else { + panic!() + }; + let (scalar, sample) = if scalar_left { + (left, right) + } else { + (right, left) + }; + assert_eq!(**scalar, ScalarExpr::literal_f64(2.0)); + assert_eq!(**sample, ScalarExpr::Column(1)); + assert!(root.schema.has_promql_series_identity()); } } #[test] fn promql_binary_arithmetic_falls_back_as_a_whole_for_unsupported_arm() { let root = lower_and_realize("rate(a[1m]) + stddev_over_time(b[1m])"); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] fn promql_binary_arithmetic_preserves_nested_structure_and_rejects_modifiers() { let nested = lower_and_realize("(rate(a[1m]) + rate(b[1m])) / 2"); - let SummaryExpr::BinaryOp { lhs, .. } = &nested.expr else { - panic!("expected outer BinaryOp, got {:?}", nested.expr); + let Some(NonASAPOp::Project { child: lhs, .. }) = nested.non_asap() else { + panic!("expected outer BinaryOp, got {:?}", nested.operator); }; - assert!(matches!(lhs.expr, SummaryExpr::BinaryOp { .. })); + assert!(matches!(lhs.non_asap(), Some(NonASAPOp::BinaryOp { .. }))); let modified = lower_and_realize("rate(a[1m]) + on(job) rate(b[1m])"); - assert!(matches!(modified.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!modified.contains_asap()); } #[test] @@ -586,8 +544,8 @@ fn promql_binary_arithmetic_never_relabels_approximate_children_as_exact() { ) .expect("lowering failed"); let root = realize(&pre).expect("binding failed"); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &root.expr else { - panic!("expected BinaryOp, got {:?}", root.expr); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp, got {:?}", root.operator); }; assert!(lhs.guarantee.as_ref().is_some_and(|g| !g.is_exact())); assert!(rhs.guarantee.as_ref().is_some_and(|g| !g.is_exact())); @@ -606,13 +564,11 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { epsilon: 0.01, delta: 0.01, }; - let query = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - target.clone(), - ) - .expect("lowering failed"), - ); + let query = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + target.clone(), + ) + .expect("lowering failed"); let evidence = FixtureQuantileDomain { lower: 1.0, @@ -620,19 +576,16 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { }; let space = search_workload_with_targets( vec![("ratio", query, Some(target.clone()))], - &asap_aware_mapping::replacement::default_strategies_with_evidence( - &DefaultCostModel, - &evidence, - ), + &asap_logical_optimizer::pass1::replacement::default_strategies_with_evidence(&evidence), &DefaultAccuracyModel, ); let root = &space.roots[0].1; - let selected = space.global_selection(&DefaultCostModel); + let selected = global_selection(&space, &DefaultCostModel); let chosen = selected .for_target(root) .and_then(|selection| selection.chosen.as_ref()) .expect("the certified DDSketch ratio should be selectable"); - let Replacement::Summary(node) = &chosen.replacement else { + let Replacement::SubDAG(node) = &chosen.replacement else { panic!("expected a summary candidate") }; let guarantee = node.guarantee.as_ref().expect("ratio guarantee"); @@ -642,23 +595,22 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { "ratio guarantee should satisfy the requested target: {guarantee:?}" ); - let shared = - asap_types::post_asap::share_common_summary_sub_dags(vec![("ratio", node.clone())]); - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &shared[0].1.expr else { + let shared = asap_types::ir::cse::share_common_sub_dags(vec![("ratio", node.clone())]); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = shared[0].1.non_asap() else { panic!("expected binary ratio") }; - let producer = |readout: &Rc| match &readout.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => Rc::clone(summary_input), - other => panic!("expected DDSketch readout, got {other:?}"), + let producer = |evaluation: &Rc| match &evaluation.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => Rc::clone(summary_input), + other => panic!("expected DDSketch evaluation, got {other:?}"), }; assert!( Rc::ptr_eq(&producer(lhs), &producer(rhs)), - "the two quantile readouts should share one DDSketch producer" + "the two quantile evaluations should share one DDSketch producer" ); } #[test] -fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() { +fn planner_only_e2e_temporal_topk_preserves_query_update_and_evaluation_contract() { // Self-contained Planner E2E: each case starts from PromQL text and ends // at the post-ASAP summary DAG. No controller/backend types, // fixtures, configuration, or runtime are involved. @@ -677,18 +629,15 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() ), ]; for (source, expected_update, expected_family, excluded_labels) in cases { - let pre = Rc::new( - lower_promql( - source, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .expect("lower temporal Top-K"), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let pre = lower_promql( + source, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .expect("lower temporal Top-K"); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopK, @@ -697,29 +646,29 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() .replacements(&TargetSubDAG::new(&pre)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains(expected_family) => { + Replacement::SubDAG(node) if candidate.rationale.contains(expected_family) => { Some(node) } _ => None, }) .expect("heap-backed temporal Top-K candidate"); - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, query: SketchStatistic::TopK { k, .. }, - } = &candidate.expr + }) = candidate.asap() else { - panic!("expected Top-K estimate, got {:?}", candidate.expr) + panic!("expected Top-K estimate, got {:?}", candidate.operator) }; assert_eq!( *k, 5, "the requested Top-K cardinality must survive binding" ); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { input: state_input, family, child, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("expected structured Top-K state input") }; @@ -742,42 +691,41 @@ fn planner_only_e2e_temporal_topk_preserves_query_update_and_readout_contract() )) ); assert_eq!(state_input.weight, expected_update); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } } /// Execute the ungrouped temporal TopK subset with exact state. This tests /// the emitted update contract, not sketch approximation or backend execution. -fn execute_topk_reference(plan: &SummaryNode) -> Vec<(String, f64)> { +fn execute_topk_reference(plan: &OperatorNode) -> Vec<(String, f64)> { use std::collections::BTreeMap; - let SummaryExpr::SummaryEstimate { + let Some(ASAPOp::SummaryEstimate { summary_input, query: SketchStatistic::TopK { k }, - } = &plan.expr + }) = plan.asap() else { - panic!("expected TopK readout") + panic!("expected TopK evaluation") }; - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { input, child, reduction, .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("expected summary updates") }; assert_eq!(reduction, &Reduction::by(vec![])); - let SummaryExpr::KeepPreAsap(raw) = &child.expr else { - panic!("expected fused raw input") - }; - let QueryExpr::TimeRange { range, child } = raw.as_ref() else { + // The fused raw input is the kept non-ASAP sub-DAG itself. + assert!(!child.contains_asap(), "expected fused raw input"); + let Some(NonASAPOp::TimeRange { range, child, .. }) = child.non_asap() else { panic!("expected temporal input") }; - let QueryExpr::Scan { - source: asap_types::pre_asap::Source::TimeSeries { metric }, + let Some(NonASAPOp::Scan { + source: asap_types::ir::operator::Source::TimeSeries { metric }, predicates, .. - } = child.as_ref() + }) = child.non_asap() else { panic!("expected metric scan") }; @@ -849,18 +797,15 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { vec![("worker", 100.0), ("cron", 30.0)], ), ] { - let pre = Rc::new( - lower_promql( - query, - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let pre = lower_promql( + query, + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopK, @@ -868,10 +813,10 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { // This reference executor consumes keyed heap updates. The inventory // also contains maintained exact values followed by sort/limit; those // have a different execution contract and must not enter this fixture. - let candidates: Vec<_> = strategy.replacements(&TargetSubDAG::new(&pre)).into_iter().filter(|candidate| matches!(&candidate.replacement, Replacement::Summary(plan) if matches!(plan.expr, SummaryExpr::SummaryEstimate { query: SketchStatistic::TopK { .. }, .. }))).collect(); + let candidates: Vec<_> = strategy.replacements(&TargetSubDAG::new(&pre)).into_iter().filter(|candidate| matches!(&candidate.replacement, Replacement::SubDAG(plan) if matches!(plan.asap(), Some(ASAPOp::SummaryEstimate { query: SketchStatistic::TopK { .. }, .. })))).collect(); assert!(!candidates.is_empty(), "no heap candidate for {query}"); for candidate in candidates { - let Replacement::Summary(plan) = candidate.replacement else { + let Replacement::SubDAG(plan) = candidate.replacement else { panic!("expected summary plan for {query}") }; let expected: Vec<_> = expected @@ -889,11 +834,11 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { /// SummaryEstimate { query: Quantile{0.99} } → {quantile_0_99: Float64} /// └─ SummaryAgg { Kll{k:269}, input: SampleValue } → {value: Sketch(Kll, {k:269})} /// └─ SummaryAgg { Rate, input: SampleValue } → {ts, value: ExactAggregate(Rate), …} -/// └─ KeepPreAsap(TimeRange{5m} → Scan) → {ts, value} +/// └─ TimeRange{5m} → Scan → {ts, value} /// ``` /// -/// The nested DAG exercises both realizations: the approximate quantile -/// binds a KLL sketch + readout; the per-series `rate` binds the exact +/// The nested tree exercises both realizations: the approximate quantile +/// binds a KLL sketch + evaluation; the per-series `rate` binds the exact /// counter-reset-aware accumulator (no estimate — its state is the value). #[test] fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { @@ -904,13 +849,13 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { .expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); - // Root: the sketch readout, back to a plain row shape. - let SummaryExpr::SummaryEstimate { + // Root: the sketch evaluation, back to a plain row shape. + let Some(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = root.asap() else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, SketchStatistic::Quantile { q } if *q == 0.99)); assert_eq!( @@ -924,15 +869,15 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { // reduction, one output row — not to be confused with the inner rate's // per-entity grouping below, even though both once collapsed to the // same empty `by: []` (issue #163). - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = summary_input.asap() else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -955,26 +900,24 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { ) ); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: asap_types::post_asap::ExecutionTiming::IngestionTime, - } = &child.expr - else { - panic!("rate needs a maintenance readout"); + let Some(ASAPOp::FinalizeExactAccumulator { child }) = child.asap() else { + panic!("rate needs a maintenance evaluation"); }; // The rate: exact counter-reset-aware accumulator, per-series (labels // and time axis preserved), no estimate wrapper. `rate(...)` has no // grouping concept at all — every entity stays its own summary. - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { child: leaf, family, reduction, .. - } = &child.expr + }) = child.asap() else { - panic!("expected inner SummaryAgg for rate, got {:?}", child.expr); + panic!( + "expected inner SummaryAgg for rate, got {:?}", + child.operator + ); }; assert_eq!( family, @@ -992,14 +935,20 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { ); // The leaf: unrewritten pass-through — TimeRange marker over the Scan. - let SummaryExpr::KeepPreAsap(kept_leaf) = &leaf.expr else { - panic!("expected KeepPreAsap leaf, got {:?}", leaf.expr); - }; - let QueryExpr::TimeRange { range, child: scan } = kept_leaf.as_ref() else { - panic!("expected TimeRange leaf, got {kept_leaf:?}"); + // The kept leaf is the non-ASAP sub-DAG itself. + assert!( + !leaf.contains_asap(), + "expected kept leaf, got {:?}", + leaf.operator + ); + let Some(NonASAPOp::TimeRange { + range, child: scan, .. + }) = leaf.non_asap() + else { + panic!("expected TimeRange leaf, got {:?}", leaf.operator); }; assert_eq!(range.as_secs(), 300); - assert!(matches!(scan.as_ref(), QueryExpr::Scan { .. })); + assert!(matches!(scan.non_asap(), Some(NonASAPOp::Scan { .. }))); assert!( leaf.schema .fields @@ -1017,11 +966,11 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { let pre_asap = lower_promql("sum by (job) (http_requests_total)", AccuracyTarget::Exact) .expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { family, reduction, .. - } = &root.expr + }) = root.asap() else { - panic!("expected SummaryAgg, got {:?}", root.expr); + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, @@ -1042,21 +991,19 @@ fn promql_exact_workload_binds_accumulators_not_sketches() { lower_promql("avg(http_requests_total)", AccuracyTarget::Exact).expect("lowering failed"); let root = realize(&pre_asap).expect("binding failed"); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + !root.contains_asap(), "avg has no mergeable accumulator — stays logical" ); } #[test] fn promql_sum_of_count_over_time_is_composed_by_default_search() { - let original = Rc::new( - lower_promql( - "sum by (service) (count_over_time(metrics[5m]))", - AccuracyTarget::Exact, - ) - .expect("lowering failed"), - ); - let original_schema = original.output_schema().unwrap(); + let original = lower_promql( + "sum by (service) (count_over_time(metrics[5m]))", + AccuracyTarget::Exact, + ) + .expect("lowering failed"); + let original_schema = original.schema.clone(); let space = search_workload(vec![("query", original)]); let root = &space.roots[0].1; let group = space.candidates_for_target(root).expect("root memo group"); @@ -1065,34 +1012,35 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { .iter() .find(|candidate| candidate.strategy == "SemanticEquivalentRewriteStrategy") .expect("default search should compose the lowered PromQL query"); - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDAG(rewritten) = &candidate.replacement else { panic!("expected logical rewrite") }; + assert!(is_logical_rewrite(rewritten), "expected logical rewrite"); - assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Project { child, .. } = rewritten.as_ref() else { + assert_eq!(rewritten.schema, original_schema); + let Some(NonASAPOp::Project { child, .. }) = rewritten.non_asap() else { panic!("sum(count_over_time) needs a Float64 cast Project") }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, child, .. - } = child.as_ref() + }) = child.non_asap() else { panic!("expected one composed aggregate") }; assert_eq!(by.keys(), &[2]); assert!(matches!( measures.as_slice(), - [asap_types::pre_asap::AggIntent::Count { + [asap_types::ir::operator::AggIntent::Count { accuracy: AccuracyTarget::Exact }] )); assert!(matches!( - child.as_ref(), - QueryExpr::TimeRange { range, child } - if range.as_secs() == 300 && matches!(child.as_ref(), QueryExpr::Scan { .. }) + child.non_asap(), + Some(NonASAPOp::TimeRange { range, child, .. }) + if range.as_secs() == 300 && matches!(child.non_asap(), Some(NonASAPOp::Scan { .. })) )); } @@ -1100,50 +1048,42 @@ fn promql_sum_of_count_over_time_is_composed_by_default_search() { fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { // Real workload selection must expose the state-to-value edge; an outer // sketch must not interpret exact accumulator bytes as input samples. - let pre = Rc::new( - lower_promql( - "quantile(0.9, sum_over_time(m[1m]))", - AccuracyTarget::Epsilon(0.05), - ) - .unwrap(), - ); + let pre = lower_promql( + "quantile(0.9, sum_over_time(m[1m]))", + AccuracyTarget::Epsilon(0.05), + ) + .unwrap(); let space = search_workload(vec![("query", pre)]); - let selected = space.global_selection(&DefaultCostModel); + let selected = global_selection(&space, &DefaultCostModel); let plan = selected .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &plan.expr else { + // Stored timings are gone: time the plan with its outer summary + // maintained and read the timed copy. + let timed_plan = maintained(&plan); + let Some(ASAPOp::SummaryEstimate { summary_input, .. }) = timed_plan.asap() else { panic!("expected selected quantile summary"); }; - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { + let Some(ASAPOp::SummaryAgg { child, .. }) = summary_input.asap() else { panic!("expected maintained outer summary"); }; - let SummaryExpr::ValueOperation { - child: source, - operation, - timing, - } = &child.expr - else { + let Some(ASAPOp::FinalizeExactAccumulator { child: source }) = child.asap() else { panic!( "missing explicit accumulator finalization: {:?}", - child.expr + child.operator ); }; - assert!(matches!( - operation, - ValueOperation::FinalizeExactAccumulator - )); assert_eq!( - *timing, - asap_types::post_asap::ExecutionTiming::IngestionTime + child.timing, + Some(asap_types::ir::properties::ExecutionTiming::IngestionTime) ); assert!(matches!( - source.expr, - SummaryExpr::SummaryAgg { + source.asap(), + Some(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. - } + }) )); assert!(child .schema @@ -1155,12 +1095,13 @@ fn nested_summary_explicitly_finalizes_exact_child_at_ingestion_time() { .fields .iter() .any(|field| matches!(field.dtype, FieldDataType::Plain(DataType::Float64)))); - compile_post_asap_dag(&plan).expect("explicit boundary is a valid post-ASAP DAG"); + // Explicit boundary is a valid post-ASAP DAG. + post_asap_dag(&plan); } #[test] fn physical_node_owns_phase_independently_of_binary_payload() { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; + use asap_types::ir::properties::ExecutionTiming; for (query, expected) in [ ( // One selector: both operands cover the same series. @@ -1173,27 +1114,32 @@ fn physical_node_owns_phase_independently_of_binary_payload() { ), ] { let input = lower_promql(query, AccuracyTarget::Epsilon(0.05)).unwrap(); - // Backend lowering carries opaque series identity before candidate export. - let input = asap_types::pre_asap::schema::with_promql_series_identity(&input).unwrap(); - let search = search_workload(vec![("q", Rc::new(input))]); - let choice = search.global_selection(&DefaultCostModel); + let search = search_workload(vec![("q", input)]); + let choice = global_selection(&search, &DefaultCostModel); let plan = choice .assemble_selected_dag(&search.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&plan).unwrap(); + let dag = maintained_post_asap_dag(&plan); let node = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Binary { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::Relational { + operator: NonASAPOpKind::BinaryOp { .. } + } + ) + }) .unwrap(); assert_eq!(node.output_state.timing, expected); let wire = serde_json::to_value(&node.payload).unwrap(); assert!(wire.get("timing").is_none()); let mut obsolete = wire.clone(); obsolete["timing"] = serde_json::json!(expected.as_str()); - assert!(serde_json::from_value::(obsolete).is_err()); - let restored: PostAsapOperatorPayload = serde_json::from_value(wire).unwrap(); + assert!(serde_json::from_value::(obsolete).is_err()); + let restored: PhysicalASAPOperatorPayload = serde_json::from_value(wire).unwrap(); assert_eq!(restored, node.payload); } } @@ -1207,15 +1153,11 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { ) .unwrap(); let root = realize(&pre).unwrap(); - assert!(matches!(root.expr, SummaryExpr::BinaryOp { .. })); + assert!(matches!(root.non_asap(), Some(NonASAPOp::BinaryOp { .. }))); assert!(root.guarantee.is_none()); let space = search_workload_with_targets( - vec![( - "unproven", - Rc::new(pre), - Some(AccuracyTarget::Epsilon(0.01)), - )], - &asap_aware_mapping::default_strategies(), + vec![("unproven", pre, Some(AccuracyTarget::Epsilon(0.01)))], + &asap_logical_optimizer::default_strategies(), &DefaultAccuracyModel, ); let root_group = space @@ -1226,15 +1168,15 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { root_group.candidates.iter().any(|candidate| { matches!( &candidate.replacement, - Replacement::Summary(node) - if matches!(node.expr, SummaryExpr::BinaryOp { .. }) + Replacement::SubDAG(node) + if matches!(node.non_asap(), Some(NonASAPOp::BinaryOp { .. })) && node.guarantee.is_none() ) }), "backend must receive the uncertified ratio candidate for its own selection" ); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); assert!( selection .for_target(&space.roots[0].1) @@ -1247,7 +1189,7 @@ fn ddsketch_ratio_without_domain_proof_is_uncertified() { .assemble_selected_dag(&space.roots[0].1) .unwrap() .expect("materialized root"); - assert!(matches!(materialized.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!materialized.contains_asap()); } struct FixtureQuantileDomain { @@ -1255,7 +1197,7 @@ struct FixtureQuantileDomain { upper: f64, } impl AccuracyEvidenceProvider for FixtureQuantileDomain { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &OperatorNode) -> Option { Some(QuantileInputDomain { lower: self.lower, upper: self.upper, @@ -1278,15 +1220,12 @@ fn ddsketch_ratio_rejects_unsafe_domains() { (f64::MIN_POSITIVE / 2., f64::MIN_POSITIVE / 2.), ] { let evidence = FixtureQuantileDomain { lower, upper }; - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, @@ -1304,13 +1243,13 @@ fn ddsketch_ratio_rejects_unsafe_domains() { fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { struct PartialUnsafeDomain; impl AccuracyEvidenceProvider for PartialUnsafeDomain { - fn quantile_input_domain(&self, operand: &QueryExpr) -> Option { - let QueryExpr::Aggregate { measures, .. } = operand else { + fn quantile_input_domain(&self, operand: &OperatorNode) -> Option { + let Some(NonASAPOp::Aggregate { measures, .. }) = operand.non_asap() else { return None; }; matches!( measures.as_slice(), - [asap_types::pre_asap::agg_intent::AggIntent::Quantile { q, .. }] if *q == 0.9 + [asap_types::ir::operator::agg_intent::AggIntent::Quantile { q, .. }] if *q == 0.9 ) .then(|| QuantileInputDomain { lower: -1.0, @@ -1321,15 +1260,12 @@ fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { } } - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &PartialUnsafeDomain, @@ -1339,40 +1275,37 @@ fn ddsketch_ratio_rejects_one_invalid_domain_when_the_other_is_missing() { /// The committed planner alpha is exercised against the pinned sketch implementation. #[test] -fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { +fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_evaluations() { for sign in [-1., 1.] { let evidence = FixtureQuantileDomain { lower: if sign < 0. { -100. } else { 1. }, upper: if sign < 0. { -1. } else { 100. }, }; - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, ); let candidates = strategy.replacements(&TargetSubDAG::new(&pre)); - let Replacement::Summary(node) = &candidates[0].replacement else { + let Replacement::SubDAG(node) = &candidates[0].replacement else { panic!("summary") }; - let SummaryExpr::BinaryOp { lhs, rhs, .. } = &node.expr else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = node.non_asap() else { panic!("ratio") }; - let alpha = |node: &SummaryNode| { - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { - panic!("readout") + let alpha = |node: &OperatorNode| { + let Some(ASAPOp::SummaryEstimate { summary_input, .. }) = node.asap() else { + panic!("evaluation") }; - let SummaryExpr::SummaryAgg { + let Some(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = summary_input.asap() else { panic!("sketch") }; @@ -1407,12 +1340,12 @@ fn ddsketch_ratio_bound_holds_for_signed_pinned_sketch_readouts() { } } -/// Empty or overlarge population contracts cannot promise a supported readout. +/// Empty or overlarge population contracts cannot promise a supported evaluation. #[test] fn ddsketch_ratio_requires_a_supported_population_size() { struct PopulationEvidence(u64); impl AccuracyEvidenceProvider for PopulationEvidence { - fn quantile_input_domain(&self, _: &QueryExpr) -> Option { + fn quantile_input_domain(&self, _: &OperatorNode) -> Option { Some(QuantileInputDomain { lower: 1., upper: 10., @@ -1421,17 +1354,14 @@ fn ddsketch_ratio_requires_a_supported_population_size() { }) } } - let pre = Rc::new( - lower_promql( - "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", - AccuracyTarget::Epsilon(0.01), - ) - .unwrap(), - ); + let pre = lower_promql( + "quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); for count in [0, (1u64 << 53) + 1] { let evidence = PopulationEvidence(count); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, @@ -1441,7 +1371,7 @@ fn ddsketch_ratio_requires_a_supported_population_size() { } // Every `without` aggregation candidate exports a valid DAG: its summary state -// column carries the family instead of the readout's Float64 value. +// column carries the family instead of the evaluation's Float64 value. #[test] fn without_aggregation_candidates_export_valid_dags() { for accuracy in [ @@ -1452,16 +1382,16 @@ fn without_aggregation_candidates_export_valid_dags() { }, ] { for query in ["sum without (pod) (m)", "quantile without (pod) (0.5, m)"] { - let root = Rc::new(lower_promql(query, accuracy.clone()).unwrap()); + let root = lower_promql(query, accuracy.clone()).unwrap(); let space = search_workload_with_targets( vec![(0, root, Some(accuracy.clone()))], - &asap_aware_mapping::default_strategies(), + &asap_logical_optimizer::default_strategies(), &DefaultAccuracyModel, ); let inventory = space.enumerate_candidate_dags_for_root(&0, 65_536).unwrap(); assert!(!inventory.candidates.is_empty(), "{query}"); for (_, node) in inventory.candidates.iter().flatten() { - compile_post_asap_dag(node).unwrap_or_else(|e| panic!("{query}: {e}")); + post_asap_dag(node); } } } diff --git a/crates/integration-tests/tests/scan.rs b/crates/integration-tests/tests/scan.rs index bb7988d7a..1aff439c1 100644 --- a/crates/integration-tests/tests/scan.rs +++ b/crates/integration-tests/tests/scan.rs @@ -1,4 +1,4 @@ -//! `QueryExpr::Scan` — label matcher / predicate tests. +//! `NonASAPOp::Scan` — label matcher / predicate tests. //! //! The Scan schema is always [ts(0), value(1), label_a(2), label_b(3), …] //! where labels are appended alphabetically after dedup by the SchemaResolver. @@ -11,60 +11,63 @@ use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{CompareOpKind, Predicate, QueryExpr, ScalarValue, Source}; +use asap_types::ir::operator::Source; +use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; +use asap_types::ir::{ + ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, TimeRangeKind, +}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn bare_scan(metric: &str, labels: &[&str]) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn bare_scan(metric: &str, labels: &[&str]) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(labels), - } + }) } -fn instant(child: QueryExpr) -> QueryExpr { - QueryExpr::TimeRange { +fn instant(child: Rc) -> Rc { + node(NonASAPOp::TimeRange { range: Duration::from_secs(1), - child: Rc::new(child), - } + kind: TimeRangeKind::Instant, + child, + }) +} + +fn label_pred(col_id: usize, op: CompareOpKind, value: &str) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col_id)), + op, + right: Box::new(ScalarExpr::Literal(ScalarValue::Utf8(value.into()))), + semantics: ExprSemantics::Promql, + }) } fn eq_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), - })) + label_pred(col_id, CompareOpKind::Eq, value) } fn ne_pred(col_id: usize, value: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Ne, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(value.into()))), - })) + label_pred(col_id, CompareOpKind::Ne, value) } fn regex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::Regex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), - })) + label_pred(col_id, CompareOpKind::Regex, pattern) } fn notregex_pred(col_id: usize, pattern: &str) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(col_id)), - op: CompareOpKind::NotRegex, - right: Rc::new(QueryExpr::Literal(ScalarValue::Utf8(pattern.into()))), - })) + label_pred(col_id, CompareOpKind::NotRegex, pattern) } // #1 — bare metric name, no matchers @@ -80,13 +83,13 @@ fn q01_bare_scan() { // schema: [ts(0), value(1), job(2)] #[test] fn q02_equality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![eq_pred(2, "api-server")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job="api-server"}"#), expected); } @@ -94,13 +97,13 @@ fn q02_equality_predicate() { // schema: [ts(0), value(1), status(2)] #[test] fn q03_inequality_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![ne_pred(2, "500")], schema: metric_schema(&["status"]), - }); + })); assert_eq!(lower(r#"http_requests_total{status!="500"}"#), expected); } @@ -108,13 +111,13 @@ fn q03_inequality_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q04_regex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![regex_pred(2, "api.*")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job=~"api.*"}"#), expected); } @@ -122,13 +125,13 @@ fn q04_regex_predicate() { // schema: [ts(0), value(1), job(2)] #[test] fn q_notregex_predicate() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![notregex_pred(2, "internal.*")], schema: metric_schema(&["job"]), - }); + })); assert_eq!(lower(r#"http_requests_total{job!~"internal.*"}"#), expected); } @@ -137,13 +140,13 @@ fn q_notregex_predicate() { // predicates in same alphabetical order: job first, then status #[test] fn q_multi_two_predicates() { - let expected = instant(QueryExpr::Scan { + let expected = instant(node(NonASAPOp::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), }, predicates: vec![eq_pred(2, "api-server"), ne_pred(3, "500")], schema: metric_schema(&["job", "status"]), - }); + })); assert_eq!( lower(r#"http_requests_total{job="api-server",status!="500"}"#), expected, diff --git a/crates/integration-tests/tests/schema.rs b/crates/integration-tests/tests/schema.rs index 08b925dd1..c616c0d5f 100644 --- a/crates/integration-tests/tests/schema.rs +++ b/crates/integration-tests/tests/schema.rs @@ -1,7 +1,7 @@ //! `Schema::closed` propagation — open/closed invariant tests. //! -//! Verifies that `QueryExpr::output_schema()` propagates the open/closed -//! completeness flag correctly through a lowered query DAG. +//! Verifies that the derived `OperatorNode::schema` propagates the open/closed +//! completeness flag correctly through a lowered query tree. //! //! Key invariant: a PromQL scan is always `closed: false` (open) because its //! label set is runtime-only. The schema freezes to `closed: true` exactly at @@ -12,14 +12,14 @@ use asap_integration_tests::fixtures::lower_promql; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> asap_types::pre_asap::QueryExpr { +fn lower(q: &str) -> std::rc::Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } // bare scan is open — the metric's full label set is unknown at plan time #[test] fn schema_bare_scan_is_open() { - let s = lower("http_requests_total").output_schema().unwrap(); + let s = lower("http_requests_total").schema.clone(); assert!(!s.closed, "PromQL scan must be open"); } @@ -27,8 +27,8 @@ fn schema_bare_scan_is_open() { #[test] fn schema_filtered_scan_is_open() { let s = lower(r#"http_requests_total{job="api-server"}"#) - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "PromQL scan with predicates must remain open"); assert_eq!(s.fields.len(), 3, "[ts, value, job]"); } @@ -36,9 +36,7 @@ fn schema_filtered_scan_is_open() { // per-series rate is label-preserving → output stays open #[test] fn schema_rate_stays_open() { - let s = lower("rate(http_requests_total[5m])") - .output_schema() - .unwrap(); + let s = lower("rate(http_requests_total[5m])").schema.clone(); assert!(!s.closed, "per-series rate is label-preserving; stays open"); } @@ -46,24 +44,22 @@ fn schema_rate_stays_open() { #[test] fn schema_count_over_time_stays_open() { let s = lower("count_over_time(http_requests_total[5m])") - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "per-series count_over_time stays open"); } // cross-series sum with no group keys freezes to closed #[test] fn schema_sum_freezes_to_closed() { - let s = lower("sum(http_requests_total)").output_schema().unwrap(); + let s = lower("sum(http_requests_total)").schema.clone(); assert!(s.closed, "cross-series aggregate must freeze to closed"); } // cross-series sum grouped by job also freezes to closed #[test] fn schema_sum_by_job_freezes_to_closed() { - let s = lower("sum by (job) (http_requests_total)") - .output_schema() - .unwrap(); + let s = lower("sum by (job) (http_requests_total)").schema.clone(); assert!( s.closed, "grouped cross-series aggregate must freeze to closed" @@ -74,8 +70,8 @@ fn schema_sum_by_job_freezes_to_closed() { #[test] fn schema_sum_over_rate_freezes_to_closed() { let s = lower("sum by (job) (rate(http_requests_total[5m]))") - .output_schema() - .unwrap(); + .schema + .clone(); assert!( s.closed, "cross-series aggregate over rate must freeze to closed" @@ -86,8 +82,8 @@ fn schema_sum_over_rate_freezes_to_closed() { #[test] fn schema_binary_op_two_open_stays_open() { let s = lower("http_requests_total / http_errors_total") - .output_schema() - .unwrap(); + .schema + .clone(); assert!(!s.closed, "binary op over two open scans must stay open"); } @@ -95,8 +91,8 @@ fn schema_binary_op_two_open_stays_open() { #[test] fn schema_binary_op_two_closed_is_closed() { let s = lower("sum by (job) (http_requests_total) / sum by (job) (http_errors_total)") - .output_schema() - .unwrap(); + .schema + .clone(); assert!( s.closed, "binary op over two closed aggregates must be closed" diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs index dd4426c22..46e0f957c 100644 --- a/crates/integration-tests/tests/sql_to_physical.rs +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -1,19 +1,21 @@ //! SQL frontend, candidate selection, physical compilation and fresh-run execution. -use asap_aware_mapping::{search_workload, DefaultCostModel}; -use asap_frontend_sql::{lower_sql, SqlCatalog}; -use asap_physical_operators::{ +mod physical_common; +use asap_executor::{ physical_planner::{compile, InputContract, Source}, runtime::{Limits, RunContext, Scope}, sources::{DataSources, MemorySource}, values::{Batch, Value}, }; -use asap_types::{ - post_asap::{compile_post_asap_dag, FieldDataType, PostAsapOperatorPayload}, - pre_asap::{DataType, Field, QueryExpr, Schema}, - types::AccuracyTarget, -}; +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_logical_optimizer::search_workload; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::DefaultCostModel; +use asap_types::ir::export::PhysicalASAPOperatorPayload; +use asap_types::ir::schema::{DataType, Field, FieldDataType, Schema}; +use asap_types::types::AccuracyTarget; use futures::StreamExt; -use std::{collections::BTreeMap, rc::Rc, sync::Arc}; +use physical_common::compile_physical_asap_dag; +use std::{collections::BTreeMap, sync::Arc}; /// SQL filtering and grouped aggregation survive logical/physical lowering; /// rebinding the compiled DAG runs against new data rather than cached results. @@ -30,26 +32,23 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { "SELECT service, SUM(value) AS total FROM metrics WHERE value > 1 GROUP BY service", "SELECT service, SUM(value) AS total FROM metrics GROUP BY service", ] { - let logical = Rc::new( - lower_sql(query, &catalog, AccuracyTarget::Exact) - .await - .unwrap(), - ); + let logical = lower_sql(query, &catalog, AccuracyTarget::Exact) + .await + .unwrap(); let space = search_workload(vec![("sql", logical)]); - let selected = space - .global_selection(&DefaultCostModel) + let selected = global_selection(&space, &DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&selected).unwrap(); + let dag = compile_physical_asap_dag(&selected).unwrap(); let scan = dag .nodes .iter() .find(|node| { matches!( &node.payload, - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Scan { .. } + PhysicalASAPOperatorPayload::Relational { + operator: asap_types::ir::export::NonASAPOpKind::Scan { .. } } ) }) @@ -62,7 +61,7 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { let plan = compile( &dag, BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], + &[u64::from(dag.roots[0].0)], ) .unwrap(); for multiplier in [1., 2.] { @@ -88,12 +87,18 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { .collect() }) .collect(); - let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { - unreachable!() - }; - let QueryExpr::Scan { source, .. } = expression else { + let PhysicalASAPOperatorPayload::Relational { + operator: + asap_types::ir::export::NonASAPOpKind::Scan { + source, + predicates: _, + schema: _scan_schema, + }, + } = &scan.payload + else { unreachable!() }; + let expression = asap_types::ir::OperatorNode::reachable(&selected).into_iter().find(|n| matches!(n.non_asap(), Some(asap_types::ir::NonASAPOp::Scan { source: s, .. }) if s == source)).unwrap(); let mut sources = DataSources::default(); sources .register( @@ -110,7 +115,7 @@ async fn sql_filter_grouped_sum_executes_and_rebinds() { let bound = plan .instantiate(BTreeMap::from([( u64::from(scan.id.0), - Box::new(sources.bind(expression).unwrap()) as Source<'_>, + Box::new(sources.bind(&expression).unwrap()) as Source<'_>, )])) .unwrap(); let mut stream = bound diff --git a/crates/integration-tests/tests/sql_to_post_asap.rs b/crates/integration-tests/tests/sql_to_post_asap.rs index 4cecd6952..e0f1b6a73 100644 --- a/crates/integration-tests/tests/sql_to_post_asap.rs +++ b/crates/integration-tests/tests/sql_to_post_asap.rs @@ -1,63 +1,100 @@ //! End-to-end SQL query-string → post-ASAP IR pin (issue #191). //! //! The SQL counterpart of `promql_to_post_asap.rs`: drives SQL text — -//! `lower_sql` (text → pre-ASAP `QueryExpr`) → -//! `SketchAlgorithmStrategy::replacements` (pre-ASAP → post-ASAP -//! `SummaryExpr`, see [`realize`] below) — and pins the resulting -//! sketch-vs-exact-accumulator shape node by node, the way -//! `promql_to_post_asap.rs` does for PromQL. +//! `lower_sql` (text → non-ASAP `OperatorNode` tree) → +//! `ASAPStrategies::replacements` (→ a tree with ASAP operators, +//! see [`realize`] below) — and pins the resulting sketch-vs-exact-accumulator +//! shape node by node, the way `promql_to_post_asap.rs` does for PromQL. //! //! ## A structural wrinkle PromQL doesn't have //! -//! `lower_promql` returns a *bare* `QueryExpr::Aggregate` for a top-level +//! `lower_promql` returns a *bare* `NonASAPOp::Aggregate` for a top-level //! aggregation (`sum by (job) (m)`, `quantile(0.99, …)`), so [`realize`] can //! bind it directly at the DAG root. `lower_sql` never does: DataFusion's //! planner always wraps even a single, unaliased aggregate in an identity //! `Project` (confirmed below), so a SQL DAG's *root* is normally `Project { //! child: Aggregate { .. } }`. Final materialization retains that projection -//! as a query-time value operation and independently plans its child, keeping +//! as a query-time non-ASAP node and independently plans its child, keeping //! both SELECT-list semantics and the summary-bound aggregate visible. use std::rc::Rc; -use asap_aware_mapping::replacement::{keep_pre_asap, RealizationError}; -use asap_aware_mapping::{ - search_workload, DefaultCostModel, Replacement, ReplacementStrategy, ReplacementSubDAG, - SketchAlgorithmStrategy, TargetSubDAG, -}; use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog}; -use asap_types::post_asap::{ - compile_post_asap_dag, EdgeRole, ExactKind, ExactParams, FieldDataType, GroupingStrategy, - PostAsapOperatorPayload, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, - SummaryExpr, SummaryNode, SummaryUpdate, ValueOperation, +use asap_integration_tests::post_asap::post_asap_dag; +use asap_logical_optimizer::pass1::replacement::{retain_exact, RealizationError}; +use asap_logical_optimizer::{ + search_workload, ASAPStrategies, Replacement, ReplacementStrategy, ReplacementSubDAG, + TargetSubDAG, +}; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::DefaultCostModel; +use asap_types::ir::export::{ + EdgeRole, NonASAPOpKind, PhysicalASAPNodeId, PhysicalASAPOperatorPayload, WirePredicate, + WireScalarExpr, }; -use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::operator::operator_properties::Reduction; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{ + ExactKind, ExactParams, FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, + SketchParams, SketchStatistic, SummaryUpdate, +}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, Predicate, ScalarExpr}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -/// This crate has no "bind me one DAG" public API any more — -/// `SketchAlgorithmStrategy::replacements` always returns every candidate, and +/// This crate has no "bind me one tree" public API any more — +/// `ASAPStrategies::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the -/// take-the-first-(`cost_model`-preferred)-candidate pattern so the +/// take-the-first-(`cost_model`-preferred)-summary-candidate pattern so the /// single-answer pins below don't all repeat it by hand. -fn realize(expr: &QueryExpr) -> Result, RealizationError> { - let root = Rc::new(expr.clone()); - let target = TargetSubDAG::new(&root); - match SketchAlgorithmStrategy::default_cost_model() - .replacements(&target) +fn realize(target: &Rc) -> Result, RealizationError> { + let target_dag = TargetSubDAG::new(target); + match ASAPStrategies::default() + .replacements(&target_dag) .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. - }) => Ok(node), - _ => keep_pre_asap(&root), + }) if node.contains_asap() => Ok(node), + _ => retain_exact(target), + } + .inspect(|node| { + node.validate_structure() + .expect("planned dag satisfies the unified IR contract") + }) +} + +/// The single input of a unary non-ASAP node (Project, Filter, Sort, ...) or +/// of a `FinalizeExactAccumulator`; `None` for anything else. +fn unary_child(node: &OperatorNode) -> Option<&Rc> { + match &node.operator { + Operator::NonASAP(op) => match op.children().as_slice() { + [child] => Some(*child), + _ => None, + }, + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => Some(child), + Operator::ASAP(_) => None, } } +/// A sub-DAG kept as plain (non-ASAP) work: no ASAP operator anywhere below. +fn is_kept_non_asap(node: &OperatorNode) -> bool { + node.non_asap().is_some() && !node.contains_asap() +} + +/// Mirror a scalar-only predicate (no operator references) to its wire form. +fn wire_pred(pred: &Predicate) -> WirePredicate { + WirePredicate(WireScalarExpr::from_expr( + &pred.0, + &mut |_: &Rc| -> PhysicalASAPNodeId { + panic!("fixture predicate references no operator") + }, + )) +} + fn dtype<'a>(schema: &'a Schema, name: &str) -> &'a FieldDataType { &schema .fields @@ -89,7 +126,7 @@ fn catalog() -> SqlCatalog { ) } -async fn lower(sql: &str, accuracy: AccuracyTarget) -> QueryExpr { +async fn lower(sql: &str, accuracy: AccuracyTarget) -> Rc { lower_sql(sql, &catalog(), accuracy) .await .unwrap_or_else(|e| panic!("lower failed for {sql:?}: {e}")) @@ -121,26 +158,27 @@ async fn clickhouse_temporal_sql_reuses_rate_and_increase_physical_summaries() { .expect("explicit temporal SQL must lower"); let physical = realize(inner_aggregate(&pre_asap)).expect("temporal reducer must be planned"); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, reduction, child, .. - } = &physical.expr + }) = &physical.operator else { - panic!("expected a shared SummaryAgg, got {:?}", physical.expr); + panic!("expected a shared SummaryAgg, got {:?}", physical.operator); }; assert_eq!(family, &expected); assert_eq!(reduction, &Reduction::PerEntity); - let SummaryExpr::KeepPreAsap(raw) = &child.expr else { - panic!( - "expected a retained temporal SQL input, got {:?}", - child.expr - ); - }; - assert!(matches!(raw.as_ref(), QueryExpr::TimeRange { range, child } + assert!( + is_kept_non_asap(child), + "expected a retained temporal SQL input, got {:?}", + child.operator + ); + assert!( + matches!(child.non_asap(), Some(NonASAPOp::TimeRange { range, child, .. }) if *range == std::time::Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Project { .. }))); + && matches!(child.non_asap(), Some(NonASAPOp::Project { .. }))) + ); } } @@ -156,47 +194,42 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { SELECT service, {function}(latency, ts, {window_ms}) AS v \ FROM metrics GROUP BY service)" ); - let pre_asap = Rc::new( - lower_sql_dialect( - &sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect("nested temporal SQL must lower"), - ); + let pre_asap = lower_sql_dialect( + &sql, + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .expect("nested temporal SQL must lower"); let space = search_workload(vec![("nested", Rc::clone(&pre_asap))]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - fn has_temporal_summary(node: &SummaryNode) -> bool { - match &node.expr { - SummaryExpr::SummaryAgg { + fn has_temporal_summary(node: &OperatorNode) -> bool { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate | ExactKind::Increase, _), .. - } => true, - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryEstimate { - summary_input: child, - .. - } => has_temporal_summary(child), - _ => false, + }) => true, + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + has_temporal_summary(summary_input) + } + _ => unary_child(node).is_some_and(|child| has_temporal_summary(child)), } } assert!( has_temporal_summary(&root), "inner {function} was hidden: {root:?}" ); - let dag = compile_post_asap_dag(&root).expect("nested SQL DAG must compile"); + let dag = post_asap_dag(&root); assert!(dag.nodes.iter().any(|node| matches!( node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Exact(_), - .. + PhysicalASAPOperatorPayload::Relational { + operator: NonASAPOpKind::Aggregate { .. }, } ))); } @@ -204,11 +237,14 @@ async fn clickhouse_outer_sum_recursively_binds_inner_temporal_aggregate() { /// The `Aggregate` node beneath the identity `Project` DataFusion's planner /// always wraps a top-level aggregate in — see the module docs above. -fn inner_aggregate(qe: &QueryExpr) -> &QueryExpr { - match qe { - QueryExpr::Project { child, .. } => inner_aggregate(child), - QueryExpr::Aggregate { .. } => qe, - other => panic!("expected a Project{{Aggregate}} shape, got {other:?}"), +fn inner_aggregate(node: &Rc) -> &Rc { + match node.non_asap() { + Some(NonASAPOp::Project { child, .. }) => inner_aggregate(child), + Some(NonASAPOp::Aggregate { .. }) => node, + _ => panic!( + "expected a Project{{Aggregate}} shape, got {:?}", + node.operator + ), } } @@ -221,32 +257,27 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { AccuracyTarget::Epsilon(0.01), ) .await; - assert!( - matches!(pre_asap, QueryExpr::Project { .. }), - "sanity: a SQL root is a Project, unlike lower_promql's bare Aggregate" - ); - let pre_asap = Rc::new(pre_asap); - let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); - let selection = space.global_selection(&DefaultCostModel); - let root = selection - .assemble_selected_dag(&space.roots[0].1) - .expect("materialization failed") - .expect("root must be discovered"); - let QueryExpr::Project { + let Some(NonASAPOp::Project { cols: expected_cols, qualifier: expected_qualifier, .. - } = pre_asap.as_ref() + }) = pre_asap.non_asap() else { - unreachable!() + panic!("sanity: a SQL root is a Project, unlike lower_promql's bare Aggregate"); }; - let SummaryExpr::ValueOperation { + let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); + let selection = global_selection(&space, &DefaultCostModel); + let root = selection + .assemble_selected_dag(&space.roots[0].1) + .expect("materialization failed") + .expect("root must be discovered"); + let Some(NonASAPOp::Project { child, - operation: asap_types::post_asap::ValueOperation::Project { cols, qualifier }, - .. - } = &root.expr + cols, + qualifier, + }) = root.non_asap() else { - panic!("expected retained Project root, got {:?}", root.expr); + panic!("expected retained Project root, got {:?}", root.operator); }; assert_eq!(cols, expected_cols, "projection expressions and aliases"); assert_eq!(qualifier, expected_qualifier, "projection qualifier"); @@ -256,7 +287,10 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { FieldDataType::Plain(DataType::Float64) ); assert!( - matches!(child.expr, SummaryExpr::SummaryEstimate { .. }), + matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ), "the Aggregate under Project must be summary-bound" ); } @@ -265,63 +299,62 @@ async fn sql_full_query_retains_project_and_binds_inner_aggregate() { /// aggregates are independently selected as physical summaries. #[tokio::test] async fn sql_join_recursively_binds_both_temporal_aggregate_children() { - let pre_asap = Rc::new( - lower_sql_dialect( - "SELECT a.service, a.v / b.v AS ratio FROM \ - (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='errors' GROUP BY service) a \ - INNER JOIN \ - (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='requests' GROUP BY service) b \ - ON b.service=a.service", - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .expect("two-subquery rate ratio must lower"), - ); + let pre_asap = lower_sql_dialect( + "SELECT a.service, a.v / b.v AS ratio FROM \ + (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='errors' GROUP BY service) a \ + INNER JOIN \ + (SELECT service, asap_rate(latency, ts, 300000) AS v FROM metrics WHERE service='requests' GROUP BY service) b \ + ON b.service=a.service", + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .expect("two-subquery rate ratio must lower"); let space = search_workload(vec![("ratio", Rc::clone(&pre_asap))]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let SummaryExpr::ValueOperation { - child: join, - operation: ValueOperation::Project { cols, .. }, - .. - } = &root.expr + let Some(NonASAPOp::Project { + child: join, cols, .. + }) = root.non_asap() else { panic!( "expected Project above relational join, got {:?}", - root.expr + root.operator ); }; assert!(matches!( &cols[1].expr, - QueryExpr::Arithmetic { - op: asap_types::pre_asap::ArithmeticOpKind::Div, + ScalarExpr::Arithmetic { + op: asap_types::ir::scalar::ArithmeticOpKind::Div, .. } )); - let SummaryExpr::RelationalJoin { + let Some(NonASAPOp::Join { left, right, kind, pred, - pruning: None, - } = &join.expr + }) = join.non_asap() else { - panic!("expected read-time relational join, got {:?}", join.expr); + panic!( + "expected read-time relational join, got {:?}", + join.operator + ); }; - assert_eq!(kind, &asap_types::pre_asap::JoinKind::Inner); + assert_eq!(kind, &asap_types::ir::operator::JoinKind::Inner); assert!(matches!( - pred.0.as_ref(), - QueryExpr::Compare { + &pred.0, + ScalarExpr::Compare { left, - op: asap_types::pre_asap::CompareOpKind::Eq, + op: asap_types::ir::scalar::CompareOpKind::Eq, right, - } if matches!(left.as_ref(), QueryExpr::Column(0)) - && matches!(right.as_ref(), QueryExpr::Column(2)) + .. + } if matches!(left.as_ref(), ScalarExpr::Column(0)) + && matches!(right.as_ref(), ScalarExpr::Column(2)) )); assert_eq!( join.schema @@ -332,39 +365,44 @@ async fn sql_join_recursively_binds_both_temporal_aggregate_children() { vec!["service", "v", "service", "v"] ); for child in [left, right] { - let SummaryExpr::ValueOperation { - child: aggregate, - operation: ValueOperation::Project { .. }, - .. - } = &child.expr + let Some(NonASAPOp::Project { + child: aggregate, .. + }) = child.non_asap() else { - panic!("derived table Project was not retained: {:?}", child.expr); + panic!( + "derived table Project was not retained: {:?}", + child.operator + ); }; - let SummaryExpr::ValueOperation { - child: aggregate, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } = &aggregate.expr + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: aggregate }) = + &aggregate.operator else { panic!("derived table Project must consume finalized exact values"); }; assert!(matches!( - aggregate.expr, - SummaryExpr::SummaryAgg { + aggregate.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), .. - } + }) )); } assert!(join .guarantee .as_ref() .is_some_and(|value| value.is_exact())); - let dag = compile_post_asap_dag(&root).expect("join DAG must compile"); + let dag = post_asap_dag(&root); let join_id = dag .nodes .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::RelationalJoin { .. })) + .find(|node| { + matches!( + node.payload, + PhysicalASAPOperatorPayload::Relational { + operator: NonASAPOpKind::Join { .. }, + } + ) + }) .expect("relational join node") .id; let roles = dag @@ -383,29 +421,27 @@ async fn unsupported_sql_join_shapes_remain_fail_closed() { "SELECT a.service FROM (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) a INNER JOIN (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) b ON a.v>b.v", "SELECT a.service FROM (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) a INNER JOIN (SELECT service, asap_rate(latency, ts, 300000) v FROM metrics GROUP BY service) b ON a.service=a.service", ] { - let pre_asap = Rc::new( - lower_sql_dialect( - sql, - &catalog(), - SqlDialect::ClickhouseSQL, - AccuracyTarget::Exact, - ) - .await - .unwrap_or_else(|error| panic!("join must lower before fail-closed mapping: {error}")), - ); + let pre_asap = lower_sql_dialect( + sql, + &catalog(), + SqlDialect::ClickhouseSQL, + AccuracyTarget::Exact, + ) + .await + .unwrap_or_else(|error| panic!("join must lower before fail-closed mapping: {error}")); let space = search_workload(vec![("unsupported-join", Rc::clone(&pre_asap))]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") .expect("root must be discovered"); - let SummaryExpr::ValueOperation { child, .. } = &root.expr else { - panic!("SQL projection must remain explicit: {:?}", root.expr); + let Some(NonASAPOp::Project { child, .. }) = root.non_asap() else { + panic!("SQL projection must remain explicit: {:?}", root.operator); }; assert!( - matches!(child.expr, SummaryExpr::KeepPreAsap(_)), + is_kept_non_asap(child), "unsupported join was partially accelerated: {:?}", - child.expr + child.operator ); } } @@ -414,18 +450,16 @@ async fn unsupported_sql_join_shapes_remain_fail_closed() { /// explicit read-time nodes while the aggregate is summary-bound. #[tokio::test] async fn sql_relational_parents_retain_summary_bound_aggregate() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.p FROM \ - (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ - FROM metrics GROUP BY service) t \ - WHERE t.p > 100 ORDER BY t.p DESC LIMIT 5", - AccuracyTarget::Epsilon(0.01), - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.p FROM \ + (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ + FROM metrics GROUP BY service) t \ + WHERE t.p > 100 ORDER BY t.p DESC LIMIT 5", + AccuracyTarget::Epsilon(0.01), + ) + .await; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") @@ -437,28 +471,29 @@ async fn sql_relational_parents_retain_summary_bound_aggregate() { let mut saw_sort = false; let mut saw_limit = false; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, operation, .. - } => { - match operation { - asap_types::post_asap::ValueOperation::Project { .. } => saw_project = true, - asap_types::post_asap::ValueOperation::Filter { .. } => saw_filter = true, - asap_types::post_asap::ValueOperation::Sort { .. } => saw_sort = true, - asap_types::post_asap::ValueOperation::Limit { n, offset, .. } => { - assert_eq!((*n, *offset), (5, 0)); - saw_limit = true; - } - _ => {} - } - node = child; - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - assert!(matches!(summary_input.expr, SummaryExpr::SummaryAgg { .. })); - break; + if let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator { + assert!(matches!( + summary_input.operator, + Operator::ASAP(ASAPOp::SummaryAgg { .. }) + )); + break; + } + match node.non_asap() { + Some(NonASAPOp::Project { .. }) => saw_project = true, + Some(NonASAPOp::Filter { .. }) => saw_filter = true, + Some(NonASAPOp::Sort { .. }) => saw_sort = true, + Some(NonASAPOp::Limit { n, offset, .. }) => { + assert_eq!((*n, *offset), (Some(5), 0)); + saw_limit = true; } - other => panic!("expected relational parents over SummaryEstimate, got {other:?}"), + _ => {} } + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected relational parents over SummaryEstimate, got {:?}", + node.operator + ) + }); } assert!(saw_project && saw_filter && saw_sort && saw_limit); } @@ -468,46 +503,54 @@ async fn sql_relational_parents_retain_summary_bound_aggregate() { /// may be dropped or moved across the aggregation boundary. #[tokio::test] async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.p FROM \ - (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ - FROM metrics WHERE service = 'api' GROUP BY service) t \ - WHERE t.p > 100", - AccuracyTarget::Epsilon(0.01), - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.p FROM \ + (SELECT service, approx_percentile_cont(latency, 0.9) AS p \ + FROM metrics WHERE service = 'api' GROUP BY service) t \ + WHERE t.p > 100", + AccuracyTarget::Epsilon(0.01), + ) + .await; let expected_read_predicate = { - let mut node = pre_asap.as_ref(); + let mut node = &pre_asap; loop { - match node { - QueryExpr::Filter { pred, .. } => break pred.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, - other => panic!("expected a Filter above the aggregate, got {other:?}"), + match node.non_asap() { + Some(NonASAPOp::Filter { pred, .. }) => break pred.clone(), + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => node = child, + _ => panic!( + "expected a Filter above the aggregate, got {:?}", + node.operator + ), } } }; let expected_source_predicates = { - let mut node = pre_asap.as_ref(); + let mut node = &pre_asap; loop { - match node { - QueryExpr::Scan { predicates, .. } => break predicates.clone(), - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Aggregate { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => node = child, - other => panic!("expected a unary SQL plan over Scan, got {other:?}"), + match node.non_asap() { + Some(NonASAPOp::Scan { predicates, .. }) => break predicates.clone(), + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => node = child, + _ => panic!( + "expected a unary SQL plan over Scan, got {:?}", + node.operator + ), } } }; assert_eq!(expected_source_predicates.len(), 1, "fixture source WHERE"); let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") @@ -516,41 +559,39 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { let mut node = root.as_ref(); let mut retained_read_predicate = None; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Filter { pred }, - .. - } => { - retained_read_predicate = Some(pred.clone()); - node = child; - } - SummaryExpr::ValueOperation { child, .. } => node = child, - SummaryExpr::SummaryEstimate { summary_input, .. } => { - let SummaryExpr::SummaryAgg { child, .. } = &summary_input.expr else { - panic!("expected SummaryAgg below SummaryEstimate"); - }; - let SummaryExpr::KeepPreAsap(raw_input) = &child.expr else { - panic!("expected raw summary population below SummaryAgg"); - }; - let QueryExpr::Scan { predicates, .. } = raw_input.as_ref() else { - panic!("expected source selection to remain a Scan"); - }; - assert_eq!(predicates, &expected_source_predicates); - break; - } - other => panic!("expected read-time operations over a summary, got {other:?}"), + if let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg below SummaryEstimate"); + }; + assert!( + is_kept_non_asap(child), + "expected raw summary population below SummaryAgg" + ); + let Some(NonASAPOp::Scan { predicates, .. }) = child.non_asap() else { + panic!("expected source selection to remain a Scan"); + }; + assert_eq!(predicates, &expected_source_predicates); + break; } + if let Some(NonASAPOp::Filter { pred, .. }) = node.non_asap() { + retained_read_predicate = Some(pred.clone()); + } + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected read-time operations over a summary, got {:?}", + node.operator + ) + }); } assert_eq!(retained_read_predicate, Some(expected_read_predicate)); - let dag = compile_post_asap_dag(&root).expect("typed DAG compilation failed"); + let dag = post_asap_dag(&root); + let expected_wire = wire_pred(retained_read_predicate.as_ref().unwrap()); assert!(dag.nodes.iter().any(|node| matches!( &node.payload, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Filter { pred }, - .. - } if pred == retained_read_predicate.as_ref().unwrap() + PhysicalASAPOperatorPayload::Relational { + operator: NonASAPOpKind::Filter { pred }, + } if *pred == expected_wire ))); } @@ -559,17 +600,15 @@ async fn sql_filter_keeps_read_predicate_and_summary_population_selection() { /// read-time operation. #[tokio::test] async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { - let pre_asap = Rc::new( - lower( - "SELECT t.service, t.avg_bytes FROM \ - (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t \ - WHERE t.avg_bytes > 100", - AccuracyTarget::Exact, - ) - .await, - ); + let pre_asap = lower( + "SELECT t.service, t.avg_bytes FROM \ + (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t \ + WHERE t.avg_bytes > 100", + AccuracyTarget::Exact, + ) + .await; let space = search_workload(vec![("query", Rc::clone(&pre_asap))]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); let root = selection .assemble_selected_dag(&space.roots[0].1) .expect("materialization failed") @@ -578,22 +617,20 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { let mut node = root.as_ref(); let mut saw_filter = false; loop { - match &node.expr { - SummaryExpr::ValueOperation { - child, operation, .. - } => { - saw_filter |= matches!(operation, ValueOperation::Filter { .. }); - node = child; - } - SummaryExpr::KeepPreAsap(fallback) => { - assert!( - matches!(fallback.as_ref(), QueryExpr::BinaryOp { .. }), - "AVG's unsupported rewritten child should be opaque, got {fallback:?}" - ); - break; - } - other => panic!("expected local value operations over fallback child, got {other:?}"), + if let Some(NonASAPOp::BinaryOp { .. }) = node.non_asap() { + assert!( + is_kept_non_asap(node), + "AVG's unsupported rewritten child should be kept whole, got {node:?}" + ); + break; } + saw_filter |= matches!(node.non_asap(), Some(NonASAPOp::Filter { .. })); + node = unary_child(node).unwrap_or_else(|| { + panic!( + "expected local value operations over fallback child, got {:?}", + node.operator + ) + }); } assert!(saw_filter, "supported Filter must remain explicit"); } @@ -604,7 +641,7 @@ async fn sql_filter_preserves_local_fallback_boundary_for_unsupported_child() { /// ```text /// SummaryEstimate { query: Quantile{0.99} } → {…: Float64} /// └─ SummaryAgg { Kll{k:269}, input: metrics.latency } → {…: Sketch(Kll, {k:269})} -/// └─ KeepPreAsap(Scan) → {ts, service, latency, bytes} +/// └─ Scan (kept non-ASAP) → {ts, service, latency, bytes} /// ``` /// /// The SQL counterpart of `promql_to_post_asap.rs`'s @@ -622,12 +659,12 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, SketchStatistic::Quantile { q } if *q == 0.99)); assert_eq!( @@ -641,15 +678,15 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { "the summary-state type must not propagate past the estimate" ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -679,10 +716,12 @@ async fn sql_quantile_binds_kll_sketch_over_named_column() { ) ); - let SummaryExpr::KeepPreAsap(kept_leaf) = &child.expr else { - panic!("expected KeepPreAsap leaf, got {:?}", child.expr); - }; - assert!(matches!(kept_leaf.as_ref(), QueryExpr::Scan { .. })); + assert!( + is_kept_non_asap(child), + "expected a kept non-ASAP leaf, got {:?}", + child.operator + ); + assert!(matches!(child.non_asap(), Some(NonASAPOp::Scan { .. }))); assert!( child .schema @@ -709,12 +748,12 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, SketchStatistic::Cardinality)); assert_eq!( @@ -723,14 +762,14 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { "COUNT(DISTINCT …) reads back out as an integer count" ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, reduction, .. - } = &summary_input.expr + }) = &summary_input.operator else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, @@ -763,11 +802,11 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { .await; let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family, reduction, .. - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryAgg, got {:?}", root.expr); + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, @@ -788,37 +827,41 @@ async fn sql_exact_workload_binds_accumulators_not_sketches() { let agg = inner_aggregate(&pre_asap); let root = realize(agg).expect("binding failed"); assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(_)), + is_kept_non_asap(&root), "avg has no mergeable accumulator — stays logical" ); + assert!( + root.guarantee.as_ref().is_some_and(|g| g.is_exact()), + "a kept logical sub_dag is exact" + ); } #[tokio::test] async fn map_projection_export_preserves_unsupported_child_boundary() { - let pre = Rc::new(lower_sql_dialect( + let pre = lower_sql_dialect( "SELECT map('job', t.service) AS labels, t.avg_bytes FROM (SELECT service, AVG(bytes) AS avg_bytes FROM metrics GROUP BY service) t WHERE t.avg_bytes > 100", &catalog(), SqlDialect::ClickhouseSQL, AccuracyTarget::Exact, - ).await.unwrap()); + ).await.unwrap(); let space = search_workload(vec![("map_query", pre)]); - let root = space - .global_selection(&DefaultCostModel) + let root = global_selection(&space, &DefaultCostModel) .assemble_selected_dag(&space.roots[0].1) .unwrap() .unwrap(); - let dag = compile_post_asap_dag(&root).unwrap(); + let dag = post_asap_dag(&root); assert!(dag.nodes.iter().any(|node| matches!(&node.payload, - PostAsapOperatorPayload::Value { operation: ValueOperation::Project { cols, .. }, .. } - if cols.iter().any(|item| matches!(&item.expr, QueryExpr::FunctionCall { name, .. } if name == "map")) + PhysicalASAPOperatorPayload::Relational { operator: NonASAPOpKind::Project { cols, .. } } + if cols.iter().any(|item| matches!(&item.expr, WireScalarExpr::FunctionCall { name, .. } if name == "map")) ))); let mut node = root.as_ref(); loop { - match &node.expr { - SummaryExpr::ValueOperation { child, .. } => node = child, - SummaryExpr::KeepPreAsap(child) => { - assert!(matches!(child.as_ref(), QueryExpr::BinaryOp { .. })); - break; - } - other => panic!("unexpected map/fallback composition: {other:?}"), + if let Some(NonASAPOp::BinaryOp { .. }) = node.non_asap() { + assert!( + is_kept_non_asap(node), + "fallback child must stay whole: {node:?}" + ); + break; } + node = unary_child(node) + .unwrap_or_else(|| panic!("unexpected map/fallback composition: {:?}", node.operator)); } } diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs deleted file mode 100644 index eeade3ae5..000000000 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ /dev/null @@ -1,1147 +0,0 @@ -//! End-to-end coverage for workload-aware summary-maintenance planning: -//! source workload -> PromQL lowering -> candidate search -> -//! summary-maintenance lifecycle selection -> materialized deployment guarantees. - -use std::rc::Rc; - -use asap_aware_mapping::cost_model::Cost; -use asap_aware_mapping::CostRate; -use asap_aware_mapping::{ - assemble_selected_dag_with_summary_maintenance_lifecycles, export_summary_maintenance_plan, - global_selection_with_summary_maintenance_lifecycles, search_workload_with, CostModel, Horizon, - SummaryMaintenanceCapabilities, SummaryMaintenanceLifecycleCapabilities, - SummaryMaintenanceLifecycleCostInputs, SummaryMaintenanceLifecycleRejection, WorkloadDemand, -}; -use asap_frontend_promql::lower_promql_workload; -use asap_types::post_asap::{ - EvaluationSchedule, SummaryMaintenanceLifecycle, SummaryMaintenanceMode, SummaryNode, -}; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::types::AccuracyTarget; -use asap_types::workload::{ - AccuracyRequirement, BatchEntry, DataArrival, DataWorkload, DurationMs, Evidence, - EvidenceSource, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, - QueryTimeScope, QueryWorkload, Rate, RepeatedDemand, RepeatingEntry, RepetitionInterval, - TimeSelection, -}; - -const NOW_MS: u64 = 1_000_000; - -struct FullyCostedRuntime; - -impl CostModel for FullyCostedRuntime { - fn raw_query_recompute_total_cost( - &self, - _target: &asap_types::pre_asap::QueryExpr, - _expected_reads: f64, - ) -> Option { - Some(Cost(1_000.0)) - } - - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(Cost(10.0)), - maintenance_cost_per_update: Some(Cost(1.0)), - summary_read_cost: Some(Cost(1.0)), - retention_cost_rate: Some(CostRate(0.1)), - retirement_cost: Some(Cost(1.0)), - } - } - - fn summary_maintenance_capabilities( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - SummaryMaintenanceCapabilities { - incremental_update: true, - merge: true, - delete: true, - } - } -} - -fn dashboard_workload() -> PlanningWorkload { - let query = Query("quantile_over_time(0.99, latency[5m])".into()); - let requirements = QueryRequirements { - accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Epsilon(0.01)), - ..QueryRequirements::default() - }; - PlanningWorkload { - query_workload: QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: Some(vec![BatchEntry { - query: query.clone(), - requirements: requirements.clone(), - predictability: Predictability::AdHoc, - invocations: 1, - execute_at: None, - time_selection: TimeSelection::default(), - }]), - repeating_queries: Some(vec![RepeatingEntry { - query, - demand: RepeatedDemand::FixedInterval(RepetitionInterval(1_000)), - requirements, - predictability: Predictability::Predictable { known_at: None }, - time_selection: TimeSelection { - scope: QueryTimeScope::RealTime, - ..TimeSelection::default() - }, - }]), - }, - data_workload: Some(DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - ingestion_rate: Evidence { - value: Some(Rate(1.0)), - source: EvidenceSource::Observed, - observed_at_ms: Some(NOW_MS), - valid_for_ms: Some(60_000), - }, - data_ingestion_interval: Evidence { - value: Some(DurationMs(1_000)), - ..Default::default() - }, - ..DataWorkload::default() - }), - } -} - -#[test] -fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() { - let workload = dashboard_workload(); - let plan = selected_plan(&workload); - - assert!(!plan.selected_raw_recompute); - assert_eq!(plan.expected_reads, Some(100.0)); - assert_eq!(plan.deployments.len(), 1); - - let deployment = &plan.deployments[0]; - let guarantee = deployment - .summary_maintenance_lifecycle_guarantee - .as_ref() - .expect("selected lifecycle guarantee"); - assert_eq!( - guarantee.summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::ContinuouslyMaintained - ); - assert_eq!(guarantee.evaluation_schedule, EvaluationSchedule::PerUpdate); - assert_eq!( - guarantee.summary_maintenance_mode, - SummaryMaintenanceMode::Incremental - ); - assert!(deployment.alternatives.iter().any(|alternative| { - matches!( - alternative.summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::Prepared { .. } - ) && alternative.rejection - == Some(SummaryMaintenanceLifecycleRejection::RequiresPredictableOneTimeQuery) - })); - assert!(deployment.alternatives.iter().any(|alternative| { - matches!( - alternative.summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::Shared { .. } - ) && alternative.rejection - == Some(SummaryMaintenanceLifecycleRejection::UnsupportedByRuntime) - })); - - let exported = serde_json::to_value(export_summary_maintenance_plan(&plan)).unwrap(); - assert_eq!( - exported["deployments"][0]["selected"]["lifecycle"]["kind"], - "continuously_maintained" - ); - assert_eq!( - exported["deployments"][0]["selected"]["maintenance_mode"], - "incremental" - ); - let alternatives = exported["deployments"][0]["alternatives"] - .as_array() - .expect("exported lifecycle alternatives"); - assert!(alternatives.iter().any(|alternative| { - alternative["lifecycle"]["kind"] == "prepared" - && alternative["rejection"] == "requires_predictable_one_time_query" - })); - assert!(alternatives.iter().any(|alternative| { - alternative["lifecycle"]["kind"] == "shared" - && alternative["rejection"] == "unsupported_by_runtime" - })); - assert!(exported["dag"]["nodes"].as_array().is_some()); - let summary_node = exported["dag"]["nodes"] - .as_array() - .unwrap() - .iter() - .find(|node| node["kind"] == "SummaryAgg") - .expect("exported SummaryAgg node"); - assert_eq!( - summary_node["detail"]["summary_maintenance"]["selected"]["lifecycle"]["kind"], - "continuously_maintained" - ); -} - -fn selected_plan( - workload: &PlanningWorkload, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { - selected_plan_with_model(workload, &FullyCostedRuntime) -} - -fn selected_plan_with_model( - workload: &PlanningWorkload, - model: &dyn CostModel, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { - selected_plan_with_horizon(workload, model, Horizon(100.)) -} - -fn selected_plan_with_horizon( - workload: &PlanningWorkload, - model: &dyn CostModel, - horizon: Horizon, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { - workload.validate().unwrap(); - - let lowered = lower_promql_workload(workload, 0) - .expect("valid PromQL workload") - .into_iter() - .next() - .expect("one normalized workload entry"); - selected_plan_for_lowered(workload, lowered, model, horizon) -} - -fn selected_plan_for_lowered( - workload: &PlanningWorkload, - lowered: asap_types::pre_asap::QueryExpr, - model: &dyn CostModel, - horizon: Horizon, -) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { - let root = Rc::new(lowered); - let strategies = asap_aware_mapping::default_strategies_with(model); - let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); - let target = Rc::clone(&space.roots[0].1); - let capabilities = SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }; - - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &[1], - }, - NOW_MS, - Some(horizon), - capabilities, - model, - ) - .unwrap(); - assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &target, - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(horizon), - capabilities, - model, - ) - .unwrap() - .expect("selected summary plan") -} - -mod physical_common; - -/// A selected continuous lifecycle supplies a materialization boundary; its -/// maintenance and query DAGs execute the selected KLL computation in fresh runs. -#[test] -fn continuous_lifecycle_compiles_and_executes_spatial_kll() { - use asap_physical_operators::{ - physical_planner::{compile_candidate, InputContract}, - runtime::Scope, - values::{Batch, Value}, - }; - use asap_types::{ - post_asap::{compile_post_asap_dag, FieldDataType, PostAsapOperatorPayload}, - pre_asap::DataType, - }; - use std::{collections::BTreeMap, sync::Arc}; - let mut workload = dashboard_workload(); - workload.query_workload.query_batch.as_mut().unwrap()[0].query = - Query("quantile(0.99, latency)".into()); - workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = - Query("quantile(0.99, latency)".into()); - let selected = selected_plan(&workload); - assert_eq!( - selected.deployments[0] - .summary_maintenance_lifecycle_guarantee - .as_ref() - .unwrap() - .summary_maintenance_lifecycle, - SummaryMaintenanceLifecycle::ContinuouslyMaintained - ); - let dag = compile_post_asap_dag(&selected.root).unwrap(); - let build = dag - .nodes - .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) - .unwrap(); - let input = dag - .edges - .iter() - .find(|edge| edge.consumer == build.id) - .unwrap() - .producer; - let raw = dag.nodes.iter().find(|node| node.id == input).unwrap(); - let schema = Arc::new(raw.output_schema.clone()); - let candidate = compile_candidate( - &dag, - BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], - &[u64::from(build.id.0)], - ) - .unwrap(); - - // A continuous input without a finite pane boundary cannot implement this - // blocking builder. Retain lifecycle ownership in the candidate payload; - // only the legal bounded request candidate reaches workload pricing. - let mut unbounded = InputContract::bounded(schema.clone()); - unbounded.properties.boundedness = asap_physical_operators::plan::Boundedness::Unbounded; - let rejected = compile_candidate( - &dag, - BTreeMap::from([(u64::from(input.0), unbounded)]), - &[u64::from(dag.root.0)], - &[u64::from(build.id.0)], - ); - assert!(rejected.is_err()); - let request = compile_candidate( - &dag, - BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], - &[], - ) - .unwrap(); - let mut priced = 0; - let feedback = asap_physical_operators::physical_planner::select_candidate( - vec![ - rejected.map(|candidate| { - ( - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - candidate, - ) - }), - Ok((SummaryMaintenanceLifecycle::Ephemeral, request)), - ], - |_| { - priced += 1; - Ok(Some( - asap_physical_operators::physical_planner::CandidateCost { - workload_scope: "dashboard".into(), - horizon_seconds: 100., - total_cost: 1000., - }, - )) - }, - ) - .unwrap(); - assert_eq!(priced, 1); - assert_eq!(feedback.candidate.0, SummaryMaintenanceLifecycle::Ephemeral); - for revision in [1, 2] { - let rows = (1..=100) - .map(|value| { - schema - .fields - .iter() - .map(|field| match field.dtype { - FieldDataType::Plain(DataType::Float64) => Value::Float64(f64::from(value)), - FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), - _ => panic!("unexpected field {field:?}"), - }) - .collect() - }) - .collect(); - let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); - let direct = physical_common::execute( - &feedback.candidate.1.query, - BTreeMap::from([(u64::from(input.0), raw_batch.clone())]), - Scope::Query { - evaluation_time_ms: 300_000, - revision, - }, - ); - let state = physical_common::execute( - candidate.precompute.as_ref().unwrap(), - BTreeMap::from([(u64::from(input.0), raw_batch)]), - Scope::Ingestion { - window_start_ms: 0, - window_end_ms: 300_000, - revision, - }, - ); - let result = physical_common::execute( - &candidate.query, - BTreeMap::from([(u64::from(build.id.0), state[0][0].clone())]), - Scope::Query { - evaluation_time_ms: 300_000, - revision, - }, - ); - let values: Vec<_> = result[0] - .iter() - .flat_map(|batch| batch.rows()) - .flat_map(|row| row.iter()) - .filter_map(|value| { - if let Value::Float64(value) = value { - Some(*value) - } else { - None - } - }) - .collect(); - let direct_values: Vec<_> = direct[0] - .iter() - .flat_map(|batch| batch.rows()) - .flat_map(|row| row.iter()) - .filter_map(|value| { - if let Value::Float64(value) = value { - Some(*value) - } else { - None - } - }) - .collect(); - assert_eq!( - values, direct_values, - "maintenance and request candidates preserve the same population" - ); - assert_eq!(values.len(), 1); - assert!( - (98. ..=100.).contains(&values[0]), - "p99 rank must reflect the supplied population" - ); - } -} - -fn quantile_workload(query: &str) -> PlanningWorkload { - let mut workload = dashboard_workload(); - workload.query_workload.query_batch.as_mut().unwrap()[0].query = Query(query.into()); - workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = Query(query.into()); - workload -} - -/// Timed DAG for `query` after binding every summary state to `lifecycle`. -/// Grouped queries carry a physical series identity, as per-entity state needs. -fn lifecycle_timed_dag( - query: &str, - lifecycle: &SummaryMaintenanceLifecycle, -) -> (asap_types::post_asap::PostAsapDAG, Vec) { - use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; - let workload = quantile_workload(query); - let mut lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); - if query.contains(" by(") { - lowered = - asap_physical_operators::physical_planner::promql_rows::with_series_identity(&lowered) - .unwrap(); - } - let root = - selected_plan_for_lowered(&workload, lowered, &FullyCostedRuntime, Horizon(100.)).root; - let candidates = enumerate_summary_maintenance_lifecycles( - root, - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &FullyCostedRuntime, - ) - .unwrap(); - let choices: Vec<_> = candidates - .deployments() - .iter() - .map(|deployment| (deployment.post_asap_node_id, lifecycle.clone())) - .collect(); - let mut states: Vec<_> = choices.iter().map(|(id, _)| u64::from(id.0)).collect(); - states.sort_unstable(); - let dag = candidates - .select(&choices) - .unwrap() - .execution_timed_dag() - .unwrap(); - (dag, states) -} - -/// Compile inputs for a timed DAG: its raw source, available at either phase. -fn raw_inputs( - dag: &asap_types::post_asap::PostAsapDAG, -) -> std::collections::BTreeMap { - let raw = dag - .nodes - .iter() - .find(|node| { - matches!( - node.payload, - asap_types::post_asap::PostAsapOperatorPayload::Fallback { .. } - ) - }) - .unwrap(); - std::collections::BTreeMap::from([( - u64::from(raw.id.0), - asap_physical_operators::physical_planner::InputContract::bounded(std::sync::Arc::new( - raw.output_schema.clone(), - )), - )]) -} - -/// For existing PromQL fixtures, Planner's own retained lifecycle selection -/// reproduces the timing that realization strategies assign today. -#[test] -fn planner_lifecycle_selection_reproduces_strategy_timing() { - for query in [ - "quantile_over_time(0.99, latency[5m])", - "quantile(0.99, latency)", - "sum by(job)(rate(m[1m]))", - ] { - let plan = selected_plan(&quantile_workload(query)); - assert!(!plan.selected_raw_recompute, "{query}"); - assert!(plan.deployments.iter().all(|deployment| { - deployment - .summary_maintenance_lifecycle_guarantee - .as_ref() - .is_some_and(|guarantee| { - guarantee.summary_maintenance_lifecycle - != SummaryMaintenanceLifecycle::Ephemeral - }) - })); - let strategy = asap_types::post_asap::compile_post_asap_dag(&plan.root).unwrap(); - assert_eq!(plan.execution_timed_dag().unwrap(), strategy, "{query}"); - } -} - -/// An explicitly chosen lifecycle reaches physical compilation through timing: -/// ContinuouslyMaintained puts the state in precompute, Ephemeral leaves -/// precompute empty and reads the raw source at query time; both answer alike. -#[test] -fn chosen_lifecycle_timing_decides_precompute_contents() { - use asap_physical_operators::{ - physical_planner::{compile_candidate, frontier_from_timing}, - runtime::Scope, - values::{Batch, Value}, - }; - use asap_types::{post_asap::FieldDataType, pre_asap::DataType}; - use std::collections::BTreeMap; - - let mut answers = Vec::new(); - for lifecycle in [ - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - SummaryMaintenanceLifecycle::Ephemeral, - ] { - let (dag, states) = lifecycle_timed_dag("quantile(0.99, latency)", &lifecycle); - let [state] = states[..] else { - panic!("one summary state"); - }; - let inputs = raw_inputs(&dag); - let (&raw_id, contract) = inputs.iter().next().unwrap(); - let schema = contract.schema.clone(); - let frontier = frontier_from_timing(&dag).unwrap(); - let candidate = - compile_candidate(&dag, inputs, &[u64::from(dag.root.0)], &frontier).unwrap(); - let rows = (1..=100) - .map(|value| { - schema - .fields - .iter() - .map(|field| match field.dtype { - FieldDataType::Plain(DataType::Float64) => Value::Float64(f64::from(value)), - FieldDataType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), - _ => panic!("unexpected field {field:?}"), - }) - .collect() - }) - .collect(); - let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); - let query_scope = Scope::Query { - evaluation_time_ms: 300_000, - revision: 1, - }; - let result = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { - assert!(frontier.is_empty()); - assert!(candidate.precompute.is_none()); - physical_common::execute( - &candidate.query, - BTreeMap::from([(raw_id, raw_batch)]), - query_scope, - ) - } else { - assert_eq!(frontier, [state]); - assert_eq!( - candidate - .materialized_outputs - .keys() - .copied() - .collect::>(), - [state] - ); - let stored = physical_common::execute( - candidate.precompute.as_ref().unwrap(), - BTreeMap::from([(raw_id, raw_batch)]), - Scope::Ingestion { - window_start_ms: 0, - window_end_ms: 300_000, - revision: 1, - }, - ); - physical_common::execute( - &candidate.query, - BTreeMap::from([(state, stored[0][0].clone())]), - query_scope, - ) - }; - answers.push( - result[0] - .iter() - .flat_map(|batch| batch.rows()) - .flat_map(|row| row.iter()) - .filter_map(|value| match value { - Value::Float64(value) => Some(*value), - _ => None, - }) - .collect::>(), - ); - } - assert_eq!(answers[0], answers[1]); - assert_eq!(answers[0].len(), 1); -} - -/// One compilation, cut by each lifecycle assignment's timing, yields exactly -/// the candidate `compile_candidate` builds for that timed DAG: the retained -/// state is the frontier under ContinuouslyMaintained, and nothing under -/// Ephemeral. Covers the KLL quantile fixture and grouped Rate→Sum. -#[test] -fn lifecycle_timing_cuts_one_compilation() { - use asap_physical_operators::physical_planner::{ - compile, compile_candidate, cut_candidate, frontier_from_timing, - }; - for query in ["quantile(0.99, latency)", "sum by(job)(rate(m[1m]))"] { - let ephemeral = SummaryMaintenanceLifecycle::Ephemeral; - let (compiled_dag, _) = lifecycle_timed_dag(query, &ephemeral); - let inputs = raw_inputs(&compiled_dag); - let roots = [u64::from(compiled_dag.root.0)]; - let compiled = compile(&compiled_dag, inputs.clone(), &roots).unwrap(); - for lifecycle in [ - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - ephemeral, - ] { - let (dag, states) = lifecycle_timed_dag(query, &lifecycle); - let frontier = frontier_from_timing(&dag).unwrap(); - // Retained states read by a query-time consumer, or the root itself. - let query_time = |id: u64| { - dag.nodes.iter().any(|node| { - u64::from(node.id.0) == id - && node.output_state.timing - == asap_types::post_asap::ExecutionTiming::QueryTime - }) - }; - let expected_frontier = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { - vec![] - } else { - states - .iter() - .copied() - .filter(|state| { - *state == u64::from(dag.root.0) - || dag.edges.iter().any(|edge| { - u64::from(edge.producer.0) == *state - && query_time(u64::from(edge.consumer.0)) - }) - }) - .collect() - }; - assert_eq!(frontier, expected_frontier, "{query} {lifecycle:?}"); - let cut = cut_candidate(&compiled, &frontier).unwrap(); - let expected = compile_candidate(&dag, inputs.clone(), &roots, &frontier).unwrap(); - assert_eq!( - serde_json::to_vec(&cut).unwrap(), - serde_json::to_vec(&expected).unwrap(), - "{query} {lifecycle:?}" - ); - } - } -} - -/// A maintained current-series population is placed by its lifecycle choice: -/// ContinuouslyMaintained stores the population in precompute, Ephemeral -/// rebuilds it from the raw source at query time; both rank alike. -#[test] -fn chosen_population_lifecycle_decides_precompute_contents() { - use asap_aware_mapping::{ - enumerate_summary_maintenance_lifecycles, - maintained_population::MaintainedPopulationStrategy, - }; - use asap_physical_operators::{ - physical_planner::{ - compile_candidate, - promql_rows::{series_row, with_series_identity}, - InputContract, - }, - runtime::Scope, - values::{Batch, Value}, - }; - use asap_types::post_asap::{ - maintained_population::PopulationInput, PostAsapOperatorPayload, ValueOperation, - }; - use std::{collections::BTreeMap, sync::Arc}; - - let workload = quantile_workload("topk by(job)(1, m)"); - let root = Rc::new( - with_series_identity(&lower_promql_workload(&workload, 0).unwrap().remove(0)).unwrap(), - ); - let root = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) - .candidate(&root) - .unwrap(); - let mut answers = Vec::new(); - for lifecycle in [ - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - SummaryMaintenanceLifecycle::Ephemeral, - ] { - let candidates = enumerate_summary_maintenance_lifecycles( - Rc::clone(&root), - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &FullyCostedRuntime, - ) - .unwrap(); - let [deployment] = candidates.deployments() else { - panic!("one population state"); - }; - let id = deployment.post_asap_node_id; - let dag = candidates - .select(&[(id, lifecycle.clone())]) - .unwrap() - .execution_timed_dag() - .unwrap(); - let population = dag.nodes.iter().find(|node| node.id == id).unwrap(); - let PostAsapOperatorPayload::Value { - operation: ValueOperation::MaintainPopulation { population }, - } = &population.payload - else { - panic!("the deployment is the maintained population"); - }; - let PopulationInput::CurrentSeries(spec) = &population.input else { - panic!("current-series population"); - }; - let lookback = i64::try_from(spec.lookback_ms).unwrap(); - let raw = dag - .nodes - .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) - .unwrap(); - let (raw_id, schema) = (u64::from(raw.id.0), Arc::new(raw.output_schema.clone())); - let frontier = - asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); - let candidate = compile_candidate( - &dag, - BTreeMap::from([(raw_id, InputContract::bounded(schema.clone()))]), - &[u64::from(dag.root.0)], - &frontier, - ) - .unwrap(); - let end = 60_000; - let rows = [("a", end - 1, 100.), ("a", end, 1.), ("b", end, 20.)] - .into_iter() - .map(|(instance, at, value)| { - series_row( - &schema, - &BTreeMap::from([ - ("job".into(), "api".into()), - ("instance".into(), instance.into()), - ]), - at, - value, - ) - .unwrap() - }) - .collect(); - let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); - let query_scope = Scope::Query { - evaluation_time_ms: end, - revision: 1, - }; - let result = if lifecycle == SummaryMaintenanceLifecycle::Ephemeral { - assert!(frontier.is_empty()); - assert!(candidate.precompute.is_none()); - physical_common::execute( - &candidate.query, - BTreeMap::from([(raw_id, raw_batch)]), - query_scope, - ) - } else { - let state = u64::from(id.0); - assert_eq!(frontier, [state]); - let stored = physical_common::execute( - candidate.precompute.as_ref().unwrap(), - BTreeMap::from([(raw_id, raw_batch)]), - Scope::Ingestion { - window_start_ms: end - lookback, - window_end_ms: end, - revision: 1, - }, - ); - physical_common::execute( - &candidate.query, - BTreeMap::from([(state, stored[0][0].clone())]), - query_scope, - ) - }; - answers.push( - result[0] - .iter() - .flat_map(|batch| batch.rows()) - .flat_map(|row| row.iter()) - .filter_map(|value| match value { - Value::Float64(value) => Some(*value), - _ => None, - }) - .collect::>(), - ); - } - assert_eq!(answers[0], answers[1]); - assert_eq!(answers[0], [20.]); -} - -/// Grouped Rate→Sum is one inventory candidate: retaining the Sum state puts -/// Rate and Sum in precompute, while an `Ephemeral` Sum over a retained Rate -/// state leaves Sum in the query DAG. -#[test] -fn grouped_rate_sum_placement_is_a_lifecycle_choice() { - use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; - use asap_physical_operators::physical_planner::{compile_candidate, InputContract}; - use asap_types::post_asap::{ExactKind, FieldDataType, PostAsapOperatorPayload, SummaryExpr}; - use std::{collections::BTreeMap, sync::Arc}; - - let workload = quantile_workload("sum by(job)(rate(m[1m]))"); - let root = Rc::new( - asap_physical_operators::physical_planner::promql_rows::with_series_identity( - &lower_promql_workload(&workload, 0).unwrap().remove(0), - ) - .unwrap(), - ); - let is_exact = |node: &SummaryNode, kind: ExactKind| { - matches!(&node.expr, SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(k, _), .. - } if *k == kind) - }; - let inventory = asap_aware_mapping::search_workload(vec![("q", root)]) - .enumerate_candidate_dags(4096) - .unwrap(); - let candidates = inventory - .candidates - .into_iter() - .map(|mut forest| forest.remove(0).1) - .filter(|candidate| { - matches!(&candidate.expr, SummaryExpr::ValueOperation { child, .. } - if is_exact(child, ExactKind::Sum)) - }) - .collect::>(); - let [candidate] = candidates.as_slice() else { - panic!("one grouped Sum candidate, got {}", candidates.len()); - }; - let mut placements = Vec::new(); - for sum_lifecycle in [ - SummaryMaintenanceLifecycle::ContinuouslyMaintained, - SummaryMaintenanceLifecycle::Ephemeral, - ] { - let lifecycles = enumerate_summary_maintenance_lifecycles( - Rc::clone(candidate), - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.)), - SummaryMaintenanceLifecycleCapabilities::ALL, - &FullyCostedRuntime, - ) - .unwrap(); - let choices = lifecycles - .deployments() - .iter() - .map(|deployment| { - let lifecycle = if is_exact(&deployment.summary, ExactKind::Sum) { - sum_lifecycle.clone() - } else { - SummaryMaintenanceLifecycle::ContinuouslyMaintained - }; - (deployment.post_asap_node_id, lifecycle) - }) - .collect::>(); - assert_eq!(choices.len(), 2, "Rate and Sum states"); - let dag = lifecycles - .select(&choices) - .unwrap() - .execution_timed_dag() - .unwrap(); - let raw = dag - .nodes - .iter() - .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) - .unwrap(); - let frontier = - asap_physical_operators::physical_planner::frontier_from_timing(&dag).unwrap(); - let [boundary] = frontier.as_slice() else { - panic!("one precompute output, got {frontier:?}"); - }; - let boundary = dag - .nodes - .iter() - .find(|node| u64::from(node.id.0) == *boundary) - .unwrap(); - let physical = compile_candidate( - &dag, - BTreeMap::from([( - u64::from(raw.id.0), - InputContract::bounded(Arc::new(raw.output_schema.clone())), - )]), - &[u64::from(dag.root.0)], - &frontier, - ) - .unwrap(); - let json = |value| String::from_utf8(serde_json::to_vec(value).unwrap()).unwrap(); - placements.push(( - boundary.payload.clone(), - json(physical.precompute.as_ref().unwrap()), - json(&physical.query), - )); - } - let builds = |json: &str, kind: &str| { - json.contains(&format!( - r#"{{"SummaryBuild":{{"family":{{"ExactAggregate":["{kind}","{kind}"]}}"# - )) - }; - let [(retained, retained_pre, retained_query), (ephemeral, ephemeral_pre, ephemeral_query)] = - placements.as_slice() - else { - unreachable!() - }; - let state = |payload: &PostAsapOperatorPayload, kind: ExactKind| { - matches!(payload, PostAsapOperatorPayload::SummaryAgg { - family: FieldDataType::ExactAggregate(k, _), .. - } if *k == kind) - }; - assert!(state(retained, ExactKind::Sum)); - assert!(builds(retained_pre, "Rate") && builds(retained_pre, "Sum")); - assert!(!retained_query.contains("SummaryBuild")); - assert!(state(ephemeral, ExactKind::Rate)); - assert!(builds(ephemeral_pre, "Rate") && !builds(ephemeral_pre, "Sum")); - assert!(builds(ephemeral_query, "Sum")); -} - -/// The lifecycle-timed DAG Planner selects for `query` with upfront series -/// typing, and whether it keeps an ingestion-time Binary. -fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDAG, bool) { - use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; - let workload = quantile_workload(query); - let lowered = asap_types::pre_asap::schema::with_promql_series_identity( - &lower_promql_workload(&workload, 0).unwrap().remove(0), - ) - .unwrap(); - let dag = selected_plan_for_lowered(&workload, lowered, &FullyCostedRuntime, Horizon(100.)) - .execution_timed_dag() - .unwrap(); - let ingestion_binary = dag.nodes.iter().any(|node| { - matches!(node.payload, PostAsapOperatorPayload::Binary { .. }) - && node.output_state.timing == ExecutionTiming::IngestionTime - }); - (dag, ingestion_binary) -} - -/// Execute a timed DAG's precompute and query DAGs over `samples` -/// (`(metric, job, seconds, value)`) at 300s; returns the root's values. -fn execute_timed( - dag: &asap_types::post_asap::PostAsapDAG, - samples: &[(&str, &str, i64, f64)], -) -> Vec { - use asap_physical_operators::{ - physical_planner::{ - compile_candidate, frontier_from_timing, promql_fallback, promql_rows, InputContract, - }, - runtime::Scope, - values::{Batch, Value}, - }; - use asap_types::{ - post_asap::PostAsapOperatorPayload, - pre_asap::{QueryExpr, Source}, - }; - use std::{collections::BTreeMap, sync::Arc}; - // Raw inputs: a selector Fallback is itself the input; a retained - // expression reads each of its selectors through its raw-series slots. - let mut raw = BTreeMap::new(); - for node in &dag.nodes { - let PostAsapOperatorPayload::Fallback { expression } = &node.payload else { - continue; - }; - let metric = |selector: &QueryExpr| match selector { - QueryExpr::TimeRange { child, .. } => match child.as_ref() { - QueryExpr::Scan { - source: Source::TimeSeries { metric }, - .. - } => Some(metric.clone()), - _ => None, - }, - QueryExpr::Scan { - source: Source::TimeSeries { metric }, - .. - } => Some(metric.clone()), - _ => None, - }; - if let Some(name) = metric(expression) { - raw.insert( - u64::from(node.id.0), - (Arc::new(node.output_schema.clone()), name), - ); - } else { - for (i, (selector, schema)) in promql_fallback::raw_series(expression) - .unwrap() - .into_iter() - .enumerate() - { - raw.insert( - promql_fallback::raw_series_input(u64::from(node.id.0), i), - (schema, metric(&selector).unwrap()), - ); - } - } - } - let batch = |schema: &asap_physical_operators::values::SchemaRef, name: &str| { - let rows = samples - .iter() - .filter(|sample| sample.0 == name) - .map(|(metric, job, seconds, value)| { - let labels = BTreeMap::from([ - ("__name__".to_string(), metric.to_string()), - ("job".to_string(), job.to_string()), - ]); - promql_rows::series_row(schema, &labels, seconds * 1000, *value).unwrap() - }) - .collect(); - Batch::try_new(schema.clone(), rows).unwrap() - }; - let frontier = frontier_from_timing(dag).unwrap(); - let candidate = compile_candidate( - dag, - raw.iter() - .map(|(id, (schema, _))| (*id, InputContract::bounded(schema.clone()))) - .collect(), - &[u64::from(dag.root.0)], - &frontier, - ) - .unwrap(); - let raw_sources = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDAG| { - plan.input_contracts() - .filter_map(|(id, _)| raw.get(&id).map(|(schema, name)| (id, batch(schema, name)))) - .collect::>() - }; - let mut query_sources = raw_sources(&candidate.query); - if let Some(precompute) = &candidate.precompute { - let stored = physical_common::execute( - precompute, - raw_sources(precompute), - Scope::Ingestion { - window_start_ms: 240_000, - window_end_ms: 300_000, - revision: 1, - }, - ); - for (root, batches) in precompute.roots().iter().zip(stored) { - query_sources.insert(*root, batches[0].clone()); - } - } - let result = physical_common::execute( - &candidate.query, - query_sources, - Scope::Query { - evaluation_time_ms: 300_000, - revision: 1, - }, - ); - result[0] - .iter() - .flat_map(|batch| batch.rows()) - .flat_map(|row| row.iter()) - .filter_map(|value| match value { - Value::Float64(value) => Some(*value), - _ => None, - }) - .collect() -} - -/// Prometheus drops series without a match: arithmetic over different -/// selectors keeps only label sets present on both sides (none when disjoint), -/// and such arithmetic never becomes aligned maintenance. -#[test] -fn maintained_arithmetic_over_different_selectors_matches_prometheus() { - let query = "sum(sum_over_time(m[1m]) + sum_over_time(n[1m]))"; - let (dag, ingestion_binary) = typed_selection(query); - let disjoint = [ - ("m", "a", 250, 1.0), - ("m", "a", 290, 2.0), - ("n", "b", 250, 5.0), - ]; - let values = execute_timed(&dag, &disjoint); - assert!(values.is_empty(), "{values:?}"); - // Only job a is on both sides: m_a + n_a = (1 + 2) + 7; m{job="b"} is dropped. - let overlapping = [ - ("m", "a", 250, 1.0), - ("m", "a", 290, 2.0), - ("m", "b", 250, 5.0), - ("n", "a", 250, 7.0), - ]; - assert_eq!(execute_timed(&dag, &overlapping), [10.0]); - assert!(!ingestion_binary); - // The quantile's exact fallback runs outside Planner; it must not be maintained either. - let (_, ingestion_binary) = - typed_selection("quantile(0.9, sum_over_time(m[1m]) + sum_over_time(n[1m]))"); - assert!(!ingestion_binary); -} - -/// Arithmetic over one selector keeps its maintained layout and adds each -/// series' two readouts before the quantile. -#[test] -fn maintained_arithmetic_over_one_selector_executes() { - let (dag, ingestion_binary) = - typed_selection("quantile(0.9, sum_over_time(m[1m]) + sum_over_time(m[1m]))"); - assert!(ingestion_binary, "one selector shares its key set"); - let values = execute_timed( - &dag, - &[ - ("m", "a", 250, 1.0), - ("m", "a", 290, 2.0), - ("m", "b", 250, 5.0), - ], - ); - // job a: 3 + 3 = 6; job b: 5 + 5 = 10 (mispairing a with b gives 8 and 8). - // KLL at epsilon 0.01 returns an input value within 0.01 of rank 0.9; of - // two values only the larger is. - assert_eq!(values, [10.0]); -} diff --git a/crates/integration-tests/tests/time_range.rs b/crates/integration-tests/tests/time_range.rs index d3ab732fe..eb7955d9a 100644 --- a/crates/integration-tests/tests/time_range.rs +++ b/crates/integration-tests/tests/time_range.rs @@ -1,46 +1,53 @@ -//! `QueryExpr::TimeRange` — range / streaming function tests. +//! `NonASAPOp::TimeRange` — range / streaming function tests. //! -//! All range functions lower to `Aggregate { child: TimeRange { range, child: Scan } }`. +//! All range functions lower to `Aggregate { child: TimeRange { range, kind: Range, child: Scan } }`. //! The temporal range lives on the `TimeRange` node, not in the `AggIntent`. //! `rate` / `increase` use `AggIntent::Rate` / `AggIntent::Increase` (no window field). //! `*_over_time` functions reuse the corresponding cross-series intents -//! (`Count`, `Sum`, `Quantile`, …) — the `TimeRange` child is what marks them -//! as per-series reductions. +//! (`Count`, `Sum`, `Quantile`, …) — the `Range` selector child is what marks +//! them as per-series reductions. use std::rc::Rc; use std::time::Duration; use asap_integration_tests::fixtures::lower_promql; use asap_integration_tests::fixtures::metric_schema; -use asap_types::pre_asap::{AggIntent, QueryExpr, Reduction, Source}; +use asap_types::ir::operator::{AggIntent, Reduction, Source}; +use asap_types::ir::{NonASAPOp, OperatorNode, TimeRangeKind}; use asap_types::types::AccuracyTarget; -fn lower(q: &str) -> QueryExpr { +fn lower(q: &str) -> Rc { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) } -fn scan(metric: &str) -> QueryExpr { - QueryExpr::Scan { +fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(op)) + .expect("fixture node derives its schema") +} + +fn scan(metric: &str) -> Rc { + node(NonASAPOp::Scan { source: Source::TimeSeries { metric: metric.into(), }, predicates: vec![], schema: metric_schema(&[]), - } + }) } -fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> QueryExpr { - QueryExpr::Aggregate { +fn range_agg(range_secs: u64, intent: AggIntent, metric: &str) -> Rc { + node(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![intent], output_names: vec!["".into()], filters: vec![], having: None, - child: Rc::new(QueryExpr::TimeRange { + child: node(NonASAPOp::TimeRange { range: Duration::from_secs(range_secs), - child: Rc::new(scan(metric)), + kind: TimeRangeKind::Range, + child: scan(metric), }), - } + }) } // #13 — rate: counter-reset-aware per-second rate; range on TimeRange node diff --git a/crates/logical-optimizer/Cargo.toml b/crates/logical-optimizer/Cargo.toml new file mode 100644 index 000000000..06ff43774 --- /dev/null +++ b/crates/logical-optimizer/Cargo.toml @@ -0,0 +1,18 @@ +[package] +name = "asap-logical-optimizer" +version = "0.1.0" +edition = "2021" + +# #509 Stage 1: logical candidate generation (Pass 1 rewrites and summary +# realization, Pass 2 ASAP-aware CSE) and the analytical accuracy model. +# Depends only on asap-types (the PromQL front end is a test-only +# dev-dependency); never on the cost model or a later stage. +# tests/stage1_cost_independence.rs checks this manifest. +[dependencies] +asap_sketchlib = { workspace = true } +asap-types = { path = "../types" } +thiserror = "2" +serde_json = "1" + +[dev-dependencies] +asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-aware-mapping/src/accuracy/allocation.rs b/crates/logical-optimizer/src/accuracy/allocation.rs similarity index 100% rename from crates/asap-aware-mapping/src/accuracy/allocation.rs rename to crates/logical-optimizer/src/accuracy/allocation.rs diff --git a/crates/asap-aware-mapping/src/accuracy/composition.rs b/crates/logical-optimizer/src/accuracy/composition.rs similarity index 99% rename from crates/asap-aware-mapping/src/accuracy/composition.rs rename to crates/logical-optimizer/src/accuracy/composition.rs index f42e8f158..17efd6572 100644 --- a/crates/asap-aware-mapping/src/accuracy/composition.rs +++ b/crates/logical-optimizer/src/accuracy/composition.rs @@ -448,11 +448,11 @@ fn composed_provenance( } pub(super) fn exact_operation_rule(operation: &ExactOperation) -> Option { - let ExactOperation::Aggregate { measures, .. } = operation else { - return None; - }; + let ExactOperation::Aggregate { measures, .. } = operation; match measures.as_slice() { - [intent] => crate::function_rules::function_rules(intent).map(|rules| rules.accuracy), + [intent] => { + crate::pass1::function_rules::function_rules(intent).map(|rules| rules.accuracy) + } // The remaining functions are exact over exact samples, but have // no definition-backed rule over approximate values yet. _ => None, @@ -707,7 +707,7 @@ pub(super) fn propagate( #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::AggIntent; + use asap_types::ir::operator::AggIntent; fn abs(bound: f64, delta: f64) -> ResultGuarantee { ResultGuarantee { metric: ErrorMetric::AbsoluteValue, @@ -1032,11 +1032,11 @@ mod tests { #[test] fn counter_functions_have_distinct_definition_rules() { let operation = |intent| ExactOperation::Aggregate { - reduction: asap_types::pre_asap::Reduction::PerEntity, + reduction: asap_types::ir::operator::Reduction::PerEntity, measures: vec![intent], output_names: vec![], - having: None, filters: vec![], + having: None, }; assert_eq!( DefaultAccuracyModel.exact_operation_rule(&operation(AggIntent::Rate)), diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/cardinality.rs b/crates/logical-optimizer/src/accuracy/estimators/cardinality.rs similarity index 100% rename from crates/asap-aware-mapping/src/accuracy/estimators/cardinality.rs rename to crates/logical-optimizer/src/accuracy/estimators/cardinality.rs diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs b/crates/logical-optimizer/src/accuracy/estimators/cms.rs similarity index 88% rename from crates/asap-aware-mapping/src/accuracy/estimators/cms.rs rename to crates/logical-optimizer/src/accuracy/estimators/cms.rs index 46559358d..131dd1dbe 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/cms.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/cms.rs @@ -41,9 +41,9 @@ mod tests { #[test] fn local_guarantee_inverts_frequency_sizing() { - use crate::replacement::default_size_params; - use asap_types::post_asap::{GroupingStrategy, SketchKind}; - let c = asap_types::pre_asap::agg_intent::default_cardinality(); + use crate::pass1::replacement::default_size_params; + use asap_types::ir::schema::{GroupingStrategy, SketchKind}; + let c = asap_types::ir::operator::agg_intent::default_cardinality(); let params = default_size_params(SketchAlgorithm::Cms, &c, 0.01, 0.001); let g = DefaultAccuracyModel .local_guarantee( @@ -65,8 +65,8 @@ mod tests { } #[test] - fn heap_readout_retains_frequency_metric() { - use asap_types::post_asap::{GroupingStrategy, SketchKind}; + fn heap_evaluation_retains_frequency_metric() { + use asap_types::ir::schema::{GroupingStrategy, SketchKind}; let cms_heap = SketchParams::CmsWithHeap { width: 272, depth: 5, diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs b/crates/logical-optimizer/src/accuracy/estimators/count_sketch.rs similarity index 90% rename from crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs rename to crates/logical-optimizer/src/accuracy/estimators/count_sketch.rs index c95059637..9e9fc66c0 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/count_sketch.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/count_sketch.rs @@ -55,9 +55,9 @@ mod tests { #[test] fn count_sketch_uses_an_l2_guarantee() { - use crate::replacement::default_size_params; - use asap_types::post_asap::{GroupingStrategy, SketchKind}; - use asap_types::pre_asap::agg_intent::default_cardinality; + use crate::pass1::replacement::default_size_params; + use asap_types::ir::operator::agg_intent::default_cardinality; + use asap_types::ir::schema::{GroupingStrategy, SketchKind}; let intent = default_cardinality(); let count_sketch = default_size_params(SketchAlgorithm::CountSketch, &intent, 0.01, 0.01); let guarantee = DefaultAccuracyModel @@ -67,7 +67,7 @@ mod tests { GroupingStrategy::default(), ), &SketchStatistic::PointCount { - key: asap_types::pre_asap::expr_ir::ColumnRef::SampleValue, + key: asap_types::ir::scalar::ColumnRef::SampleValue, value: None, }, ) diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/ddsketch.rs b/crates/logical-optimizer/src/accuracy/estimators/ddsketch.rs similarity index 100% rename from crates/asap-aware-mapping/src/accuracy/estimators/ddsketch.rs rename to crates/logical-optimizer/src/accuracy/estimators/ddsketch.rs diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs b/crates/logical-optimizer/src/accuracy/estimators/hll.rs similarity index 95% rename from crates/asap-aware-mapping/src/accuracy/estimators/hll.rs rename to crates/logical-optimizer/src/accuracy/estimators/hll.rs index 4431a4118..c7c26b435 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/hll.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/hll.rs @@ -1,7 +1,7 @@ //! Estimator-specific confidence for classic HLL's linear-counting branch. //! //! This is conditional on independent uniform bucket hashes and an enforced -//! upper bound on distinct items in the complete readout population (including +//! upper bound on distinct items in the complete evaluation population (including //! all merged panes). It is not an RSE-to-normal conversion or an ERP fit. use super::*; @@ -56,13 +56,13 @@ impl ClassicHllConfidence { value: self.relative_error, }, failure_probability: ProbabilityExpr::Constant { value: delta }, - provenance: vec![GuaranteeSource::SketchReadout { + provenance: vec![GuaranteeSource::SketchEvaluation { algorithm: "Hll".into(), contract: "classic_hll_linear_counting_collision_bound_v1".into(), params: serde_json::json!({"precision": precision, "max_distinct": self.max_distinct, "relative_error": self.relative_error, "hash_assumption": "independent_uniform_buckets", - "population_scope": "complete_readout_including_merged_panes"}), + "population_scope": "complete_evaluation_including_merged_panes"}), query: "Cardinality".into(), }], }) @@ -187,7 +187,7 @@ mod tests { } } } - /// The model's readout formula matches the actual classic estimator after merge. + /// The model's evaluation formula matches the actual classic estimator after merge. #[test] fn native_classic_estimator_and_merged_registers_use_the_same_contract() { use asap_sketchlib::sketches::hll::{Classic, HyperLogLogP16}; @@ -230,9 +230,9 @@ mod tests { } #[test] fn generic_rse_sizing_does_not_certify_confidence() { - use crate::replacement::default_size_params; - use asap_types::post_asap::{GroupingStrategy, SketchKind}; - use asap_types::pre_asap::agg_intent::default_cardinality; + use crate::pass1::replacement::default_size_params; + use asap_types::ir::operator::agg_intent::default_cardinality; + use asap_types::ir::schema::{GroupingStrategy, SketchKind}; let c = default_cardinality(); let params = default_size_params(SketchAlgorithm::Hll, &c, 0.01, 0.01); let g = DefaultAccuracyModel diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs b/crates/logical-optimizer/src/accuracy/estimators/kll.rs similarity index 90% rename from crates/asap-aware-mapping/src/accuracy/estimators/kll.rs rename to crates/logical-optimizer/src/accuracy/estimators/kll.rs index 4cfe6f073..efa3feaa0 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/kll.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/kll.rs @@ -42,9 +42,9 @@ mod tests { #[test] fn local_guarantee_inverts_rank_sizing() { - use crate::replacement::default_size_params; - use asap_types::post_asap::{GroupingStrategy, SketchKind}; - use asap_types::pre_asap::agg_intent::default_quantile; + use crate::pass1::replacement::default_size_params; + use asap_types::ir::operator::agg_intent::default_quantile; + use asap_types::ir::schema::{GroupingStrategy, SketchKind}; let q = default_quantile(0.99); let params = default_size_params(SketchAlgorithm::Kll, &q, 0.01, 0.01); let g = DefaultAccuracyModel @@ -69,7 +69,7 @@ mod tests { assert_eq!(g.approximate_layer_count(), 1); assert!(g.provenance.iter().any(|source| matches!( source, - GuaranteeSource::SketchReadout { contract, .. } + GuaranteeSource::SketchEvaluation { contract, .. } if contract == "apache_datasketches_kll_empirical_99_a9b42755072b" ))); } diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs b/crates/logical-optimizer/src/accuracy/estimators/mod.rs similarity index 95% rename from crates/asap-aware-mapping/src/accuracy/estimators/mod.rs rename to crates/logical-optimizer/src/accuracy/estimators/mod.rs index 4eabf40be..6b094e30a 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/mod.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/mod.rs @@ -1,7 +1,7 @@ //! Dispatch committed estimator parameters to their accuracy models. use super::*; -use asap_types::post_asap::GroupingStrategy; -use asap_types::pre_asap::AggIntent; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::schema::GroupingStrategy; pub mod cardinality; pub mod cms; @@ -46,7 +46,7 @@ fn bounded_guarantee( metric, bound: BoundExpr::Constant { value: bound }, failure_probability: delta, - provenance: vec![GuaranteeSource::SketchReadout { + provenance: vec![GuaranteeSource::SketchEvaluation { algorithm: format!("{algorithm:?}"), contract: contract.into(), params: serde_json::to_value(params).unwrap_or(serde_json::Value::Null), @@ -157,7 +157,7 @@ impl<'a> EstimatorAccuracy<'a> { target: Option<&AccuracyTarget>, ) -> Self { let (epsilon, delta) = target - .map(crate::replacement::accuracy_budget) + .map(crate::pass1::replacement::accuracy_budget) .unwrap_or((0.0, 0.0)); Self { base, @@ -169,9 +169,9 @@ impl<'a> EstimatorAccuracy<'a> { fn hll(&self) -> Option { let EstimatorContract::ClassicHll { - max_distinct_per_readout, + max_distinct_per_evaluation, } = self.contract?; - hll::ClassicHllConfidence::new(max_distinct_per_readout, self.epsilon) + hll::ClassicHllConfidence::new(max_distinct_per_evaluation, self.epsilon) } pub(crate) fn size_params(&self, algorithm: &SketchAlgorithm) -> Option { diff --git a/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs b/crates/logical-optimizer/src/accuracy/estimators/univmon.rs similarity index 97% rename from crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs rename to crates/logical-optimizer/src/accuracy/estimators/univmon.rs index 297cf53d3..6a89a105e 100644 --- a/crates/asap-aware-mapping/src/accuracy/estimators/univmon.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/univmon.rs @@ -1,4 +1,4 @@ -//! UnivMon currently certifies only its exact unit-update total readout. +//! UnivMon currently certifies only its exact unit-update total evaluation. use super::*; pub(super) fn guarantee(query: &SketchStatistic) -> Option { diff --git a/crates/asap-aware-mapping/src/accuracy/evidence.rs b/crates/logical-optimizer/src/accuracy/evidence.rs similarity index 89% rename from crates/asap-aware-mapping/src/accuracy/evidence.rs rename to crates/logical-optimizer/src/accuracy/evidence.rs index 6391c746e..7ba650674 100644 --- a/crates/asap-aware-mapping/src/accuracy/evidence.rs +++ b/crates/logical-optimizer/src/accuracy/evidence.rs @@ -2,12 +2,12 @@ use super::*; /// A trusted source assertion scoped by `AccuracyEvidenceProvider` to one -/// complete readout. Choosing this variant asserts the estimator and hash +/// complete evaluation. Choosing this variant asserts the estimator and hash /// assumptions; it must not be inferred from sampled population statistics. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum EstimatorContract { /// Classic HLL with independent uniform bucket hashing, including merged panes. - ClassicHll { max_distinct_per_readout: u32 }, + ClassicHll { max_distinct_per_evaluation: u32 }, } /// An enforced domain for every sample of a direct quantile operand, in every @@ -18,7 +18,7 @@ pub enum EstimatorContract { pub struct QuantileInputDomain { pub lower: f64, pub upper: f64, - /// Upper bound on samples per evaluation, matching the pinned readout's + /// Upper bound on samples per evaluation, matching the pinned evaluation's /// exact Float64 rank limit. The population must also be nonempty. pub max_samples: u64, pub contract: String, @@ -87,31 +87,22 @@ pub struct PropagationStats { /// Supplies typed planning-time evidence required by propagation rules. pub trait AccuracyEvidenceProvider { /// Trusted estimator contract for this complete aggregate expression, - /// including source, filters, grouping and all panes in each readout. + /// including source, filters, grouping and all panes in each evaluation. /// An observed cardinality is not an enforced population bound. - fn estimator_contract( - &self, - _expression: &asap_types::pre_asap::QueryExpr, - ) -> Option { + fn estimator_contract(&self, _expression: &OperatorNode) -> Option { None } /// Enforced upper bound on distinct (partition, item) identities across a - /// complete TopK readout. Used to union-bound score errors for adaptively + /// complete TopK evaluation. Used to union-bound score errors for adaptively /// selected candidates. Observed cardinality is not sufficient evidence. - fn topk_max_distinct_items( - &self, - _expression: &asap_types::pre_asap::QueryExpr, - ) -> Option { + fn topk_max_distinct_items(&self, _expression: &OperatorNode) -> Option { None } /// Proof scoped to this complete quantile expression, including its source, /// filters, grouping and window. `None` means unknown, including emptiness. - fn quantile_input_domain( - &self, - _operand: &asap_types::pre_asap::query_expr::QueryExpr, - ) -> Option { + fn quantile_input_domain(&self, _operand: &OperatorNode) -> Option { None } @@ -181,8 +172,8 @@ mod tests { let fresh = provider.propagation_stats( &CompositionOperator::ExactSum, &FieldDataType::ExactAggregate( - asap_types::post_asap::ExactKind::Sum, - asap_types::post_asap::ExactParams::Sum, + asap_types::ir::schema::ExactKind::Sum, + asap_types::ir::schema::ExactParams::Sum, ), None, ); @@ -196,8 +187,8 @@ mod tests { .propagation_stats( &CompositionOperator::ExactSum, &FieldDataType::ExactAggregate( - asap_types::post_asap::ExactKind::Sum, - asap_types::post_asap::ExactParams::Sum, + asap_types::ir::schema::ExactKind::Sum, + asap_types::ir::schema::ExactParams::Sum, ), None, ); diff --git a/crates/asap-aware-mapping/src/accuracy/mod.rs b/crates/logical-optimizer/src/accuracy/mod.rs similarity index 91% rename from crates/asap-aware-mapping/src/accuracy/mod.rs rename to crates/logical-optimizer/src/accuracy/mod.rs index 9026de468..08ec27471 100644 --- a/crates/asap-aware-mapping/src/accuracy/mod.rs +++ b/crates/logical-optimizer/src/accuracy/mod.rs @@ -9,7 +9,6 @@ pub mod allocation; pub mod composition; pub mod estimators; pub mod evidence; -pub mod reconciliation; pub use allocation::{ AccuracyAllocation, AccuracyBudgetAllocator, CompositionShape, EqualSplitAllocator, @@ -20,18 +19,21 @@ pub use evidence::{ QuantileInputDomain, WorkloadAccuracyEvidence, }; -use asap_types::post_asap::{ - AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ExactOperation, FieldDataType, - GuaranteeSource, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchParams, - SketchStatistic, +use asap_types::ir::properties::{ + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, GuaranteeSource, ProbabilityExpr, + ResultGuarantee, }; +use asap_types::ir::schema::{FieldDataType, SketchAlgorithm, SketchParams, SketchStatistic}; +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; -/// The deployment-extensible accuracy algebra. `asap-aware-mapping` ships +use crate::pass1::exact_composition::ExactOperation; + +/// The deployment-extensible accuracy algebra. `asap-logical-optimizer` ships /// [`DefaultAccuracyModel`]; a deployment with a proof for a composition the /// default rejects (a registered cross-metric conversion, say) implements /// this trait and passes it to -/// [`crate::replacement::SketchAlgorithmStrategy::new_with_planning_inputs`]. +/// [`crate::pass1::replacement::ASAPStrategies::new_with_planning_inputs`]. pub trait AccuracyModel { /// The definition-registered rule for applying `operation` to an /// approximate input. `None` means the function is exact only over exact @@ -43,7 +45,7 @@ pub trait AccuracyModel { /// The guarantee of reading `query` out of a summary of family `family` /// built over an **exact** input — derived from the family's committed /// parameters by inverting the same sizing formulas - /// [`crate::replacement::default_size_params`] uses. `None` when this + /// [`crate::pass1::replacement::default_size_params`] uses. `None` when this /// model has no error model for the family (the default has none for /// `Sample`/`Wavelet`/`StatModel`). fn local_guarantee( @@ -78,7 +80,7 @@ pub struct DefaultAccuracyModel; const SATISFACTION_TOLERANCE: f64 = 1e-9; impl DefaultAccuracyModel { - /// Derive the guarantee for the committed estimator parameters and readout. + /// Derive the guarantee for the committed estimator parameters and evaluation. pub fn sketch_guarantee( algorithm: &SketchAlgorithm, params: &SketchParams, diff --git a/crates/logical-optimizer/src/lib.rs b/crates/logical-optimizer/src/lib.rs new file mode 100644 index 000000000..d5d6f458a --- /dev/null +++ b/crates/logical-optimizer/src/lib.rs @@ -0,0 +1,80 @@ +//! `asap-logical-optimizer` — #509 Stage 1: logical candidate generation. +//! +//! It takes the pre-ASAP [`OperatorNode`](asap_types::ir::OperatorNode) DAGs a +//! front end produces and proposes the logical alternatives for each target +//! sub-DAG: which summary (if any) realizes each approximate intent, and which +//! semantic rewrites, roll-ups, groupings and exact compositions apply. It +//! never prices a plan: only Stage 3 uses the cost model (#572, decision +//! Q36(a)). Cargo enforces the stage order: this crate depends only on +//! `asap-types`, never on a front end, a later stage or the executor. +//! +//! - [`pass1`] — local alternatives per target sub-DAG. +//! - [`pass2`] — ASAP-aware sharing across targets. +//! - [`accuracy`] — the analytical accuracy model: per-family error bounds and +//! sizing ([`accuracy::estimators`]), propagation through compositions +//! ([`accuracy::composition`]) and error-budget allocation +//! ([`accuracy::allocation`]). Candidates no analytical rule can prove +//! invalid are kept. +//! +//! **Common sub-expression elimination (CSE) of identical sub-DAGs is not +//! implemented here.** It runs over the pre-ASAP IR itself +//! (`asap_types::ir::cse`, issues #222/#223). Pass 2's identical-expression +//! rule ([`pass2::identical_expressions`]) decides when to use it: the stage +//! pipeline keeps a variant with and without it. Pass 2 also recognizes +//! sharing that is invisible at that level, such as `Quantile(x, 0.99)` and +//! `Quantile(x, 0.95)` reading one built sketch. +//! +//! ## Candidate search +//! +//! Search returns [`CandidateLogicalASAPDAGs`](pass1::replacement::CandidateLogicalASAPDAGs), +//! a compact logical choice space with one [`TargetSubDAGCandidates`] per +//! target sub-DAG. [`ReplacementStrategy`] implementations propose local +//! alternatives; search applies the applicable semantic and accuracy checks. +//! Candidate presence does not certify physical deployability or an unknown +//! accuracy guarantee. [`GlobalSelection`] assembles a DAG from given per-target +//! choices; choosing them is a later stage's job. +//! +//! | Term | Meaning | Entry point | +//! |---|---|---| +//! | Realization | Enumerate the physical forms for one aggregate intent | `pass1::replacement::realizations_for_intent` | +//! | Replacement | Construct each candidate summary sub-DAG | [`ASAPStrategies`] | +//! | Search | Enumerate alternatives across a workload | [`search_workload`] | +//! | Local candidates | The #509 stage pipeline's Stage 1 entry point | [`pass1::logical_candidates`] | +//! +//! [`Matcher`] asks whether an already available `Realization` satisfies a +//! required one. It has no shipped implementation: which realizations are +//! available is a deployment's concern. +//! +//! [`explanation`](pass1::explanation) reports, for each discovered target +//! with a non-trivial candidate list, why a replacement exists, reusing the +//! candidate's own rationale. + +pub mod accuracy; +pub mod pass1; +pub mod pass2; +#[cfg(test)] +mod test_support; + +pub use accuracy::{ + AccuracyAllocation, AccuracyBudgetAllocator, AccuracyEvidenceProvider, AccuracyModel, + CompositionShape, DefaultAccuracyModel, EqualSplitAllocator, NoAccuracyEvidence, + PropagationStats, WorkloadAccuracyEvidence, +}; +pub use pass1::exact_composition::{ + ExactComposition, ExactCompositionStrategy, OperationPlacement, +}; +pub use pass1::explanation::{ + explain_replacements, explain_replacements_with, ExplanationKind, ReplacementExplanation, +}; +pub use pass1::grouping::{has_subpopulations, HydraGroupingStrategy}; +pub use pass1::replacement::{ + default_strategies, is_logical_rewrite, search_workload, search_workload_with, + search_workload_with_targets, summary_candidates, ASAPStrategies, CandidateLogicalASAPDAGs, + GlobalSelection, Matcher, Proposals, Realization, RealizationError, RejectedCandidate, + Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, + SharedSubDAGStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, + MAX_SEARCH_ITERATIONS, +}; +pub use pass1::rewrite::{AvgToSumOverCountStrategy, SemanticEquivalentRewriteStrategy}; +pub use pass2::reconciliation::AccuracyReconciliationStrategy; +pub use pass2::topk_reuse::TopKLimitReuseStrategy; diff --git a/crates/asap-aware-mapping/src/exact_composition.rs b/crates/logical-optimizer/src/pass1/exact_composition.rs similarity index 53% rename from crates/asap-aware-mapping/src/exact_composition.rs rename to crates/logical-optimizer/src/pass1/exact_composition.rs index 54bce26f6..8b5fab902 100644 --- a/crates/asap-aware-mapping/src/exact_composition.rs +++ b/crates/logical-optimizer/src/pass1/exact_composition.rs @@ -4,23 +4,25 @@ //! `construct_summary_agg` already nests accumulator realizations, such as //! KLL over exact `Sum` state or a quantile over `Rate` state. This strategy //! covers the more general cases where an exact function must consume a -//! summary readout, or where a maintained summary consumes the values of an +//! summary evaluation, or where a maintained summary consumes the values of an //! exact function that has no accumulator realization. //! -//! Both cases use the general [`SummaryExpr::ValueOperation`] node. Its -//! semantic [`ValueOperation`] is independent from [`ExecutionTiming`], so -//! adding a function does not require adding a new physical node type. +//! Both cases use an ordinary `NonASAPOp::Aggregate` node over the child +//! plan. The node carries no timing: it runs when its consumer runs, so the +//! same operator serves both placements and adding a function does not +//! require adding a new physical node type. [`OperationPlacement`] is the +//! search-time placement choice. //! //! ## Reference, don't select //! //! A composed candidate needs a child plan to compose *with* — the inner -//! quantile's own summary readout, say. This strategy deliberately does +//! quantile's own summary evaluation, say. This strategy deliberately does //! **not** pick that child itself (the way `construct_summary_agg`'s //! `realize_child` takes the head of the child's own ranking): a //! [`Replacement::ExactComposition`] carries only the child *target* -//! (`ExactComposition::child_target`, the same `Rc` whose +//! (`ExactComposition::child_target`, the same `Rc` whose //! `TargetSubDAGCandidates` in `CandidateLogicalASAPDAGs` already holds every candidate for it). It is -//! [`CandidateLogicalASAPDAGs::global_selection`](crate::replacement::CandidateLogicalASAPDAGs::global_selection) +//! `candidate_selection::global_selection` //! that commits the compatible parent/child pair — so the child's own //! cost-model ranking, workload-wide effective consumer count, and shared //! `Rc` identity (one inner summary serving two outer folds) all stay @@ -34,20 +36,19 @@ //! //! - the target is a single-measure, `HAVING`-free exact aggregate; //! - read-time operation: the child is a bindable aggregate that has at least one -//! readout-producing summary implementation (a sketch/sample/wavelet/ +//! evaluation-producing summary implementation (a sketch/sample/wavelet/ //! model — the shapes a maintained accumulator can't sit above), and the //! target's grouping keys resolve in the child's output schema; //! transform: the target is a per-entity exact function with no //! accumulator form (its only implementation is `PassThrough`); //! - the exact operator consumes only `Plain` values in its data_state — checked -//! again, structurally, when the pair is composed; -//! - the plugged-in [`CostModel`] has not disproven the matching -//! [`ValueOperationCapabilities`](crate::cost_model::ValueOperationCapabilities). -//! Unknown support keeps the candidate visible; only explicit positive -//! support evidence permits global selection. +//! again, structurally, when the pair is composed. +//! +//! Runtime support is not checked here: selection admits a composition only +//! with explicit positive support evidence from its cost model. //! //! `avg` gets a read-time operation candidate *and* keeps -//! [`crate::rewrite::AvgToSumOverCountStrategy`]'s rewrite in the same +//! [`crate::pass1::rewrite::AvgToSumOverCountStrategy`]'s rewrite in the same //! group; the cost model picks between them, nothing here hard-codes one. //! //! ## What this strategy never does @@ -57,36 +58,91 @@ //! `RealizationError` regardless. //! - Decide whether a composition is *worth it*: that is //! `global_selection`'s job, using the issue's cost-units-per-second -//! formulas (see `crate::cost_model::read_operation_plan_cost_rate` and -//! siblings). Missing statistics keep the conservative `KeepPreAsap`. +//! formulas (see `cost_model::read_operation_plan_cost_rate` and +//! siblings). Missing statistics keep the conservative kept sub-DAG. +use asap_types::ir::operator::non_asap::any_measure_filtered; use std::rc::Rc; -use asap_types::post_asap::execution_data_state::validate_execution_data_states_at; -use asap_types::post_asap::{ - exact_operation_output_schema, produced_data_state, AccuracyError, ExactOperation, - ExecutionDataState, ExecutionDataStateError, ResultGuarantee, Schema, SummaryExpr, SummaryNode, - ValueOperation, +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::Reduction; +use asap_types::ir::properties::timing::{planned_data_state, validate_maintained}; +use asap_types::ir::properties::{ + AccuracyError, ExecutionDataState, ExecutionDataStateError, ResultGuarantee, }; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, QueryExpr, Reduction}; +use asap_types::ir::schema::aggregate_schema::aggregate_output_schema; +use asap_types::ir::schema::Schema; +use asap_types::ir::{NonASAPOp, Operator, OperatorNode, Predicate}; +use asap_types::physical::execution_data_state::lift_plain; +use asap_types::physical::ExactOperationSchemaError; use asap_types::types::AccuracyTarget; -use crate::cost_model::CostModel; -use crate::replacement::{ +use crate::pass1::replacement::{ bindable_intent, describe_intent, realizations_for_intent, Realization, RealizationError, Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; use crate::{AccuracyModel, DefaultAccuracyModel, PropagationStats}; #[cfg(test)] -use asap_types::post_asap::ExecutionTiming; +use asap_types::ir::properties::ExecutionTiming; /// Which side of the maintenance/read boundary an [`ExactComposition`]'s /// exact function executes on. +/// The exact function an [`ExactComposition`] applies: the parameters of +/// the `NonASAPOp::Aggregate` node the composition builds over its child. +#[derive(Debug, Clone, PartialEq)] +pub enum ExactOperation { + Aggregate { + reduction: Reduction, + measures: Vec, + output_names: Vec, + filters: Vec>, + having: Option, + }, +} + +impl ExactOperation { + /// Output schema of this operation over a child whose edge carries + /// `input` — the same canonical derivation the pre-ASAP `Aggregate` + /// node uses. `Err` when the child carries non-plain state the operator + /// cannot read. + pub fn output_schema(&self, input: &Schema) -> Result { + if !input.is_all_plain() { + return Err(ExactOperationSchemaError::NonPlainInput); + } + let plain = lift_plain(input); + let ExactOperation::Aggregate { + reduction, + measures, + output_names, + .. + } = self; + let out = aggregate_output_schema(&plain, reduction, measures, output_names)?; + Ok(lift_plain(&out)) + } + + fn into_op(self, child: Rc) -> NonASAPOp { + let ExactOperation::Aggregate { + reduction, + measures, + output_names, + filters, + having, + } = self; + NonASAPOp::Aggregate { + reduction, + measures, + output_names, + filters, + having, + child, + } + } +} + #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] pub enum OperationPlacement { - /// After the child's summary readout. + /// After the child's summary evaluation. Read, /// On the maintenance path, feeding /// maintained state above. @@ -120,7 +176,7 @@ pub struct ExactComposition { pub op: ExactOperation, /// The pre-ASAP child the operator consumes; its `TargetSubDAGCandidates` holds the /// candidates `global_selection` may commit this composition with. - pub child_target: Rc, + pub child_target: Rc, /// The composed node's output schema — the target's own pre-ASAP /// output schema, lifted with every column `Plain` (an exact operator /// only ever produces plain values). @@ -128,24 +184,27 @@ pub struct ExactComposition { } impl ExactComposition { + /// The data state `child` produces when this operation (its consumer) + /// runs at the placement's timing. + fn child_data_state(&self, child: &Rc) -> ExecutionDataState { + planned_data_state(child, self.placement.data_state().timing) + } + /// Can `child` legally be this composition's input? Phase legality - /// (the child's produced data_state — a `KeepPreAsap` leaf takes the + /// (the child's produced data_state — a kept pre-ASAP sub-DAG takes the /// phase this edge assigns) plus the plain-operand rule, checked /// through the same schema derivation [`Self::compose`] uses. - pub fn accepts_child(&self, child: &SummaryNode) -> bool { - let phase_ok = match produced_data_state(&child.expr) { - None => true, - Some(avail) => avail == self.placement.data_state(), - }; - phase_ok && exact_operation_output_schema(&self.op, &child.schema).is_ok() + pub fn accepts_child(&self, child: &Rc) -> bool { + self.child_data_state(child) == self.placement.data_state() + && self.op.output_schema(&child.schema).is_ok() } /// Build the composed, data_state-validated node over `child`. Every edge of /// the result (including everything beneath `child`) is checked by - /// `asap_types::post_asap::validate_execution_data_states`; an illegal + /// `asap_types::ir::properties::timing::validate_maintained`; an illegal /// placement is a typed [`RealizationError::ExecutionDataState`], never deferred to a /// runtime. - pub fn compose(&self, child: Rc) -> Result, RealizationError> { + pub fn compose(&self, child: Rc) -> Result, RealizationError> { self.compose_with_accuracy(child, &DefaultAccuracyModel) } @@ -154,24 +213,23 @@ impl ExactComposition { /// unsupported folds fail closed with a typed accuracy error. pub fn compose_with_accuracy( &self, - child: Rc, + child: Rc, accuracy_model: &dyn AccuracyModel, - ) -> Result, RealizationError> { - if let Some(produced) = produced_data_state(&child.expr) { - if produced != self.placement.data_state() { - let edge = match self.placement { - OperationPlacement::Maintenance => "ValueOperation.child (maintenance time)", - OperationPlacement::Read => "ValueOperation.child (read time)", - }; - return Err(RealizationError::ExecutionDataState( - ExecutionDataStateError::IllegalChildDataState { - edge, - child: produced, - }, - )); - } + ) -> Result, RealizationError> { + let produced = self.child_data_state(&child); + if produced != self.placement.data_state() { + let edge = match self.placement { + OperationPlacement::Maintenance => "exact operation child (maintenance time)", + OperationPlacement::Read => "exact operation child (read time)", + }; + return Err(RealizationError::ExecutionDataState( + ExecutionDataStateError::IllegalChildDataState { + edge, + child: produced, + }, + )); } - let schema = exact_operation_output_schema(&self.op, &child.schema)?; + let schema = self.op.output_schema(&child.schema)?; let guarantee = match &child.guarantee { None => None, Some(input) if input.is_exact() => Some(ResultGuarantee::exact(format!( @@ -198,36 +256,24 @@ impl ExactComposition { None => None, }, }; - let timing = match self.placement { - OperationPlacement::Read => asap_types::post_asap::ExecutionTiming::QueryTime, - OperationPlacement::Maintenance => { - asap_types::post_asap::ExecutionTiming::IngestionTime - } - }; - let expr = SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Exact(self.op.clone()), - timing, - }; - let node = Rc::new(SummaryNode { - expr, - schema, - guarantee, - }); - validate_execution_data_states_at(&node, self.placement.data_state())?; + let node = Rc::new( + OperatorNode::with_schema(Operator::NonASAP(self.op.clone().into_op(child)), schema) + .with_guarantee(guarantee), + ); + validate_maintained(&node, self.placement.data_state().timing)?; Ok(node) } /// Structural identity for `TargetSubDAGCandidates` dedup: same placement, same /// operator, same child `Rc`. - pub(crate) fn same_as(&self, other: &Self) -> bool { + pub fn same_as(&self, other: &Self) -> bool { self.placement == other.placement && self.op == other.op && Rc::ptr_eq(&self.child_target, &other.child_target) } } -/// Which exact reducers may run as a query-time fold over readout rows. +/// Which exact reducers may run as a query-time fold over evaluation rows. /// `Count` only at `Exact` accuracy (an approximate count is a sketch /// target, not an exact fold). fn is_query_time_reducer(intent: &AggIntent) -> bool { @@ -245,9 +291,9 @@ fn is_query_time_reducer(intent: &AggIntent) -> bool { ) } -/// Does `implementation` need a `SummaryEstimate` readout to yield a value +/// Does `implementation` need a `SummaryEstimate` evaluation to yield a value /// — i.e. is it a shape a maintained accumulator can't legally sit above? -fn needs_readout(implementation: &Realization) -> bool { +fn needs_evaluation(implementation: &Realization) -> bool { matches!( implementation, Realization::Sketch(_) @@ -258,18 +304,15 @@ fn needs_readout(implementation: &Realization) -> bool { } /// The `(op, child)` of a read-time operation-shaped target, or `None`. -fn query_time_shape( - root: &QueryExpr, - cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { +fn query_time_shape(root: &OperatorNode) -> Option<(ExactOperation, Rc, AggIntent)> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names, filters, having: None, child, - } = root + }) = root.non_asap() else { return None; }; @@ -289,15 +332,14 @@ fn query_time_shape( return None; } let child_intent = bindable_intent(child)?; - if !realizations_for_intent(child_intent, cost_model) + if !realizations_for_intent(child_intent) .iter() - .any(needs_readout) + .any(needs_evaluation) { return None; } // Grouping keys must resolve in the child's output schema — the same // derivation the composed node's own schema will use. - root.output_schema().ok()?; Some(( ExactOperation::Aggregate { reduction: reduction.clone(), @@ -314,17 +356,16 @@ fn query_time_shape( /// The `(op, child)` of a function-shaped target — a per-entity exact /// transform with no accumulator form — or `None`. fn ingestion_time_shape( - root: &QueryExpr, - cost_model: &dyn CostModel, -) -> Option<(ExactOperation, Rc, AggIntent)> { - let QueryExpr::Aggregate { + root: &OperatorNode, +) -> Option<(ExactOperation, Rc, AggIntent)> { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, output_names, filters, having: None, child, - } = root + }) = root.non_asap() else { return None; }; @@ -340,13 +381,12 @@ fn ingestion_time_shape( // Exact accumulators (`Rate`/`Increase`) are already directly nestable // as `SummaryAgg(ExactAggregate)`; only a pass-through function needs // an explicit update-path node. - if realizations_for_intent(intent, cost_model) + if realizations_for_intent(intent) .iter() .any(|i| *i != Realization::PassThrough) { return None; } - root.output_schema().ok()?; Some(( ExactOperation::Aggregate { reduction: Reduction::PerEntity, @@ -361,97 +401,62 @@ fn ingestion_time_shape( } /// Proposes [`Replacement::ExactComposition`] candidates — see the module -/// docs. Holds a [`CostModel`] only to ask it which mixed-execution shapes -/// the runtime advertises and which implementations the child has; it -/// never uses it to *rank* anything. -pub struct ExactCompositionStrategy<'a> { - cost_model: &'a dyn CostModel, -} - -static DEFAULT_COST_MODEL: crate::cost_model::DefaultCostModel = - crate::cost_model::DefaultCostModel; - -impl ExactCompositionStrategy<'static> { - /// A strategy consulting the built-in [`DefaultCostModel`](crate::cost_model::DefaultCostModel). - pub fn default_cost_model() -> Self { - Self { - cost_model: &DEFAULT_COST_MODEL, - } - } -} - -impl<'a> ExactCompositionStrategy<'a> { - pub fn new(cost_model: &'a dyn CostModel) -> Self { - Self { cost_model } - } +/// docs. +pub struct ExactCompositionStrategy; +impl ExactCompositionStrategy { fn candidates(&self, target: &TargetSubDAG<'_>) -> Vec { - let Ok(schema) = target.root.output_schema() else { - return Vec::new(); - }; - let schema = asap_types::post_asap::execution_data_state::lift_plain(&schema); + let schema = lift_plain(&target.root.schema); let mut out = Vec::new(); - if let Some((op, child, intent)) = query_time_shape(target.root, self.cost_model) { - if self - .cost_model - .value_operation_support_evidence(&op, OperationPlacement::Read) - != Some(false) - { - let child_desc = - describe_intent(bindable_intent(&child).expect("checked by query_time_shape")); - out.push(ReplacementSubDAG { - strategy: "ExactCompositionStrategy", - replacement: Replacement::ExactComposition(ExactComposition { - placement: OperationPlacement::Read, - op, - child_target: child, - schema: schema.clone(), - }), - provenance: ReplacementProvenance::ValueOperationAtQueryTime, - rationale: format!( - "{} is an exact fold whose input is the readout of {} — a maintained \ - accumulator cannot consume query-time values, so instead of collapsing \ - the whole DAG into KeepPreAsap this applies the fold as an \ - ExactRead over whichever summary readout global_selection \ - commits for the child target (asap_aware_mapping::exact_composition)", - describe_intent(&intent), - child_desc - ), - }); - } + if let Some((op, child, intent)) = query_time_shape(target.root) { + let child_desc = + describe_intent(bindable_intent(&child).expect("checked by query_time_shape")); + out.push(ReplacementSubDAG { + strategy: "ExactCompositionStrategy", + replacement: Replacement::ExactComposition(ExactComposition { + placement: OperationPlacement::Read, + op, + child_target: child, + schema: schema.clone(), + }), + provenance: ReplacementProvenance::ValueOperationAtQueryTime, + rationale: format!( + "{} is an exact fold whose input is the evaluation of {} — a maintained \ + accumulator cannot consume query-time values, so instead of keeping \ + the whole tree pre-ASAP this applies the fold as an \ + ExactRead over whichever summary evaluation global_selection \ + commits for the child target (asap_logical_optimizer::pass1::exact_composition)", + describe_intent(&intent), + child_desc + ), + }); } - if let Some((op, child, intent)) = ingestion_time_shape(target.root, self.cost_model) { - if self - .cost_model - .value_operation_support_evidence(&op, OperationPlacement::Maintenance) - != Some(false) - { - out.push(ReplacementSubDAG { - strategy: "ExactCompositionStrategy", - replacement: Replacement::ExactComposition(ExactComposition { - placement: OperationPlacement::Maintenance, - op, - child_target: child, - schema, - }), - provenance: ReplacementProvenance::ValueOperationAtIngestionTime, - rationale: format!( - "{} is an exact per-entity function with no accumulator form; as an \ - explicit ExactMaintenance on the update path its output can feed a \ - maintained summary above it instead of being handed over as an opaque \ - raw KeepPreAsap blob (asap_aware_mapping::exact_composition)", - describe_intent(&intent) - ), - }); - } + if let Some((op, child, intent)) = ingestion_time_shape(target.root) { + out.push(ReplacementSubDAG { + strategy: "ExactCompositionStrategy", + replacement: Replacement::ExactComposition(ExactComposition { + placement: OperationPlacement::Maintenance, + op, + child_target: child, + schema, + }), + provenance: ReplacementProvenance::ValueOperationAtIngestionTime, + rationale: format!( + "{} is an exact per-entity function with no accumulator form; as an \ + explicit ExactMaintenance on the update path its output can feed a \ + maintained summary above it instead of being handed over as an opaque \ + raw kept sub_dag (asap_logical_optimizer::pass1::exact_composition)", + describe_intent(&intent) + ), + }); } out } } -impl ReplacementStrategy for ExactCompositionStrategy<'_> { +impl ReplacementStrategy for ExactCompositionStrategy { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { !self.candidates(target).is_empty() } @@ -464,67 +469,28 @@ impl ReplacementStrategy for ExactCompositionStrategy<'_> { #[cfg(test)] mod tests { use super::*; - use crate::cost_model::{DefaultCostModel, ValueOperationCapabilities}; - use crate::replacement::keep_pre_asap; - use asap_types::post_asap::{ExecutionDataStateError, FieldDataType, SketchAlgorithm}; - use asap_types::pre_asap::agg_intent::default_quantile; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{DataType, Field, Schema}; - - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - - fn per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } + use crate::pass1::replacement::retain_exact; + use crate::test_support::{agg, agg_per_entity as per_entity, metric_scan, timed}; + use asap_types::ir::operator::agg_intent::default_quantile; + use asap_types::ir::properties::ExecutionDataStateError; + use asap_types::ir::schema::FieldDataType; + use asap_types::ir::ASAPOp; /// `max by (zone) (quantile by (zone, host) (m))`. - fn max_over_quantile() -> Rc { + fn max_over_quantile() -> Rc { let inner = agg( vec![2, 3], default_quantile(0.99), metric_scan(&["zone", "host"]), ); - Rc::new(agg(vec![0], AggIntent::Max { col: None }, inner)) + agg(vec![0], AggIntent::Max { col: None }, inner) } #[test] fn proposes_query_time_operation_for_max_over_quantile() { let root = max_over_quantile(); let target = TargetSubDAG::new(&root); - let strategy = ExactCompositionStrategy::default_cost_model(); + let strategy = ExactCompositionStrategy; assert!(strategy.matches(&target)); let candidates = strategy.replacements(&target); assert_eq!(candidates.len(), 1); @@ -539,7 +505,7 @@ mod tests { candidates[0].provenance, ReplacementProvenance::ValueOperationAtQueryTime ); - let QueryExpr::Aggregate { child, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = root.non_asap() else { unreachable!() }; assert!( @@ -553,23 +519,18 @@ mod tests { #[test] fn proposes_query_time_operation_for_avg_over_quantile_alongside_the_rewrite() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); - let root = Rc::new(agg(vec![0], AggIntent::Avg { col: None }, inner)); + let root = agg(vec![0], AggIntent::Avg { col: None }, inner); let target = TargetSubDAG::new(&root); - assert_eq!( - ExactCompositionStrategy::default_cost_model() - .replacements(&target) - .len(), - 1 - ); + assert_eq!(ExactCompositionStrategy.replacements(&target).len(), 1); // `avg` competes with AvgToSumOverCountStrategy in the same group. - assert!(crate::rewrite::AvgToSumOverCountStrategy.matches(&target)); + assert!(crate::pass1::rewrite::AvgToSumOverCountStrategy.matches(&target)); } #[test] fn proposes_ingestion_time_operation_for_a_per_entity_pass_through_over_raw_input() { - let root = Rc::new(per_entity(AggIntent::Deriv, metric_scan(&["zone"]))); + let root = per_entity(AggIntent::Deriv, metric_scan(&["zone"])); let target = TargetSubDAG::new(&root); - let candidates = ExactCompositionStrategy::default_cost_model().replacements(&target); + let candidates = ExactCompositionStrategy.replacements(&target); assert_eq!(candidates.len(), 1); assert_eq!( candidates[0].provenance, @@ -579,63 +540,38 @@ mod tests { #[test] fn does_not_propose_for_shapes_already_covered_by_accumulators() { - // sum by (zone) over an exact Sum child: the child has no readout, + // sum by (zone) over an exact Sum child: the child has no evaluation, // so SummaryAgg(Sum) over SummaryAgg(Sum) is already legal. let inner = agg( vec![2, 3], AggIntent::Sum { col: None }, metric_scan(&["zone", "host"]), ); - let root = Rc::new(agg(vec![0], AggIntent::Sum { col: None }, inner)); - assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&root))); + let root = agg(vec![0], AggIntent::Sum { col: None }, inner); + assert!(!ExactCompositionStrategy.matches(&TargetSubDAG::new(&root))); // rate is an exact accumulator — directly nestable, no separate value operation. - let rate = Rc::new(per_entity(AggIntent::Rate, metric_scan(&[]))); - assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&rate))); + let rate = per_entity(AggIntent::Rate, metric_scan(&[])); + assert!(!ExactCompositionStrategy.matches(&TargetSubDAG::new(&rate))); // A sketch-capable outer intent is not an exact fold. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["zone"])); - let root = Rc::new(agg(vec![0], default_quantile(0.99), inner)); - assert!(!ExactCompositionStrategy::default_cost_model().matches(&TargetSubDAG::new(&root))); - } - - struct NoMixedExecution; - impl CostModel for NoMixedExecution { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - fn value_operation_capabilities(&self) -> ValueOperationCapabilities { - ValueOperationCapabilities::NONE - } - } - - #[test] - fn a_runtime_without_the_capability_gets_no_candidate() { - let root = max_over_quantile(); - let target = TargetSubDAG::new(&root); - let strategy = ExactCompositionStrategy::new(&NoMixedExecution); - assert!(!strategy.matches(&target)); - assert!(strategy.replacements(&target).is_empty()); - let deriv = Rc::new(per_entity(AggIntent::Deriv, metric_scan(&[]))); - assert!(!strategy.matches(&TargetSubDAG::new(&deriv))); + let root = agg(vec![0], default_quantile(0.99), inner); + assert!(!ExactCompositionStrategy.matches(&TargetSubDAG::new(&root))); } #[test] fn compose_rejects_a_maintained_state_child_for_a_query_time_operation() { let root = max_over_quantile(); let target = TargetSubDAG::new(&root); - let candidates = ExactCompositionStrategy::default_cost_model().replacements(&target); + let candidates = ExactCompositionStrategy.replacements(&target); let Replacement::ExactComposition(comp) = &candidates[0].replacement else { unreachable!() }; - // A bare SummaryAgg (state, no readout) is not a legal read-time operation + // A bare SummaryAgg (state, no evaluation) is not a legal read-time operation // input — the operator would be consuming sketch state. - let state_child = - crate::replacement::realize_child(&comp.child_target, &DefaultCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &state_child.expr else { - panic!("expected the child to realize to a readout"); + let state_child = crate::pass1::replacement::realize_child(&comp.child_target).unwrap(); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &state_child.operator + else { + panic!("expected the child to realize to a evaluation"); }; assert!(!comp.accepts_child(summary_input)); assert!(matches!( @@ -644,7 +580,7 @@ mod tests { ExecutionDataStateError::IllegalChildDataState { .. } )) )); - // The readout itself is accepted and composes to a plain schema. + // The evaluation itself is accepted and composes to a plain schema. assert!(comp.accepts_child(&state_child)); let composed = comp.compose(state_child).unwrap(); assert!( @@ -652,12 +588,13 @@ mod tests { "rank error has no registered conversion through max" ); assert!(matches!( - composed.expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } + composed.operator, + Operator::NonASAP(NonASAPOp::Aggregate { .. }) )); + // Timing is no longer stored by composition: under the default + // materialization assignment the composed read-time operation runs at + // query time. + assert_eq!(timed(&composed).timing, Some(ExecutionTiming::QueryTime)); assert!(composed .schema .fields @@ -666,32 +603,76 @@ mod tests { } #[test] - fn compose_rejects_a_readout_child_for_a_ingestion_time_operation() { + fn compose_rejects_a_evaluation_child_for_a_ingestion_time_operation() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["zone"])); - let root = Rc::new(per_entity(AggIntent::Deriv, inner)); - let candidates = - ExactCompositionStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + let root = per_entity(AggIntent::Deriv, inner); + let candidates = ExactCompositionStrategy.replacements(&TargetSubDAG::new(&root)); let Replacement::ExactComposition(comp) = &candidates[0].replacement else { unreachable!() }; - let readout = - crate::replacement::realize_child(&comp.child_target, &DefaultCostModel).unwrap(); - assert!(!comp.accepts_child(&readout)); + let evaluation = crate::pass1::replacement::realize_child(&comp.child_target).unwrap(); + assert!(!comp.accepts_child(&evaluation)); assert!(matches!( - comp.compose(readout), + comp.compose(evaluation), Err(RealizationError::ExecutionDataState( ExecutionDataStateError::IllegalChildDataState { .. } )) )); // Raw update input is fine. - let raw = keep_pre_asap(&comp.child_target).unwrap(); + let raw = retain_exact(&comp.child_target).unwrap(); assert!(comp.accepts_child(&raw)); + // Timing is no longer stored by composition: the composition's + // placement is maintenance time, the composed exact operation is a + // plain Aggregate over the raw rows, and it is legal (and planned to + // run) at ingestion time. + assert_eq!(comp.placement, OperationPlacement::Maintenance); + let composed = comp.compose(raw).unwrap(); assert!(matches!( - comp.compose(raw).unwrap().expr, - SummaryExpr::ValueOperation { - timing: ExecutionTiming::IngestionTime, - .. - } + composed.operator, + Operator::NonASAP(NonASAPOp::Aggregate { .. }) + )); + validate_maintained(&composed, ExecutionTiming::IngestionTime).unwrap(); + assert_eq!( + planned_data_state(&composed, ExecutionTiming::IngestionTime).timing, + ExecutionTiming::IngestionTime + ); + } + + fn max_op(by: Vec) -> ExactOperation { + ExactOperation::Aggregate { + reduction: Reduction::by(by), + measures: vec![AggIntent::Max { col: None }], + output_names: vec![], + filters: vec![], + having: None, + } + } + + #[test] + fn exact_operator_schema_matches_pre_asap_aggregate_derivation() { + let child = lift_plain(&metric_scan(&["zone"]).schema); + let out = max_op(vec![2]).output_schema(&child).unwrap(); + let names: Vec<_> = out.fields.iter().map(|f| f.name.as_str()).collect(); + assert_eq!(names, vec!["zone", "max"]); + assert!(out.is_all_plain()); + } + + #[test] + fn exact_operator_rejects_non_plain_input() { + let state = Schema::lifted( + vec![asap_types::ir::schema::Field::new( + "state", + FieldDataType::ExactAggregate( + asap_types::ir::schema::ExactKind::Sum, + asap_types::ir::schema::ExactParams::Sum, + ), + false, + )], + None, + ); + assert!(matches!( + max_op(vec![]).output_schema(&state), + Err(ExactOperationSchemaError::NonPlainInput) )); } } diff --git a/crates/asap-aware-mapping/src/explanation.rs b/crates/logical-optimizer/src/pass1/explanation.rs similarity index 73% rename from crates/asap-aware-mapping/src/explanation.rs rename to crates/logical-optimizer/src/pass1/explanation.rs index e7afff2ba..7acaeebe6 100644 --- a/crates/asap-aware-mapping/src/explanation.rs +++ b/crates/logical-optimizer/src/pass1/explanation.rs @@ -1,5 +1,5 @@ //! This crate's **explanation of a replacement**: for a `TargetSubDAG` that -//! [`crate::replacement::search_workload`] found something to say about, why +//! [`crate::pass1::replacement::search_workload`] found something to say about, why //! does that candidate exist? (issue #33: "Add logic to detect which //! optimizations are applicable to a query workload"; this module: issue //! #257.) @@ -7,7 +7,7 @@ //! This module does not answer "is optimization X applicable here, yes or //! no" — that framing implies a classifier deciding admissibility from //! scratch. What it actually does is narrower and more mechanical: reuse a -//! matching candidate's own [`crate::replacement::ReplacementSubDAG::rationale`] +//! matching candidate's own [`crate::pass1::replacement::ReplacementSubDAG::rationale`] //! to explain, in the candidate's own words, why a [`Replacement`] exists at //! a given target. No new prose is invented here; see "The reframing" below //! for exactly what's being reused and why. @@ -18,10 +18,10 @@ //! Earlier (PR #247, superseded by this module — see "What this replaces" //! below), "is optimization X applicable here?" was a yes/no fact each rule //! re-derived by walking the DAG itself. That made sense before there was -//! any other structure to consult. But [`crate::replacement::search_workload`] +//! any other structure to consult. But [`crate::pass1::replacement::search_workload`] //! (issue #252) now *already* computes, for every -//! [`TargetSubDAG`](crate::replacement::TargetSubDAG) in the workload, every -//! semantically valid [`crate::replacement::ReplacementSubDAG`] a registered +//! `TargetSubDAG` in the workload, every +//! semantically valid [`crate::pass1::replacement::ReplacementSubDAG`] a registered //! [`ReplacementStrategy`] can propose — a [`CandidateLogicalASAPDAGs`] of [`TargetSubDAGCandidates`]s. A //! rule re-deriving the same yes/no fact from scratch would be answering a //! question the search already answered, via a second, independently @@ -31,7 +31,7 @@ //! collapses into a single question this module asks of *that* data instead: //! **for a given `TargetSubDAG`, does its candidate list contain anything //! other than the trivial, no-op realization?** A `TargetSubDAG` whose only -//! candidate is "the one thing `SketchAlgorithmStrategy` would have committed +//! candidate is "the one thing `ASAPStrategies` would have committed //! to anyway, with no alternative" has no optimization to report — that //! candidate isn't an *opportunity*, it's just the target's existing shape //! reflected back. A `TargetSubDAG` with more than one candidate (several @@ -43,10 +43,10 @@ //! [`TargetSubDAGCandidates`]s into that shape: //! //! - [`ExplanationKind::SketchApproximation`] — the `TargetSubDAG`'s -//! candidate list contains at least one [`Replacement::Summary`] that +//! candidate list contains at least one summary-realization [`Replacement::SubDAG`] that //! actually realizes a sketch family (`FieldDataType::Sketch`), i.e. -//! [`SketchAlgorithmStrategy`] found something to offer beyond whatever -//! exact/pass-through candidate [`crate::replacement`]'s own +//! [`ASAPStrategies`] found something to offer beyond whatever +//! exact/pass-through candidate [`crate::pass1::replacement`]'s own //! `realizations_for_intent` would have committed to on its own. //! - [`ExplanationKind::CommonSubexpressionReuse`] — the `TargetSubDAG` //! has two or more consumers *and* its candidate list contains the @@ -56,7 +56,7 @@ //! just an accident of how the workload happened to be built. //! //! Each finding's `reason` is literally the matching candidate's own -//! [`crate::replacement::ReplacementSubDAG::rationale`] (joined, if more than one candidate +//! [`crate::pass1::replacement::ReplacementSubDAG::rationale`] (joined, if more than one candidate //! qualifies) — this module invents no new prose to explain *why* a //! candidate is valid; that explanation already exists on the candidate a //! [`ReplacementStrategy`] produced, and repeating it here (rather than @@ -73,7 +73,7 @@ //! unrepresented). So do the two top-level entry points, //! [`explain_replacements`] and [`explain_replacements_with`] //! — same "workload roots in, findings out" contract, mirroring -//! [`crate::replacement::search_workload`]/[`crate::replacement::search_workload_with`]'s +//! [`crate::pass1::replacement::search_workload`]/[`crate::pass1::replacement::search_workload_with`]'s //! own signature shape. Only the *data source* changed: this module now //! calls those two functions and translates the result, rather than running //! its own rules and their supporting traversal over the DAG a second time. @@ -86,15 +86,15 @@ //! //! PR #247 gave this module its own extension-point trait, `ApplicabilityRule` //! (`fn optimization(&self) -> ExplanationKind` + `fn evaluate(&self, roots) -//! -> Vec`), the same shape [`crate::cost_model::CostModel`] -//! and [`crate::replacement::Matcher`] use elsewhere in this crate. Once +//! -> Vec`), the same shape `cost_model::CostModel` +//! and [`crate::pass1::replacement::Matcher`] use elsewhere in this crate. Once //! findings are a *view* over [`CandidateLogicalASAPDAGs`] rather than an independent //! computation, that trait would be a second extension point answering a //! question [`ReplacementStrategy`] (issue #251) already answers: "does this //! `TargetSubDAG` have an alternative worth reporting, and why". A caller who //! wants a new optimization represented as a finding needs a new //! `impl ReplacementStrategy` wired into -//! [`crate::replacement::search_workload_with`]'s strategy set *regardless* +//! [`crate::pass1::replacement::search_workload_with`]'s strategy set *regardless* //! (that's the only way its candidates end up in the [`CandidateLogicalASAPDAGs`] this //! module reads) — adding an `ApplicabilityRule` too would mean maintaining //! two extension points for the same new capability, one of which (the rule) @@ -102,15 +102,14 @@ //! produced. So this module ships no extension-point trait of its own: //! [`ReplacementStrategy`] already *is* that extension point, one layer //! down, and [`explain_replacements_with`]'s own `strategies` -//! parameter is where a caller plugs in a custom one (or a custom -//! `CostModel`, via [`crate::replacement::SketchAlgorithmStrategy::new`]) — the identical spot -//! [`crate::replacement::search_workload_with`] itself exposes. +//! parameter is where a caller plugs in a custom one — the identical spot +//! [`crate::pass1::replacement::search_workload_with`] itself exposes. //! //! ## Two guarantees the old traversal made, re-verified against the new one //! //! 1. **A finding is reported at the maximal `TargetSubDAG`, never once more -//! per subsumed descendant.** [`crate::replacement`]'s own -//! `discover_targets` (used by [`crate::replacement::search_workload_with`], +//! per subsumed descendant.** [`crate::pass1::replacement`]'s own +//! `discover_targets` (used by [`crate::pass1::replacement::search_workload_with`], //! and so by this module) walks every workload root's whole DAG but only //! *recurses into a node's children the first time that node's `Rc` is //! seen*; every subsequent occurrence still counts towards @@ -120,7 +119,7 @@ //! references it — identical to PR #247's own discovery pass, which //! reported "the highest point sharing starts," not a finding at every //! subsumed level below it. Same guarantee, same mechanism, just living in -//! [`crate::replacement`] now instead of here. +//! [`crate::pass1::replacement`] now instead of here. //! 2. **A node reachable via more than one path is one finding, not one per //! path.** [`TargetSubDAGCandidates`]s are keyed by `Rc` pointer identity in //! [`CandidateLogicalASAPDAGs`]'s internal map — there is exactly one group per distinct @@ -134,7 +133,7 @@ //! ## One thing [`CandidateLogicalASAPDAGs`] doesn't carry that this module still needs: //! human-readable `location` text //! -//! [`TargetSubDAGCandidates`]/[`CandidateLogicalASAPDAGs`] deliberately track only `Rc` +//! [`TargetSubDAGCandidates`]/[`CandidateLogicalASAPDAGs`] deliberately track only `Rc` //! pointer identity — the currency the search itself needs — not //! caller-facing prose. [`ReplacementExplanation::location`] is prose (a //! breadcrumb like `root "dash_a" > lhs`), so this module keeps one small, @@ -144,9 +143,9 @@ //! the deleted rule traversal: it makes no applicability decision (it runs //! the same regardless of what any strategy found), and duplicating this //! small, self-contained shape rather than threading location strings -//! through [`crate::replacement`]'s own `discover_targets` matches the same +//! through [`crate::pass1::replacement`]'s own `discover_targets` matches the same //! call that module's own docs already make for its (test-only) -//! `count_consumers` counterpart — see [`crate::replacement`]'s "Where +//! `count_consumers` counterpart — see [`crate::pass1::replacement`]'s "Where //! `TargetSubDAG` discovery comes from" section. //! //! ## Catalog primitives deliberately left as future work @@ -157,40 +156,39 @@ //! this codebase today**. Faking a variant for one of them would report a //! finding this codebase cannot back with a real candidate, so none of the //! below get an [`ExplanationKind`] variant yet — each gets one once a real -//! strategy exists and is wired into [`crate::replacement::default_strategies`]: +//! strategy exists and is wired into [`crate::pass1::replacement::default_strategies`]: //! //! | Catalog entry | Status | Where a future `ExplanationKind` would come from | //! |---|---|---| -//! | Semantic-equivalent rewriting (e.g. `avg` → `sum`/`count`) | [`AvgToSumOverCountStrategy`](crate::rewrite::AvgToSumOverCountStrategy) exists and is wired into `default_strategies()` (issue #253) — but still no `ExplanationKind` of its own below, since this table is about *direct* findings for a catalog entry, and this strategy's whole point is indirect: its `Replacement::Rewrite` candidate exposes `sum`/`count` as independently bindable discovered targets, which can then earn `CommonSubexpressionReuse` findings when the workload actually reuses them | A dedicated variant would need `findings_from_candidate_logical_asap_dags` to recognize a `LogicalRewrite`-provenance candidate as a finding in its own right, not just rely on what it exposes downstream | -//! | Roll-ups (fine-to-coarse group-by reuse) | [`RollupStrategy`](crate::rollup::RollupStrategy), derived from workload siblings after CSE/target discovery (issue #254) | Any `Replacement::Rewrite` candidate that rolls a coarse aggregate up from a compatible finer aggregate | -//! | Wavelets/OMP | Params type exists (`WaveletKind`/`WaveletParams`), reachable only via a deployment `CostModel::realize_extension` (no core `AggIntent` dispatch picks it) | A `ReplacementStrategy` that inspects a deployment's own `CostModel`, once some intent shape actually maps to `Realization::Wavelet` | +//! | Semantic-equivalent rewriting (e.g. `avg` → `sum`/`count`) | `AvgToSumOverCountStrategy` exists and is wired into `default_strategies()` (issue #253) — but still no `ExplanationKind` of its own below, since this table is about *direct* findings for a catalog entry, and this strategy's whole point is indirect: its `Replacement::SubDAG` rewrite candidate exposes `sum`/`count` as independently bindable discovered targets, which can then earn `CommonSubexpressionReuse` findings when the workload actually reuses them | A dedicated variant would need `findings_from_candidate_logical_asap_dags` to recognize a `LogicalRewrite`-provenance candidate as a finding in its own right, not just rely on what it exposes downstream | +//! | Roll-ups (fine-to-coarse group-by reuse) | `RollupStrategy`, derived from workload siblings after CSE/target discovery (issue #254) | Any `Replacement::SubDAG` rewrite candidate that rolls a coarse aggregate up from a compatible finer aggregate | +//! | Wavelets/OMP | Params type exists (`WaveletKind`/`WaveletParams`), not reachable (no core `AggIntent` dispatch picks it) | A `ReplacementStrategy` for it, once some intent shape actually maps to `Realization::Wavelet` | //! | Sampling | Same story as Wavelets: `SamplingKind`/`SamplingParams` exist, unreachable from core dispatch | Same hook as Wavelets, for `Realization::Sample` | -//! | Deep generative compression | No representation at all — no `Realization`/`FieldDataType` variant | Needs a new summary family added to `asap_types::post_asap` first | +//! | Deep generative compression | No representation at all — no `Realization`/`FieldDataType` variant | Needs a new summary family added to `asap_types::ir::schema::state_type` first | //! | Approximation frameworks for windows | No representation — `TimeRange`/`PromqlSubquery` windows are always evaluated exactly | Would key off those node types once an approximate-window operator exists | //! | Function decomposition | No representation anywhere | No hook point identified yet | //! | Continuous distributed monitoring | No representation — `RepeatingEntry`/`RepetitionInterval` in `asap_types::workload` describe *that* a query repeats, not any monitoring-specific decomposition | Would likely key off `RepeatingEntry` once such logic exists | //! | Incremental computation across time | No representation — nothing carries state across repeated evaluations of a `RepeatingEntry` today | Would key off `RepeatingEntry` + `TimeShift`/`TimeRange` once incremental state-carry exists | //! | Delta encoding | No representation — `AggIntent::Delta`/`IDelta` are PromQL *value*-difference semantics, not a wire/storage delta-encoding optimization | Would plug into a future deployment-side wire/storage encoding decision (post-ASAP), not this crate's IR-level dispatch | //! -//! [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy -//! [`ReplacementSubDAG`]: crate::replacement::ReplacementSubDAG -//! [`Replacement`]: crate::replacement::Replacement -//! [`Replacement::Summary`]: crate::replacement::Replacement::Summary -//! [`Replacement::Rewrite`]: crate::replacement::Replacement::Rewrite -//! [`SketchAlgorithmStrategy`]: crate::replacement::SketchAlgorithmStrategy -//! [`SharedSubDAGStrategy`]: crate::replacement::SharedSubDAGStrategy -//! [`CandidateLogicalASAPDAGs`]: crate::replacement::CandidateLogicalASAPDAGs -//! [`TargetSubDAGCandidates`]: crate::replacement::TargetSubDAGCandidates +//! [`ReplacementStrategy`]: crate::pass1::replacement::ReplacementStrategy +//! [`ReplacementSubDAG`]: crate::pass1::replacement::ReplacementSubDAG +//! [`Replacement`]: crate::pass1::replacement::Replacement +//! [`Replacement::SubDAG`]: crate::pass1::replacement::Replacement::SubDAG +//! [`ASAPStrategies`]: crate::pass1::replacement::ASAPStrategies +//! [`SharedSubDAGStrategy`]: crate::pass1::replacement::SharedSubDAGStrategy +//! [`CandidateLogicalASAPDAGs`]: crate::pass1::replacement::CandidateLogicalASAPDAGs +//! [`TargetSubDAGCandidates`]: crate::pass1::replacement::TargetSubDAGCandidates use std::collections::HashMap; use std::fmt::Display; use std::rc::Rc; -use asap_types::post_asap::{FieldDataType, SummaryExpr, SummaryNode}; -use asap_types::pre_asap::cse::{structural_hash, HashCache}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::ir::cse::{structural_hash, HashCache}; +use asap_types::ir::schema::FieldDataType; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; -use crate::replacement::{ +use crate::pass1::replacement::{ self, CandidateLogicalASAPDAGs, Replacement, ReplacementStrategy, TargetSubDAGCandidates, }; @@ -201,29 +199,29 @@ use crate::replacement::{ /// deliberately left as future work" table for everything else in the /// catalog). /// -/// [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy +/// [`ReplacementStrategy`]: crate::pass1::replacement::ReplacementStrategy #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] #[non_exhaustive] pub enum ExplanationKind { /// A `TargetSubDAG`'s candidate list contains at least one - /// [`Replacement::Summary`] that realizes a sketch family — - /// [`crate::replacement::SketchAlgorithmStrategy`] found a genuine sketch + /// [`Replacement::SubDAG`] that realizes a sketch family — + /// [`crate::pass1::replacement::ASAPStrategies`] found a genuine sketch /// alternative for this `Aggregate`, beyond whatever exact/pass-through - /// candidate `crate::replacement`'s own `realizations_for_intent` would + /// candidate `crate::pass1::replacement`'s own `realizations_for_intent` would /// have committed to on its own. SketchApproximation, /// A `TargetSubDAG` has two or more consumers *and* its candidate list - /// contains [`crate::replacement::SharedSubDAGStrategy`]'s "build once + /// contains [`crate::pass1::replacement::SharedSubDAGStrategy`]'s "build once /// and share" candidate — the catalog's cross-statistic / cross-metrics / /// cross-subpopulation reuse entries, all the same underlying structural /// fact. CommonSubexpressionReuse, /// A `TargetSubDAG`'s candidate list contains at least one /// [`Replacement::ExactComposition`] — - /// [`crate::exact_composition::ExactCompositionStrategy`] found an exact + /// [`crate::pass1::exact_composition::ExactCompositionStrategy`] found an exact /// operator that can be composed with a summary plan across an explicit - /// update/readout boundary instead of collapsing the whole DAG into - /// `KeepPreAsap` (issue #171). + /// update/evaluation boundary instead of keeping the whole tree as it is + /// (issue #171). ExactComposition, } @@ -231,13 +229,13 @@ pub enum ExplanationKind { /// breadcrumb into the workload — e.g. `root "dashboard_p99"` or /// `root "ratio" > lhs`): `reason` (human-readable, meant for a report/log, /// not machine parsing — literally the matching candidate's own -/// [`crate::replacement::ReplacementSubDAG::rationale`]). +/// [`crate::pass1::replacement::ReplacementSubDAG::rationale`]). /// -/// `node_hash` is [`structural_hash`](asap_types::pre_asap::cse::structural_hash) +/// `node_hash` is [`structural_hash`](asap_types::ir::cse::structural_hash) /// of the `TargetSubDAG`'s own `target` sub-DAG — the same function, on the -/// same `Rc` shape, that [`asap_types::dag_export::DAGNode::hash`] +/// same `Rc` shape, that [`asap_types::dag_export::DAGNode::hash`] /// is computed with. A downstream consumer that independently exported the -/// same `QueryExpr` (e.g. via `asap_types::dag_export::export`) can match +/// same node (e.g. via `asap_types::dag_export::export`) can match /// this explanation to a `DAGNode` by first comparing hashes and then /// confirming structural equality with [`ReplacementExplanation::target`]. #[derive(Debug, Clone, PartialEq)] @@ -249,42 +247,39 @@ pub struct ReplacementExplanation { /// The exact target expression the explanation describes. Reporting /// integrations use this together with `node_hash`: the hash narrows the /// search, and structural equality makes the final match collision-safe. - pub target: Rc, + pub target: Rc, } -/// Explain every replacement [`crate::replacement::search_workload`] finds +/// Explain every replacement [`crate::pass1::replacement::search_workload`] finds /// across a workload's pre-ASAP query roots, using -/// [`crate::replacement::default_strategies`]. +/// [`crate::pass1::replacement::default_strategies`]. /// -/// `roots` — like [`crate::replacement::search_workload`]'s own `Id` type +/// `roots` — like [`crate::pass1::replacement::search_workload`]'s own `Id` type /// parameter — is caller-chosen: a `QueryWorkload` entry's own key, an index, /// a query name. It only needs [`Display`], since a finding's `location` is /// prose, not a structured key back to the caller. /// -/// Internally runs [`crate::replacement::search_workload`] to build the +/// Internally runs [`crate::pass1::replacement::search_workload`] to build the /// candidate-plan space, then reads findings off it — see the module docs' /// "The reframing" section for what that translation actually checks. pub fn explain_replacements( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, Rc)>, ) -> Vec { explain_replacements_with(roots, &replacement::default_strategies()) } /// Like [`explain_replacements`], but searches with `strategies` -/// instead of [`crate::replacement::default_strategies`] — the extension -/// point for a deployment-specific [`ReplacementStrategy`], or a custom -/// `CostModel` plugged into -/// [`crate::replacement::SketchAlgorithmStrategy::new`] (e.g. via -/// [`crate::replacement::default_strategies_with`]). +/// instead of [`crate::pass1::replacement::default_strategies`] — the extension +/// point for a deployment-specific [`ReplacementStrategy`]. /// -/// [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy +/// [`ReplacementStrategy`]: crate::pass1::replacement::ReplacementStrategy pub fn explain_replacements_with<'s, Id: Display>( - roots: Vec<(Id, QueryExpr)>, + roots: Vec<(Id, Rc)>, strategies: &[Box], ) -> Vec { - let ided: Vec<(String, Rc)> = roots + let ided: Vec<(String, Rc)> = roots .into_iter() - .map(|(id, expr)| (id.to_string(), Rc::new(expr))) + .map(|(id, expr)| (id.to_string(), expr)) .collect(); let space = replacement::search_workload_with(ided, strategies); findings_from_candidate_logical_asap_dags(&space) @@ -297,7 +292,7 @@ pub fn explain_replacements_with<'s, Id: Display>( /// /// `space`'s own `Id` is always `String` here: [`explain_replacements_with`] /// already converted the caller's `Id: Display` into a `String` (via -/// `to_string()`) before calling [`crate::replacement::search_workload_with`], +/// `to_string()`) before calling [`crate::pass1::replacement::search_workload_with`], /// so this function (and [`collect_locations`], which formats `id` with /// [`std::fmt::Debug`] for the breadcrumb text) doesn't need its own generic /// `Id` bound. @@ -375,7 +370,7 @@ fn sketch_finding_reason(group: &TargetSubDAGCandidates) -> Option { .candidates .iter() .filter( - |c| matches!(&c.replacement, Replacement::Summary(node) if is_sketch_realization(node)), + |c| matches!(&c.replacement, Replacement::SubDAG(node) if is_sketch_realization(node)), ) .map(|c| c.rationale.as_str()) .collect(); @@ -387,7 +382,7 @@ fn sketch_finding_reason(group: &TargetSubDAGCandidates) -> Option { } /// Does `group` have two or more consumers *and* a "build once and share" -/// candidate (the [`Replacement::Rewrite`] whose `Rc` is the group's own +/// candidate (the [`Replacement::SubDAG`] whose `Rc` is the group's own /// `target`) in its candidate list? If so, the finding's `reason` is that /// candidate's own `rationale`. fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option { @@ -398,18 +393,18 @@ fn shared_subexpr_finding_reason(group: &TargetSubDAGCandidates) -> Option bool { +fn is_sketch_realization(node: &OperatorNode) -> bool { if node .guarantee .as_ref() @@ -417,9 +412,13 @@ fn is_sketch_realization(node: &SummaryNode) -> bool { { return false; } - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => is_sketch_realization(summary_input), - SummaryExpr::SummaryAgg { family, .. } => matches!(family, FieldDataType::Sketch(..)), + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + is_sketch_realization(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => { + matches!(family, FieldDataType::Sketch(..)) + } _ => false, } } @@ -432,8 +431,10 @@ fn is_sketch_realization(node: &SummaryNode) -> bool { /// every breadcrumb path that reaches a given `Rc`, not just the first: a /// shared node referenced from two workload roots (or two branches of one /// root) needs both breadcrumbs in its finding's `location`, not just one. -fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryExpr, Vec> { - let mut locations: HashMap<*const QueryExpr, Vec> = HashMap::new(); +fn collect_locations( + roots: &[(String, Rc)], +) -> HashMap<*const OperatorNode, Vec> { + let mut locations: HashMap<*const OperatorNode, Vec> = HashMap::new(); for (id, root) in roots { visit(root, format!("root {id:?}"), &mut locations); } @@ -444,9 +445,9 @@ fn collect_locations(roots: &[(String, Rc)]) -> HashMap<*const QueryE /// through its children. A shared ancestor is intentionally traversed once /// per incoming path so every descendant receives every valid breadcrumb. fn visit( - node: &Rc, + node: &Rc, label: String, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const OperatorNode, Vec>, ) { let ptr = Rc::as_ptr(node); locations.entry(ptr).or_default().push(label.clone()); @@ -454,21 +455,25 @@ fn visit( } /// `node`'s own **relational-skeleton** operator children — the same scope -/// `crate::replacement`'s own target-discovery `walk_children` (and -/// `asap_types::pre_asap::cse::share_common_sub_dags`'s `rebuild_children`) -/// use. Exhaustive over every `QueryExpr` variant: a new variant fails to -/// compile here until this match is extended too. +/// `crate::pass1::replacement`'s own target-discovery `walk_children` (and +/// `asap_types::ir::cse::share_common_sub_dags`) use. Exhaustive over every +/// `NonASAPOp` variant: a new variant fails to compile here until this match +/// is extended too. An ASAP node never occurs in a workload root. fn visit_children( - node: &QueryExpr, + node: &OperatorNode, label: &str, - locations: &mut HashMap<*const QueryExpr, Vec>, + locations: &mut HashMap<*const OperatorNode, Vec>, ) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { + use asap_types::ir::{NonASAPOp::*, ScalarExpr}; + let Operator::NonASAP(op) = &node.operator else { + return; + }; + match op { + Scan { .. } | Values { .. } => {} + PromqlVectorFromScalar(ScalarExpr::PromqlScalarFromVector(c)) => { visit(c, format!("{label} > child"), locations) } + PromqlVectorFromScalar(_) => {} PromqlRelabel { child, .. } | PromqlInfoEnrich { child, .. } | PromqlSeriesSample { child, .. } @@ -484,7 +489,7 @@ fn visit_children( | Limit { child, .. } => visit(child, format!("{label} > child"), locations), Concat { children, .. } => { for (i, c) in children.iter().enumerate() { - visit_children(c, &format!("{label} > concat[{i}]"), locations); + visit(c, format!("{label} > concat[{i}]"), locations); } } Join { left, right, .. } | SetOp { left, right, .. } => { @@ -495,31 +500,20 @@ fn visit_children( visit(lhs, format!("{label} > lhs"), locations); visit(rhs, format!("{label} > rhs"), locations); } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} } } #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::agg_intent::{default_quantile, AggIntent}; - use asap_types::pre_asap::query_expr::{Reduction, Source}; - use asap_types::pre_asap::schema::{DataType, Field, Schema}; + use asap_types::ir::operator::agg_intent::{default_quantile, AggIntent}; + use asap_types::ir::operator::operator_properties::{BinaryOpKind, Reduction, Source}; + use asap_types::ir::schema::{DataType, Field, Schema}; + use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, Predicate, ScalarExpr}; + use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { + fn metric_scan(labels: &[&str]) -> Rc { let mut columns = vec![ Field::plain("ts", DataType::Timestamp, false), Field::plain("value", DataType::Float64, false), @@ -529,22 +523,43 @@ mod tests { .iter() .map(|n| Field::plain(*n, DataType::Utf8, true)), ); - QueryExpr::Scan { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index(columns, 0, vec![]), - } + })) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], filters: vec![], having: None, - child: Rc::new(child), - } + child, + })) + .unwrap() + } + + fn binary( + kind: BinaryOpKind, + lhs: Rc, + rhs: Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs, + rhs, + })) + .unwrap() } // ── SketchApproximation ────────────────────────────────────────────── @@ -567,7 +582,7 @@ mod tests { } /// `node_hash` must be the literal `structural_hash` a downstream - /// consumer would compute over the *same* `QueryExpr` sub-DAG via + /// consumer would compute over the *same* `OperatorNode` sub-DAG via /// `asap_types::dag_export::export` — the whole point of carrying it is /// that two independent exports of the same DAG agree, with no /// string-matching against `location` required. @@ -586,7 +601,7 @@ mod tests { Some(sketch.node_hash), expected_hash, "ReplacementExplanation::node_hash must match dag_export's DAGNode::hash \ - for the same QueryExpr sub-DAG" + for the same OperatorNode sub_dag" ); } @@ -641,21 +656,18 @@ mod tests { /// A sketch-applicable `Aggregate` reachable via two paths that CSE /// collapses onto one `Rc` — the same `median(x) == median(x)` shape - /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_sub_dag` + /// `ir::cse`'s own `single_query_shares_its_own_repeated_sub-DAG` /// test uses — must be reported once, not once per path: it is exactly - /// one [`crate::replacement::TargetSubDAGCandidates`], keyed by `Rc` pointer identity, + /// one [`crate::pass1::replacement::TargetSubDAGCandidates`], keyed by `Rc` pointer identity, /// not one per path that reaches it. #[test] fn a_shared_sketchable_aggregate_is_reported_only_once() { let quantile = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let root = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( - asap_types::pre_asap::expr_ir::CompareOpKind::Eq, - ), - lhs: Rc::new(quantile.clone()), - rhs: Rc::new(quantile), - vector_match: None, - }; + let root = binary( + BinaryOpKind::Compare(asap_types::ir::scalar::CompareOpKind::Eq), + Rc::clone(&quantile), + quantile, + ); let findings = explain_replacements(vec![("ratio", root)]); let sketch: Vec<_> = findings .iter() @@ -695,8 +707,8 @@ mod tests { #[test] fn descendant_of_a_shared_root_keeps_every_root_breadcrumb() { let inner = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let outer = agg(vec![2], AggIntent::Sum { col: None }, inner); - let findings = explain_replacements(vec![("dash_a", outer.clone()), ("dash_b", outer)]); + let outer = agg(vec![0], AggIntent::Sum { col: None }, inner); + let findings = explain_replacements(vec![("dash_a", Rc::clone(&outer)), ("dash_b", outer)]); let inner_sketch = findings .iter() .find(|f| { @@ -742,14 +754,11 @@ mod tests { // The same shared branch appearing twice within one query (an `a/a` // shape) — single-query CSE. let branch = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let q = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Arithmetic( - asap_types::pre_asap::expr_ir::ArithmeticOpKind::Div, - ), - lhs: Rc::new(branch.clone()), - rhs: Rc::new(branch), - vector_match: None, - }; + let q = binary( + BinaryOpKind::Arithmetic(asap_types::ir::scalar::ArithmeticOpKind::Div), + Rc::clone(&branch), + branch, + ); let findings = explain_replacements(vec![("ratio", q)]); let reuse: Vec<_> = findings .iter() @@ -765,26 +774,29 @@ mod tests { } /// A shared node nested three levels under two *different*, unshared - /// `Filter` parents (mirrors `crate::replacement::tests:: - /// nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered`) + /// `Filter` parents (mirrors `crate::pass1::replacement::tests:: + /// nested_shared_sub-DAG_below_an_unshared_parent_is_still_discovered`) /// must still be exactly one finding — the maximal-`TargetSubDAG` /// guarantee the module docs describe, now provided by - /// `crate::replacement`'s own target discovery rather than this module's + /// `crate::pass1::replacement`'s own target discovery rather than this module's /// (deleted) traversal. #[test] fn a_deeply_shared_sub_dag_under_different_parents_is_reported_once() { - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; + use asap_types::ir::scalar::ScalarValue; let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::new(shared.clone()), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::new(shared), - }; + let root_a = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: Rc::clone(&shared), + })) + .unwrap(); + let root_b = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: shared, + })) + .unwrap(); let findings = explain_replacements(vec![("a", root_a), ("b", root_b)]); let reuse: Vec<_> = findings .iter() @@ -798,37 +810,4 @@ mod tests { assert!(reuse[0].location.contains('a')); assert!(reuse[0].location.contains('b')); } - - // ── Custom strategy set / cost model plumbing ─────────────────────── - - struct AlwaysDDSketch; - impl crate::cost_model::CostModel for AlwaysDDSketch { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { - let mut v = candidates.to_vec(); - if let Some(pos) = v - .iter() - .position(|k| *k == asap_types::post_asap::SketchAlgorithm::DDSketch) - { - let dd = v.remove(pos); - v.insert(0, dd); - } - v - } - } - - #[test] - fn custom_cost_model_changes_the_reported_sketch_kind() { - let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let custom_model = AlwaysDDSketch; - let strategies: Vec> = vec![Box::new( - crate::replacement::SketchAlgorithmStrategy::new(&custom_model), - )]; - let findings = explain_replacements_with(vec![("q", q)], &strategies); - assert_eq!(findings.len(), 1); - assert!(findings[0].reason.to_lowercase().contains("ddsketch")); - } } diff --git a/crates/asap-aware-mapping/src/function_rules.rs b/crates/logical-optimizer/src/pass1/function_rules.rs similarity index 93% rename from crates/asap-aware-mapping/src/function_rules.rs rename to crates/logical-optimizer/src/pass1/function_rules.rs index 0b96e0184..3c83f2e2c 100644 --- a/crates/asap-aware-mapping/src/function_rules.rs +++ b/crates/logical-optimizer/src/pass1/function_rules.rs @@ -1,7 +1,8 @@ //! Function facts shared by value-operation propagation and accumulator realization. //! Runtime support remains a deployment decision in `CostModel`. -use asap_types::post_asap::{CompositionOperator, ExactKind, ExactParams}; -use asap_types::pre_asap::AggIntent; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::properties::CompositionOperator; +use asap_types::ir::schema::{ExactKind, ExactParams}; pub(crate) struct FunctionRules { pub accuracy: CompositionOperator, diff --git a/crates/asap-aware-mapping/src/grouping.rs b/crates/logical-optimizer/src/pass1/grouping.rs similarity index 73% rename from crates/asap-aware-mapping/src/grouping.rs rename to crates/logical-optimizer/src/pass1/grouping.rs index fc4abfe62..413acc6cc 100644 --- a/crates/asap-aware-mapping/src/grouping.rs +++ b/crates/logical-optimizer/src/pass1/grouping.rs @@ -3,11 +3,11 @@ //! per `by` subpopulation (today's only, implicit behavior) or as one //! shared Hydra-family structure serving all of them — orthogonal to //! *which* summary family/kind answers the intent, the same way -//! [`asap_types::post_asap::GroupingStrategy`]'s own doc explains. +//! [`asap_types::ir::schema::GroupingStrategy`]'s own doc explains. //! //! ## Placement: planning metadata and edge-state type //! -//! `SummaryExpr::SummaryAgg` carries the grouping choice next to the +//! `ASAPOp::SummaryAgg` carries the grouping choice next to the //! `Reduction` whose `by` keys determine legality. The same choice is also //! committed to `FieldDataType::Sketch` on the aggregate's output edge. //! That duplication is intentional: the node field makes the choice easy to @@ -15,7 +15,7 @@ //! CMS state and a Hydra-backed state cannot be accepted as compatible inputs //! to a downstream `SummaryMerge`. [`with_grouping`] updates both atomically. //! -//! ## Legality vs. cost (same split [`crate::replacement::realizations_for_intent`] +//! ## Legality vs. cost (same split [`crate::pass1::replacement::realizations_for_intent`] //! already draws) //! //! This module only answers "is `SharedMultiSubpopulation` valid here at @@ -26,7 +26,7 @@ //! with no grouping concept at all) has nothing for a //! shared-multi-subpopulation structure to multiplex across. //! - **The family has a Hydra variant** -//! ([`asap_types::post_asap::hydra_kind_for`]): `Cms` and `CountSketch` +//! ([`asap_types::ir::schema::hydra_kind_for`]): `Cms` and `CountSketch` //! have structural Hydra mappings. `HydraKll` remains an explicit //! experimental IR value, but the paper excludes quantiles and search //! therefore never emits it. The shared-grid term is represented @@ -41,19 +41,19 @@ //! An earlier draft of this module (written against the very first draft of //! #251) reused a `CostModel`-wrapping adapter that "steered" a //! whole-recursive-bind decision procedure toward a specific `SketchKind`, -//! the same pattern [`crate::replacement::SketchAlgorithmStrategy`]'s own module +//! the same pattern [`crate::pass1::replacement::ASAPStrategies`]'s own module //! docs explain was deliberately deleted from this crate as an anti-pattern: //! forcing a choice via a whole-DAG `CostModel` adapter had a real bug where //! the forced choice could leak into a target's own nested aggregates. This -//! module never needs that: [`crate::replacement::realizations_for_intent`] +//! module never needs that: [`crate::pass1::replacement::realizations_for_intent`] //! already returns every ranked candidate `Realization` directly, so //! [`build_candidate`](HydraGroupingStrategy::build_candidate) just finds the //! one whose `Realization::Sketch(kind)` has `kind.algorithm()` matching //! the Hydra-eligible `sketch_kind` it's building a candidate for, and //! passes that exact, //! already-decided `Realization` to -//! [`crate::replacement::construct_summary`] — the same first-class, -//! one-candidate-at-a-time primitive [`crate::replacement::SketchAlgorithmStrategy`] +//! [`crate::pass1::replacement::construct_summary`] — the same first-class, +//! one-candidate-at-a-time primitive [`crate::pass1::replacement::ASAPStrategies`] //! itself calls once per candidate. No adapter, no steering, no risk of a //! forced choice leaking into nested aggregates. //! @@ -71,19 +71,22 @@ use std::rc::Rc; -use asap_types::post_asap::{ - default_hydra_params, hydra_kind_for, AccuracyError, BoundExpr, CompositionOperator, - FieldDataType, GroupingStrategy, GuaranteeSource, HydraKind, ProbabilityExpr, ResultGuarantee, - SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::Reduction; +use asap_types::ir::properties::{ + AccuracyError, BoundExpr, CompositionOperator, GuaranteeSource, ProbabilityExpr, + ResultGuarantee, }; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; +use asap_types::ir::schema::{ + default_hydra_params, hydra_kind_for, FieldDataType, GroupingStrategy, HydraKind, + SketchAlgorithm, SketchParams, +}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; use crate::accuracy::{ AccuracyBudgetAllocator, AccuracyEvidenceProvider, AccuracyModel, PropagationStats, }; -use crate::cost_model::{CostModel, DefaultCostModel}; -use crate::replacement::{ +use crate::pass1::replacement::{ accuracy_target, bindable_intent, construct_summary_with, describe_intent, realizations_for_intent, summary_candidates, CandidatePlanningInputs, Proposals, Realization, RejectedCandidate, Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, @@ -108,17 +111,12 @@ pub fn has_subpopulations(reduction: &Reduction) -> bool { } } -/// A single static instance so [`HydraGroupingStrategy::default_cost_model`] -/// can hand out a `&'static dyn CostModel` without heap-allocating one — same -/// pattern [`crate::replacement::SketchAlgorithmStrategy`] uses. -static DEFAULT_COST_MODEL: DefaultCostModel = DefaultCostModel; - /// Wraps the `GroupingStrategy` axis (issue #256) as a -/// [`ReplacementStrategy`]: for a target [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) +/// [`ReplacementStrategy`]: for a target `ASAPStrategies` /// already has an opinion on, offers an additional /// `GroupingStrategy::SharedMultiSubpopulation` candidate wherever the /// legality conditions in the module docs above hold — alongside, not -/// instead of, the per-subpopulation candidates `SketchAlgorithmStrategy` +/// instead of, the per-subpopulation candidates `ASAPStrategies` /// itself enumerates. The workload search composes both strategies over the /// same target, so it sees every summary-family alternative *and* the Hydra /// alternative; the built-in workload search registers both strategies, and @@ -129,37 +127,24 @@ pub struct HydraGroupingStrategy<'a> { planning_inputs: CandidatePlanningInputs<'a>, } -impl HydraGroupingStrategy<'static> { - /// A strategy that ranks/binds via the built-in [`DefaultCostModel`] — - /// what a deployment gets with no custom cost model plugged in, the same - /// default [`crate::replacement::SketchAlgorithmStrategy::default_cost_model`] - /// offers. - pub fn default_cost_model() -> Self { +impl Default for HydraGroupingStrategy<'static> { + /// The built-in accuracy models, the same default + /// [`crate::pass1::replacement::ASAPStrategies`] uses. + fn default() -> Self { Self { - planning_inputs: CandidatePlanningInputs::with_default_accuracy(&DEFAULT_COST_MODEL), + planning_inputs: CandidatePlanningInputs::with_default_accuracy(), } } } impl<'a> HydraGroupingStrategy<'a> { - /// A strategy that ranks/binds via `cost_model` instead of the built-in - /// static preference order — the same customization point - /// [`crate::replacement::SketchAlgorithmStrategy::new`] already offers. - pub fn new(cost_model: &'a dyn CostModel) -> Self { - Self { - planning_inputs: CandidatePlanningInputs::with_default_accuracy(cost_model), - } - } - pub fn new_with_planning_inputs_and_evidence( - cost_model: &'a dyn CostModel, accuracy_model: &'a dyn AccuracyModel, allocator: &'a dyn AccuracyBudgetAllocator, evidence: &'a dyn AccuracyEvidenceProvider, ) -> Self { Self { planning_inputs: CandidatePlanningInputs { - cost: cost_model, accuracy: accuracy_model, allocator, evidence, @@ -173,7 +158,7 @@ impl<'a> HydraGroupingStrategy<'a> { /// variant modeled. fn hydra_proposals(&self, target: &TargetSubDAG<'_>) -> Proposals { let mut proposals = Proposals::default(); - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = target.root.non_asap() else { return proposals; }; if !has_subpopulations(reduction) { @@ -202,23 +187,23 @@ impl<'a> HydraGroupingStrategy<'a> { /// Find the already-ranked candidate [`Realization::Sketch`] matching /// `sketch_kind` among [`realizations_for_intent`]'s exhaustive list for /// `intent`, bind `root` to that exact, already-decided candidate via - /// [`crate::replacement::construct_summary_with`] (no steering/forcing — see + /// [`crate::pass1::replacement::construct_summary_with`] (no steering/forcing — see /// the module docs' "No `ForceSketchKind`-style steering"), then swap the /// resulting `SummaryAgg`'s `grouping` field from the default /// `PerSubpopulationInstance` to /// `SharedMultiSubpopulation { kind: hydra_kind, .. }` — reusing the /// entire bind decision procedure (schema derivation, column resolution, - /// readout construction) unchanged, patching only the one field this + /// evaluation construction) unchanged, patching only the one field this /// axis owns. fn build_candidate( &self, - root: &Rc, + root: &Rc, intent: &AggIntent, sketch_kind: SketchAlgorithm, hydra_kind: HydraKind, rejected: &mut Vec, ) -> Option { - let realization = realizations_for_intent(intent, self.planning_inputs.cost) + let realization = realizations_for_intent(intent) .into_iter() .find(|candidate| { matches!(candidate, Realization::Sketch(kind) if *kind.algorithm() == sketch_kind) @@ -233,12 +218,12 @@ impl<'a> HydraGroupingStrategy<'a> { params, }; - let (family, query) = match &node.expr { - SummaryExpr::SummaryEstimate { + let (family, query) = match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => match &summary_input.expr { - SummaryExpr::SummaryAgg { family, .. } => (family, Some(query)), + }) => match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => (family, Some(query)), _ => return None, }, _ => return None, @@ -295,14 +280,14 @@ impl<'a> HydraGroupingStrategy<'a> { } Some(ReplacementSubDAG { strategy: "HydraGroupingStrategy", - replacement: Replacement::Summary(patched), - provenance: crate::replacement::ReplacementProvenance::SummaryRealization, + replacement: Replacement::SubDAG(patched), + provenance: crate::pass1::replacement::ReplacementProvenance::SummaryRealization, rationale: format!( "{} realizes as a shared {hydra_kind:?} structure over {sketch_kind:?} \ serving every subpopulation of this grouped aggregate, instead of one \ {sketch_kind:?} instance per distinct `by` key — legal because this \ aggregate has a non-empty subpopulation concept and {sketch_kind:?} has a \ - modeled Hydra variant (asap_types::post_asap::hydra_kind_for); whether it's \ + modeled Hydra variant (asap_types::ir::schema::hydra_kind_for); whether it's \ *worth* the shared/independent trade-off for the actual subpopulation \ cardinality is a CostModel's call, not this strategy's", describe_intent(intent) @@ -313,7 +298,7 @@ impl<'a> HydraGroupingStrategy<'a> { impl ReplacementStrategy for HydraGroupingStrategy<'_> { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - let QueryExpr::Aggregate { reduction, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = target.root.non_asap() else { return false; }; if !has_subpopulations(reduction) { @@ -322,7 +307,7 @@ impl ReplacementStrategy for HydraGroupingStrategy<'_> { let Some(intent) = bindable_intent(target.root) else { return false; }; - realizations_for_intent(intent, self.planning_inputs.cost) + realizations_for_intent(intent) .into_iter() .any(|realization| { matches!(realization, @@ -357,15 +342,15 @@ impl ReplacementStrategy for HydraGroupingStrategy<'_> { /// destructures the right variant for `kind`; this function's only job is /// to find whatever `SketchParams` the bind decision already committed to /// and hand the whole thing over unchanged. -fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { +fn per_subpopulation_sketch_params(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { per_subpopulation_sketch_params(summary_input) } - SummaryExpr::SummaryAgg { + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } => Some(kind.params().clone()), + }) => Some(kind.params().clone()), _ => None, } } @@ -373,33 +358,35 @@ fn per_subpopulation_sketch_params(node: &SummaryNode) -> Option { /// Rebuild `node`, replacing its `SummaryAgg`'s `grouping` field with /// `grouping` — patching the one field this axis owns onto an /// already-correctly-bound node rather than re-deriving the rest of it. -/// Recurses through a `SummaryEstimate` readout wrapper (the shape every +/// Recurses through a `SummaryEstimate` evaluation wrapper (the shape every /// sketch candidate this module builds actually has) to reach the /// `SummaryAgg` underneath. fn with_grouping( - node: Rc, + node: Rc, grouping: GroupingStrategy, stats: &PropagationStats, -) -> Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { +) -> Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: with_grouping(Rc::clone(summary_input), grouping, stats), - query: query.clone(), - }, - schema: node.schema.clone(), - guarantee: node.guarantee.as_ref().map(|g| hydra_guarantee(g, stats)), - }), - SummaryExpr::SummaryAgg { + }) => std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: with_grouping(Rc::clone(summary_input), grouping, stats), + query: query.clone(), + }), + node.schema.clone(), + ) + .with_guarantee(node.guarantee.as_ref().map(|g| hydra_guarantee(g, stats))), + ), + Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, .. - } => { + }) => { let grouped_family = match family { FieldDataType::Sketch(kind, _) => { FieldDataType::Sketch(kind.clone(), grouping.clone()) @@ -412,17 +399,21 @@ fn with_grouping( field.dtype = FieldDataType::Sketch(kind.clone(), grouping.clone()); } } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(child), - family: grouped_family, - input: input.clone(), - reduction: reduction.clone(), - grouping, - filter: None, - }, - schema: grouped_schema, - guarantee: None, + // Regrouping the same state leaves the observations it covers unchanged. + std::rc::Rc::new(OperatorNode { + coverage: node.coverage.clone(), + ..OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(child), + family: grouped_family, + input: input.clone(), + reduction: reduction.clone(), + grouping, + filter: None, + }), + grouped_schema, + ) + .with_guarantee(None) }) } // Never reached by this module's own callers (they only ever pass a @@ -492,51 +483,11 @@ fn hydra_guarantee(inner: &ResultGuarantee, stats: &PropagationStats) -> ResultG mod tests { use super::*; use crate::accuracy::{DefaultAccuracyModel, EqualSplitAllocator}; - use asap_types::post_asap::ErrorMetric; - use asap_types::pre_asap::agg_intent::{default_cardinality, default_quantile}; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{DataType, Field, Schema}; + use crate::test_support::{agg, agg_per_entity, metric_scan}; + use asap_types::ir::operator::agg_intent::{default_cardinality, default_quantile}; + use asap_types::ir::properties::ErrorMetric; use asap_types::types::AccuracyTarget; - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } - } - - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - // ── has_subpopulations ──────────────────────────────────────────────── #[test] @@ -556,7 +507,7 @@ mod tests { #[test] fn without_grouping_has_a_subpopulation_concept_even_when_empty() { - use asap_types::pre_asap::query_expr::GroupKeys; + use asap_types::ir::operator::operator_properties::GroupKeys; // `without([])` groups by every remaining label — a real // subpopulation concept, unlike `by([])`'s genuine full reduction. assert!(has_subpopulations(&Reduction::Reduce(GroupKeys::without( @@ -603,45 +554,42 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - assert!(HydraGroupingStrategy::default_cost_model().matches(&target)); + assert!(HydraGroupingStrategy::default().matches(&target)); } #[test] fn does_not_match_an_empty_by_aggregate() { // Global reduction — no subpopulation concept, no Hydra alternative. - let q = Rc::new(agg(vec![], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let strategy = HydraGroupingStrategy::default_cost_model(); + let strategy = HydraGroupingStrategy::default(); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); } #[test] fn does_not_match_a_per_entity_aggregate() { - let q = Rc::new(agg_per_entity( - default_quantile(0.99), - metric_scan(&["job"]), - )); + let q = agg_per_entity(default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let strategy = HydraGroupingStrategy::default_cost_model(); + let strategy = HydraGroupingStrategy::default(); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); } #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); - assert!(!HydraGroupingStrategy::default_cost_model().matches(&target)); + assert!(!HydraGroupingStrategy::default().matches(&target)); } #[test] fn quantile_has_no_hydra_candidate_without_a_modeled_error_bound() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = HydraGroupingStrategy::default_cost_model().replacements(&target); + let replacements = HydraGroupingStrategy::default().replacements(&target); assert!(replacements.is_empty(), "{replacements:?}"); } @@ -653,13 +601,13 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let replacements = HydraGroupingStrategy::default_cost_model().replacements(&target); + let replacements = HydraGroupingStrategy::default().replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|guarantee| guarantee.bound.evaluate().is_none() && guarantee.failure_probability.evaluate().is_none()) @@ -673,7 +621,7 @@ mod tests { &self, _op: &CompositionOperator, _family: &FieldDataType, - _query: Option<&asap_types::post_asap::SketchStatistic>, + _query: Option<&asap_types::ir::schema::SketchStatistic>, ) -> PropagationStats { PropagationStats { hydra_shared_grid_collision_bound: Some(0.0), @@ -691,9 +639,8 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, &ZeroSharedGridEvidence, @@ -702,7 +649,7 @@ mod tests { assert_eq!(replacements.len(), 2, "{replacements:?}"); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|g| g.bound.evaluate().is_some() && g.failure_probability.evaluate().is_some()) @@ -717,7 +664,7 @@ mod tests { &self, _op: &CompositionOperator, _family: &FieldDataType, - _query: Option<&asap_types::post_asap::SketchStatistic>, + _query: Option<&asap_types::ir::schema::SketchStatistic>, ) -> PropagationStats { PropagationStats { hydra_shared_grid_failure_probability: Some(1.5), @@ -731,9 +678,8 @@ mod tests { delta: 0.01, }, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, &InvalidEvidence, @@ -746,7 +692,7 @@ mod tests { .iter() .all(|r| matches!(r.error, AccuracyError::UnsupportedComposition { .. }))); - let space = crate::replacement::search_workload_with( + let space = crate::pass1::replacement::search_workload_with( vec![("q", Rc::clone(&q))], &[Box::new(strategy)], ); @@ -763,7 +709,7 @@ mod tests { &self, _op: &CompositionOperator, _family: &FieldDataType, - _query: Option<&asap_types::post_asap::SketchStatistic>, + _query: Option<&asap_types::ir::schema::SketchStatistic>, ) -> PropagationStats { PropagationStats { hydra_shared_grid_collision_bound: Some(0.1), @@ -771,7 +717,7 @@ mod tests { } } } - let q = Rc::new(agg( + let q = agg( vec![2], AggIntent::Count { accuracy: AccuracyTarget::EpsilonDelta { @@ -780,9 +726,8 @@ mod tests { }, }, metric_scan(&["job"]), - )); + ); let strategy = HydraGroupingStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, &DefaultAccuracyModel, &EqualSplitAllocator, &ExcessiveCollision, @@ -801,9 +746,9 @@ mod tests { // summary_candidates(Cardinality) = [Hll, Theta, Kmv] — none have a // modeled Hydra variant, so no candidate at all (not an error, just // an empty result, same conservatism as every other strategy here). - let q = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); + let q = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let strategy = HydraGroupingStrategy::default_cost_model(); + let strategy = HydraGroupingStrategy::default(); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); } @@ -817,9 +762,9 @@ mod tests { q: 0.99, accuracy: AccuracyTarget::Exact, }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); + let q = agg(vec![2], intent, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let strategy = HydraGroupingStrategy::default_cost_model(); + let strategy = HydraGroupingStrategy::default(); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); } @@ -828,60 +773,29 @@ mod tests { fn exact_mergeable_intent_has_no_hydra_candidate() { // Sum's exact accumulator has no candidate summary families at all // (summary_candidates only covers approximate-capable intents). - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); - let strategy = HydraGroupingStrategy::default_cost_model(); + let strategy = HydraGroupingStrategy::default(); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); } #[test] fn does_not_match_a_multi_intent_or_having_aggregate() { - let strategy = HydraGroupingStrategy::default_cost_model(); - - let multi = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }); + let strategy = HydraGroupingStrategy::default(); + + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&multi); assert!(!strategy.matches(&target)); assert!(strategy.replacements(&target).is_empty()); } - - /// A custom `CostModel` doesn't change *which* candidate is offered — - /// only which sketch candidate `realizations_for_intent` itself would - /// have ranked first, and how that candidate's own params are sized — - /// same guarantee `SketchAlgorithmStrategy` makes for its own candidates. - struct PreferDDSketch; - impl CostModel for PreferDDSketch { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - let mut v = candidates.to_vec(); - if let Some(pos) = v.iter().position(|k| *k == SketchAlgorithm::DDSketch) { - let dd = v.remove(pos); - v.insert(0, dd); - } - v - } - } - - #[test] - fn custom_cost_model_cannot_enable_unproven_hydra_kll() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let target = TargetSubDAG::new(&q); - let custom = PreferDDSketch; - let replacements = HydraGroupingStrategy::new(&custom).replacements(&target); - assert!(replacements.is_empty(), "{replacements:?}"); - } } diff --git a/crates/logical-optimizer/src/pass1/logical_candidates.rs b/crates/logical-optimizer/src/pass1/logical_candidates.rs new file mode 100644 index 000000000..4a195f4f2 --- /dev/null +++ b/crates/logical-optimizer/src/pass1/logical_candidates.rs @@ -0,0 +1,782 @@ +//! Pass 1 local alternatives over the unified logical IR. +//! +//! Alternatives are nominal realization descriptors attached to their original +//! target, not ranked plans or accuracy certificates. Workload composition and +//! physical planning consume this inventory later; empirical models belong to +//! selection. The legacy search API remains until planner cutover. +use std::collections::{BTreeMap, HashMap, HashSet}; +use std::rc::Rc; + +use asap_types::ir::operator::{AggIntent, Reduction}; +use asap_types::ir::properties::summary_coverage::{CoverageRegion, SummaryCoverage}; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::Schema; +use asap_types::ir::schema::{ + EntityIdentity, ExactKind, ExactParams, FieldDataType, GroupingStrategy, + NonNegativeWeightProof, SketchAlgorithm, SketchKind, SketchStatistic, SummaryInputExpr, + SummaryUpdate, WeightDomain, +}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, QueryRoot, SchemaDerivationError}; +use asap_types::types::AccuracyTarget; +use thiserror::Error; + +use crate::pass1::replacement::{ + accuracy_budget, accuracy_target, default_size_params, summary_candidates, Realization, +}; + +/// All local realizations of one single-measure aggregate. The target retains +/// source, grouping, filters, input expressions and evaluation context. +#[derive(Debug, Clone)] +pub struct LocalLogicalTarget { + pub target: Rc, + pub alternatives: Vec, + /// Per alternative, the target directly beneath this one that it + /// absorbs: its summary reads that target's input, so that target is + /// not computed and has no choice of its own (#509 whole-expression + /// realization). `None`: the alternative reads this target's input. + pub absorbs: Vec>, +} + +/// Compact Pass 1 inventory; roots and nested producer dependencies are retained. +#[derive(Debug, Clone)] +pub struct LocalLogicalCandidates { + pub roots: Vec<(Id, QueryRoot)>, + pub targets: Vec, +} + +#[derive(Debug, Error)] +pub enum LogicalCandidateError { + #[error(transparent)] + Structure(#[from] SchemaDerivationError), + #[error("logical candidate input already has assigned execution timing")] + AssignedTiming, + #[error("approximate accuracy requires finite positive epsilon and delta in (0, 1)")] + InvalidAccuracy, + #[error("choice must name one listed alternative per target")] + InvalidChoice, + #[error("unsupported local realization: {0}")] + Unsupported(&'static str), +} + +/// Enumerate exact and summary choices in stable catalog order, without ranking +/// or empirical assessment. Parameters are candidate dimensions, not a claim +/// that a deployment meets the request's accuracy requirement. +pub fn local_realizations_for_intent( + intent: &AggIntent, +) -> Result, LogicalCandidateError> { + let mut choices = vec![Realization::PassThrough]; + let exact = match intent { + AggIntent::Count { .. } => Some((ExactKind::Count, ExactParams::Count)), + AggIntent::Sum { .. } => Some((ExactKind::Sum, ExactParams::Sum)), + AggIntent::Min { .. } => Some((ExactKind::Min, ExactParams::Min)), + AggIntent::Max { .. } => Some((ExactKind::Max, ExactParams::Max)), + AggIntent::Rate => Some((ExactKind::Rate, ExactParams::Rate)), + AggIntent::IRate => Some((ExactKind::IRate, ExactParams::IRate)), + AggIntent::Increase => Some((ExactKind::Increase, ExactParams::Increase)), + _ => None, + }; + if let Some((kind, params)) = exact { + choices.push(Realization::ExactAggregate { kind, params }); + } + if let Some(target) = accuracy_target(intent) { + if *target != AccuracyTarget::Exact { + let (epsilon, delta) = accuracy_budget(target); + if !epsilon.is_finite() + || epsilon <= 0.0 + || !delta.is_finite() + || !(0.0..1.0).contains(&delta) + || delta == 0.0 + { + return Err(LogicalCandidateError::InvalidAccuracy); + } + for algorithm in summary_candidates(intent) { + choices.push(Realization::Sketch(SketchKind::new( + algorithm.clone(), + default_size_params(algorithm.clone(), intent, epsilon, delta), + ))); + } + } + } + Ok(choices) +} + +/// Discover single-measure targets, including operator plans read by scalar roots. +/// Multi-measure aggregates remain intact pending a semantics-preserving split. +pub fn enumerate_local_logical_candidates( + roots: Vec<(Id, QueryRoot)>, +) -> Result, LogicalCandidateError> { + let mut seen = HashSet::new(); + let mut targets = Vec::new(); + // Consumers of each node: the distinct nodes reading it, plus roots. + let mut consumers: HashMap<*const OperatorNode, usize> = HashMap::new(); + for (_, root) in &roots { + root.validate_structure()?; + let operators = match root { + QueryRoot::Operator(node) => vec![node], + QueryRoot::Scalar(expr) => expr.operator_refs(), + }; + for root in operators { + *consumers.entry(Rc::as_ptr(root)).or_default() += 1; + for node in OperatorNode::reachable(root) { + if !seen.insert(Rc::as_ptr(&node)) { + continue; + } + for child in node.children() { + *consumers.entry(Rc::as_ptr(child)).or_default() += 1; + } + if node.timing.is_some() { + return Err(LogicalCandidateError::AssignedTiming); + } + if let Some(NonASAPOp::Aggregate { measures, .. }) = node.non_asap() { + if let [intent] = measures.as_slice() { + let alternatives = local_realizations_for_intent(intent)?; + targets.push(LocalLogicalTarget { + absorbs: vec![None; alternatives.len()], + alternatives, + target: node, + }); + } + } + } + } + } + add_whole_expression_alternatives(&mut targets, &consumers); + Ok(LocalLogicalCandidates { roots, targets }) +} + +/// Whole-expression top-k (#509 Example 1's Q2): a top-k over a per-item sum +/// or count is realized as one heap sketch over the inner aggregate's input, +/// keyed by the ranked item and weighted by the summed value, instead of a +/// sketch over the inner aggregate's exact result. The decision is the +/// legacy keyed-additive rule's. Offered only when the inner target has no +/// other consumer, so absorbing it removes its work. +fn add_whole_expression_alternatives( + targets: &mut [LocalLogicalTarget], + consumers: &HashMap<*const OperatorNode, usize>, +) { + let position: HashMap<_, _> = targets + .iter() + .enumerate() + .map(|(i, t)| (Rc::as_ptr(&t.target), i)) + .collect(); + for target in targets.iter_mut() { + let Some(NonASAPOp::Aggregate { child, .. }) = target.target.non_asap() else { + continue; + }; + let Some(&inner) = position.get(&Rc::as_ptr(child)) else { + continue; + }; + if consumers.get(&Rc::as_ptr(child)) != Some(&1) + || whole_expression_input(&target.target).is_none() + { + continue; + } + let heaps: Vec<_> = target + .alternatives + .iter() + .filter(|a| { + matches!(a, Realization::Sketch(kind) if matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + )) + }) + .cloned() + .collect(); + for heap in heaps { + target.alternatives.push(heap); + target.absorbs.push(Some(inner)); + } + } +} + +/// The input and update of a whole-expression top-k over `target`'s inner +/// aggregate, by the legacy keyed-additive rule, or `None` when it does not +/// apply. Rows that carry the full series identity rank it as a column, as +/// [`summary_update`] does. +fn whole_expression_input(target: &OperatorNode) -> Option<(Rc, SummaryUpdate)> { + use crate::pass1::replacement::{ + realize_keyed_additive_summary_input, PhysicalSummaryInputRuleResult, + }; + let Some(NonASAPOp::Aggregate { + child, + reduction, + measures, + filters, + having: None, + .. + }) = target.non_asap() + else { + return None; + }; + let ([intent @ AggIntent::TopK { .. }], true) = (measures.as_slice(), filters.is_empty()) + else { + return None; + }; + // The rule reads only the family's algorithm, and rejects Count-Min over + // signed weights. Ask with CountSketch so one update serves both heap + // sketches; Stage 3 decides whether Count-Min is admissible, as it does + // for every other Count-Min candidate. + let family = FieldDataType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + asap_types::ir::schema::SketchParams::CountSketchWithHeap { + width: 1, + depth: 1, + heap_size: 1, + }, + ), + GroupingStrategy::default(), + ); + let PhysicalSummaryInputRuleResult::Realized(realized) = + realize_keyed_additive_summary_input(intent, &family, reduction, child) + else { + return None; + }; + let mut input = realized.input; + let per_series = matches!( + child.non_asap(), + Some(NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + .. + }) + ); + if per_series && realized.child.schema.has_promql_series_identity() { + input.item = Some(SummaryInputExpr::Column(ColumnRef::Named( + asap_types::ir::schema::PROMQL_SERIES_IDENTITY.into(), + ))); + } else if realized.child.schema.closed { + // An encoded label set needs an open PromQL schema. + return None; + } + Some((realized.child, input)) +} + +/// Number of whole-workload candidates: one per choice of an alternative for +/// every target, where a target absorbed by the alternative above it takes +/// only its pass-through (it is not computed). Saturates rather than +/// overflowing. +pub fn combination_count(inventory: &LocalLogicalCandidates) -> usize { + completions(inventory, &[]) +} + +/// Valid choices extending `prefix` (choices for the first `prefix.len()` +/// targets); 0 when `prefix` is invalid. A target and the target its +/// alternatives absorb are counted together. +fn completions(inventory: &LocalLogicalCandidates, prefix: &[usize]) -> usize { + let targets = &inventory.targets; + let mut paired = vec![false; targets.len()]; + let mut total = 1usize; + for (t, target) in targets.iter().enumerate() { + let Some(u) = target.absorbs.iter().flatten().next().copied() else { + continue; + }; + paired[t] = true; + paired[u] = true; + let absorbing = |c: usize| target.absorbs[c].is_some(); + let own = target.alternatives.len(); + let inner = targets[u].alternatives.len(); + let absorbing_count = (0..own).filter(|&c| absorbing(c)).count(); + let options = match (prefix.get(t), prefix.get(u)) { + (Some(&c), Some(&d)) => usize::from(!absorbing(c) || d == 0), + (Some(&c), None) if absorbing(c) => 1, + (Some(_), None) => inner, + (None, Some(0)) => own, + (None, Some(_)) => own - absorbing_count, + (None, None) => (own - absorbing_count) + .saturating_mul(inner) + .saturating_add(absorbing_count), + }; + total = total.saturating_mul(options); + } + for (j, target) in targets.iter().enumerate().skip(prefix.len()) { + if !paired[j] { + total = total.saturating_mul(target.alternatives.len()); + } + } + total +} + +/// The first `max` choices in enumeration order: mixed radix, the last target +/// varying fastest, skipping choices for a target its outer choice absorbs. +/// `choice[i]` indexes `inventory.targets[i].alternatives`. +pub fn enumerate_choices( + inventory: &LocalLogicalCandidates, + max: usize, +) -> Vec> { + let count = combination_count(inventory).min(max); + let mut choices = Vec::with_capacity(count); + let mut choice = vec![0; inventory.targets.len()]; + while choices.len() < count { + if completions(inventory, &choice) == 1 { + choices.push(choice.clone()); + } + for (digit, target) in choice.iter_mut().zip(&inventory.targets).rev() { + *digit += 1; + if *digit < target.alternatives.len() { + break; + } + *digit = 0; + } + } + choices +} + +/// Position of `choice` in [`enumerate_choices`] order: the valid choices +/// that precede it. +pub fn choice_index(inventory: &LocalLogicalCandidates, choice: &[usize]) -> usize { + let mut index = 0usize; + let mut prefix = Vec::with_capacity(choice.len()); + for &digit in choice { + for smaller in 0..digit { + prefix.push(smaller); + index = index.saturating_add(completions(inventory, &prefix)); + prefix.pop(); + } + prefix.push(digit); + } + index +} + +/// The targets alternative `c` of target `t` reads directly: those beneath +/// `t`, except one it absorbs, whose own targets beneath it are read instead. +/// `beneath` is [`nested_targets`]. +pub fn read_targets( + inventory: &LocalLogicalCandidates, + beneath: &[Vec], + t: usize, + c: usize, +) -> Vec { + match inventory.targets[t].absorbs[c] { + None => beneath[t].clone(), + Some(u) => { + let mut read: Vec<_> = beneath[t].iter().copied().filter(|&v| v != u).collect(); + read.extend(&beneath[u]); + read.sort_unstable(); + read.dedup(); + read + } + } +} + +/// For each target, the targets directly beneath it: reachable from its input +/// without passing through another target. +pub fn nested_targets(inventory: &LocalLogicalCandidates) -> Vec> { + let position: HashMap<_, _> = inventory + .targets + .iter() + .enumerate() + .map(|(i, t)| (Rc::as_ptr(&t.target), i)) + .collect(); + inventory + .targets + .iter() + .map(|target| { + let mut found = Vec::new(); + let mut seen = HashSet::new(); + let mut stack: Vec<_> = target.target.children().into_iter().cloned().collect(); + while let Some(node) = stack.pop() { + if !seen.insert(Rc::as_ptr(&node)) { + continue; + } + match position.get(&Rc::as_ptr(&node)) { + Some(&index) => found.push(index), + None => stack.extend(node.children().into_iter().cloned()), + } + } + found.sort_unstable(); + found + }) + .collect() +} + +/// Build one whole-workload candidate (#509 Stage 1): `choice[i]` indexes +/// `inventory.targets[i].alternatives`. Each chosen non-pass-through target is +/// replaced by `SummaryAgg` followed by `SummaryEstimate` (sketch) or +/// `FinalizeExactAccumulator` (exact accumulator). Untouched sub-DAGs keep +/// their identity, so sharing between roots is preserved. +pub fn compose_logical_candidate( + inventory: &LocalLogicalCandidates, + choice: &[usize], +) -> Result, LogicalCandidateError> { + if choice.len() != inventory.targets.len() { + return Err(LogicalCandidateError::InvalidChoice); + } + let chosen = inventory + .targets + .iter() + .zip(choice) + .map(|(target, &index)| { + target + .alternatives + .get(index) + .map(|alternative| { + let absorbs = target.absorbs[index].is_some(); + (Rc::as_ptr(&target.target), (alternative, absorbs)) + }) + .ok_or(LogicalCandidateError::InvalidChoice) + }) + .collect::, _>>()?; + let mut memo = HashMap::new(); + inventory + .roots + .iter() + .map(|(id, root)| { + let root = match root { + QueryRoot::Operator(node) => { + QueryRoot::Operator(rewrite(node, &chosen, &mut memo)?) + } + QueryRoot::Scalar(expr) => { + for node in expr.operator_refs() { + rewrite(node, &chosen, &mut memo)?; + } + QueryRoot::Scalar( + expr.map_operator_refs(&mut |node| memo[&Rc::as_ptr(node)].clone()), + ) + } + }; + Ok((id.clone(), root)) + }) + .collect() +} + +type Memo = HashMap<*const OperatorNode, Rc>; +/// Each target's chosen alternative, and whether it absorbs the target beneath. +type Chosen<'a> = HashMap<*const OperatorNode, (&'a Realization, bool)>; + +fn rewrite( + node: &Rc, + chosen: &Chosen<'_>, + memo: &mut Memo, +) -> Result, LogicalCandidateError> { + if let Some(done) = memo.get(&Rc::as_ptr(node)) { + return Ok(done.clone()); + } + for child in node.children() { + rewrite(child, chosen, memo)?; + } + let changed = node + .children() + .iter() + .any(|child| !Rc::ptr_eq(child, &memo[&Rc::as_ptr(child)])); + let rebuilt = match chosen.get(&Rc::as_ptr(node)) { + Some((realization, absorbs)) if **realization != Realization::PassThrough => { + realize(node, realization, *absorbs, memo)? + } + _ if changed => Rc::new(node.map_children(|child| memo[&Rc::as_ptr(child)].clone())?), + _ => node.clone(), + }; + memo.insert(Rc::as_ptr(node), rebuilt.clone()); + Ok(rebuilt) +} + +fn realize( + target: &OperatorNode, + realization: &Realization, + absorbs: bool, + memo: &Memo, +) -> Result, LogicalCandidateError> { + let Some(NonASAPOp::Aggregate { + child, + reduction, + measures, + filters, + having, + .. + }) = target.non_asap() + else { + return Err(LogicalCandidateError::Unsupported( + "target is not an aggregate", + )); + }; + let [intent] = measures.as_slice() else { + return Err(LogicalCandidateError::Unsupported( + "multi-measure aggregate", + )); + }; + if !filters.is_empty() || having.is_some() { + return Err(LogicalCandidateError::Unsupported( + "filtered or HAVING aggregate", + )); + } + let whole = + match absorbs { + true => Some(whole_expression_input(target).ok_or( + LogicalCandidateError::Unsupported("whole-expression top-k input"), + )?), + false => None, + }; + let child = match &whole { + Some((input, _)) => memo[&Rc::as_ptr(input)].clone(), + None => memo[&Rc::as_ptr(child)].clone(), + }; + let (family, query) = match realization { + Realization::ExactAggregate { kind, params } => ( + FieldDataType::ExactAggregate(kind.clone(), params.clone()), + None, + ), + Realization::Sketch(kind) => ( + FieldDataType::Sketch(kind.clone(), GroupingStrategy::default()), + Some(statistic(intent)?), + ), + _ => return Err(LogicalCandidateError::Unsupported("summary family")), + }; + let input = match whole { + Some((_, update)) => update, + None => summary_update(intent, &family, reduction, &child.schema)?, + }; + let state = OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: child.clone(), + family, + input, + reduction: reduction.clone(), + grouping: GroupingStrategy::default(), + filter: None, + }))?; + // Whole-source coverage is declared, not proven: Pass 1 trusts that the + // state holds every observation of its source that reaches it (#570). + let coverage = SummaryCoverage { + source: single_source(&child)?, + regions: vec![CoverageRegion { + time_ms: None, + population: BTreeMap::new(), + }], + }; + let state = Rc::new(state.with_coverage(coverage)?); + let evaluation = match query { + Some(query) => ASAPOp::SummaryEstimate { + summary_input: state, + query, + }, + None => ASAPOp::FinalizeExactAccumulator { child: state }, + }; + Ok(OperatorNode::new_shared(Operator::ASAP(evaluation))?) +} + +fn single_source( + node: &Rc, +) -> Result { + let mut sources = Vec::new(); + for node in OperatorNode::reachable(node) { + if let Some(NonASAPOp::Scan { source, .. }) = node.non_asap() { + if !sources.contains(source) { + sources.push(source.clone()); + } + } + } + match <[_; 1]>::try_from(sources) { + Ok([source]) => Ok(source), + Err(_) => Err(LogicalCandidateError::Unsupported( + "summary coverage needs exactly one source", + )), + } +} + +/// What each input row contributes, following the legacy realization rules: +/// heap sketches rank series identities, frequency sketches count values, and +/// every other family reads the measure's input column. +fn summary_update( + intent: &AggIntent, + family: &FieldDataType, + reduction: &Reduction, + child: &Schema, +) -> Result { + let algorithm = match family { + FieldDataType::Sketch(kind, _) => Some(kind.algorithm()), + _ => None, + }; + let weight = crate::pass1::replacement::summarised_input(intent, child) + .map_err(|_| LogicalCandidateError::Unsupported("input column outside child schema"))?; + Ok(match (intent, algorithm) { + (AggIntent::TopK { .. }, Some(_)) => { + // SQL rows carry no implicit series identity to rank; closed + // PromQL rows carry it as a column. + if child.closed && !child.has_promql_series_identity() { + return Err(LogicalCandidateError::Unsupported( + "Top-K item identity for closed schemas", + )); + } + if reduction.group_keys().is_some_and(|keys| keys.is_without()) { + return Err(LogicalCandidateError::Unsupported( + "Top-K partitions given by `without`", + )); + } + let excluding = reduction + .group_keys() + .into_iter() + .flat_map(|keys| keys.iter()) + .filter_map(|&index| child.fields.get(index)) + .map(crate::pass1::replacement::column_ref) + .collect(); + // Rows that carry the full series identity rank it as a column, + // the item form the runtime builds keyed summaries from. + let item = if child.has_promql_series_identity() { + SummaryInputExpr::Column(ColumnRef::Named( + asap_types::ir::schema::PROMQL_SERIES_IDENTITY.into(), + )) + } else { + SummaryInputExpr::EntityIdentity(EntityIdentity::PromqlLabelSet { excluding }) + }; + SummaryUpdate { + item: Some(item), + weight: SummaryInputExpr::Column(ColumnRef::SampleValue), + // Not proven non-negative; selection decides whether CMS is admissible. + weight_domain: WeightDomain::UnknownOrSigned, + } + } + (_, Some(SketchAlgorithm::UnivMon)) + | (AggIntent::Count { .. }, Some(SketchAlgorithm::Cms | SketchAlgorithm::CountSketch)) => { + SummaryUpdate { + item: Some(weight), + weight: SummaryInputExpr::Constant(1.0), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::UnitCount, + }, + } + } + _ => SummaryUpdate { + item: None, + weight, + weight_domain: WeightDomain::UnknownOrSigned, + }, + }) +} + +fn statistic(intent: &AggIntent) -> Result { + Ok(match intent { + AggIntent::Quantile { q, .. } => SketchStatistic::Quantile { q: *q }, + AggIntent::Cardinality { .. } => SketchStatistic::Cardinality, + AggIntent::FrequencyL2 { .. } => SketchStatistic::FrequencyL2, + AggIntent::FrequencyEntropy { .. } => SketchStatistic::FrequencyEntropy, + AggIntent::TopK { k, .. } => SketchStatistic::TopK { k: *k }, + AggIntent::Count { .. } => SketchStatistic::PointCount { + key: ColumnRef::SampleValue, + value: None, + }, + _ => return Err(LogicalCandidateError::Unsupported("sketch evaluation")), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::test_support::lower_promql; + + /// Every realization of a top-k returns the same selected-rows schema, so + /// an aggregate over a pass-through top-k composes like one over a sketch. + #[test] + fn aggregate_over_any_topk_realization_composes() { + let root = lower_promql( + "count(topk by (job) (10, sum_over_time(m[1m])))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }, + ); + let root = asap_types::ir::schema_support::with_promql_series_identity(&root).unwrap(); + let inventory = + enumerate_local_logical_candidates(vec![(0, QueryRoot::Operator(root))]).unwrap(); + let topk = inventory + .targets + .iter() + .position(|t| { + matches!(t.target.non_asap(), Some(NonASAPOp::Aggregate { measures, .. }) + if matches!(measures.as_slice(), [AggIntent::TopK { .. }])) + }) + .unwrap(); + let mut schemas = Vec::new(); + for (count, alternatives) in inventory.targets.iter().enumerate() { + if count == topk { + continue; + } + for outer in 0..alternatives.alternatives.len() { + for inner in 0..inventory.targets[topk].alternatives.len() { + let mut choice = vec![0; inventory.targets.len()]; + choice[count] = outer; + choice[topk] = inner; + compose_logical_candidate(&inventory, &choice) + .unwrap_or_else(|e| panic!("{choice:?}: {e}")); + } + } + } + for inner in 0..inventory.targets[topk].alternatives.len() { + let mut choice = vec![0; inventory.targets.len()]; + choice[topk] = inner; + let roots = compose_logical_candidate(&inventory, &choice).unwrap(); + let QueryRoot::Operator(root) = &roots[0].1 else { + panic!("operator root") + }; + let NonASAPOp::Aggregate { child, .. } = root.expect_non_asap() else { + panic!("count over top-k") + }; + schemas.push(child.schema.clone()); + } + assert!(schemas.windows(2).all(|w| w[0] == w[1]), "{schemas:#?}"); + } + /// A top-k over a per-series `sum_over_time` gets whole-expression heap + /// sketches that absorb the inner target; enumeration skips the absorbed + /// target's choices, and `choice_index` numbers choices in that order. + #[test] + fn whole_expression_topk_absorbs_the_inner_sum() { + let root = lower_promql( + "topk by (job) (10, sum_over_time(m[1m]))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }, + ); + let root = asap_types::ir::schema_support::with_promql_series_identity(&root).unwrap(); + let inventory = + enumerate_local_logical_candidates(vec![(0, QueryRoot::Operator(root))]).unwrap(); + let (topk, inner) = inventory + .targets + .iter() + .enumerate() + .find_map(|(t, target)| Some((t, target.absorbs.iter().flatten().next().copied()?))) + .expect("an absorbing alternative"); + let absorbing = inventory.targets[topk].absorbs.iter().flatten().count(); + assert_eq!(absorbing, 2, "Count-Min and CountSketch with heap"); + // (pass-through, CMS+heap, CountSketch+heap) × (raw, Sum acc) + 2. + let choices = enumerate_choices(&inventory, usize::MAX); + assert_eq!(choices.len(), 8); + assert_eq!(combination_count(&inventory), 8); + for (index, choice) in choices.iter().enumerate() { + assert_eq!(choice_index(&inventory, choice), index); + if inventory.targets[topk].absorbs[choice[topk]].is_some() { + assert_eq!(choice[inner], 0); + let roots = compose_logical_candidate(&inventory, choice).unwrap(); + let QueryRoot::Operator(root) = &roots[0].1 else { + panic!("operator root") + }; + assert!( + !OperatorNode::reachable(root) + .iter() + .any(|n| Rc::ptr_eq(n, &inventory.targets[inner].target)), + "the inner sum is not computed" + ); + } + } + } + + /// Approximate requests must retain the exact execution alternative too. + #[test] + fn approximate_count_keeps_exact_and_universal_choices() { + let choices = local_realizations_for_intent(&AggIntent::Count { + accuracy: AccuracyTarget::EpsilonDelta { + epsilon: 0.05, + delta: 0.01, + }, + }) + .unwrap(); + assert!(choices + .iter() + .any(|choice| matches!(choice, Realization::PassThrough))); + assert!(choices.iter().any(|choice| matches!( + choice, + Realization::ExactAggregate { + kind: ExactKind::Count, + .. + } + ))); + assert!(choices.iter().any(|choice| matches!(choice, Realization::Sketch(kind) if *kind.algorithm() == asap_types::ir::schema::SketchAlgorithm::UnivMon))); + } +} diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/logical-optimizer/src/pass1/maintained_population.rs similarity index 63% rename from crates/asap-aware-mapping/src/maintained_population.rs rename to crates/logical-optimizer/src/pass1/maintained_population.rs index 4764212f3..ca22ffbef 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/logical-optimizer/src/pass1/maintained_population.rs @@ -1,34 +1,29 @@ //! Shared maintained-population candidates over canonical relational IR. -use crate::replacement::{ +use crate::pass1::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; -use asap_types::post_asap::{ - maintained_population::*, ExecutionTiming, ResultGuarantee, SummaryExpr, SummaryNode, - ValueOperation, -}; -use asap_types::pre_asap::{ - any_measure_filtered, AggIntent, CompareOpKind, DataType, QueryExpr, Reduction, ScalarValue, - Schema, Source, -}; +use asap_types::ir::operator::maintained_population::*; +use asap_types::ir::operator::non_asap::any_measure_filtered; +use asap_types::ir::operator::{AggIntent, Reduction, Source}; +use asap_types::ir::properties::ResultGuarantee; +use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; +use asap_types::ir::schema::DataType; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; use std::rc::Rc; -fn plain(schema: Schema) -> Schema { - Schema::lifted(schema.fields, schema.time_index) -} - -fn strip_projection(mut root: &QueryExpr) -> &QueryExpr { - while let QueryExpr::Project { child, .. } = root { +fn strip_projection(mut root: &OperatorNode) -> &OperatorNode { + while let Some(NonASAPOp::Project { child, .. }) = root.non_asap() { root = child; } root } fn recognize( - root: &QueryExpr, -) -> Option<(MaintainedPopulation, PopulationStatistic, Rc)> { + root: &OperatorNode, +) -> Option<(MaintainedPopulation, PopulationStatistic, Rc)> { let root = strip_projection(root); - let (source, grouping, readout, value_column) = match root { - QueryExpr::Aggregate { + let (source, grouping, evaluation, value_column) = match root.non_asap()? { + NonASAPOp::Aggregate { child, reduction: Reduction::Reduce(grouping), measures, @@ -42,7 +37,7 @@ fn recognize( if any_measure_filtered(filters) { return None; } - let (col, readout) = match intent { + let (col, evaluation) = match intent { AggIntent::Quantile { q, col, .. } if q.is_finite() => { (*col, PopulationStatistic::Quantile { q: *q }) } @@ -52,29 +47,30 @@ fn recognize( AggIntent::Avg { col } => (*col, PopulationStatistic::Average), _ => return None, }; - let schema = child.output_schema().ok()?; + let schema = &child.schema; if col.is_some_and(|c| schema.fields.get(c).is_none()) { return None; } - (child, grouping, readout, col) + (child, grouping, evaluation, col) } - QueryExpr::Limit { - n, + NonASAPOp::Limit { + n: Some(n), offset: 0, child, + .. } => { - let QueryExpr::Sort { + let Some(NonASAPOp::Sort { child, keys, partition_by, - } = child.as_ref() + }) = child.non_asap() else { return None; }; let [key] = keys.as_slice() else { return None; }; - let QueryExpr::Column(col) = &key.expr else { + let ScalarExpr::Column(col) = &key.expr else { return None; }; if key.ascending { @@ -89,11 +85,11 @@ fn recognize( } _ => return None, }; - if let QueryExpr::Scan { + if let Some(NonASAPOp::Scan { source: Source::Table { .. }, schema, .. - } = source.as_ref() + }) = source.non_asap() { let value_column = value_column.or_else(|| { schema @@ -110,29 +106,29 @@ fn recognize( max_k: 0, quantiles: false, }; - if !schema.closed || !population.matches_input(source) { + if !schema.closed || !population.matches_node(source) { return None; } - return Some((population, readout, Rc::clone(source))); + return Some((population, evaluation, Rc::clone(source))); } // A bare PromQL selector carries the declared ingestion interval as a // temporal input scope. Membership must expire at that horizon; retain // the wrapper as the maintained input so validation can check agreement. - let (series_source, lookback_ms) = match source.as_ref() { - QueryExpr::TimeRange { range, child } => { + let (series_source, lookback_ms) = match source.non_asap() { + Some(NonASAPOp::TimeRange { range, child, .. }) => { let ms = u64::try_from(range.as_millis()).ok()?; if ms == 0 || std::time::Duration::from_millis(ms) != *range { return None; } (child.as_ref(), ms) } - other => (other, 300_000), + _ => (source.as_ref(), 300_000), }; - let QueryExpr::Scan { + let Some(NonASAPOp::Scan { source: Source::TimeSeries { metric }, predicates, schema, - } = series_source + }) = series_source.non_asap() else { return None; }; @@ -152,10 +148,13 @@ fn recognize( }; let mut matchers = Vec::new(); for predicate in predicates { - let QueryExpr::Compare { left, op, right } = predicate.0.as_ref() else { + let ScalarExpr::Compare { + left, op, right, .. + } = &predicate.0 + else { return None; }; - let (QueryExpr::Column(col), QueryExpr::Literal(ScalarValue::Utf8(value))) = + let (ScalarExpr::Column(col), ScalarExpr::Literal(ScalarValue::Utf8(value))) = (left.as_ref(), right.as_ref()) else { return None; @@ -194,46 +193,46 @@ fn recognize( max_k: 0, quantiles: false, }, - readout, + evaluation, Rc::clone(source), )) } -/// Workload-aware rule: compatible readouts share one retractable population. +/// Workload-aware rule: compatible evaluations share one retractable population. /// Deployments opt in by registering this strategy when they can maintain complete -/// population updates and price the maintenance/readout boundary. -/// The population is exact; max_k bounds the shared readout cache, not its members. +/// population updates and price the maintenance/evaluation boundary. +/// The population is exact; max_k bounds the shared evaluation cache, not its members. pub struct MaintainedPopulationStrategy { - roots: Vec>, + roots: Vec>, } impl MaintainedPopulationStrategy { - pub fn new(roots: &[Rc]) -> Self { + pub fn new(roots: &[Rc]) -> Self { Self { roots: roots.to_vec(), } } - pub fn candidate(&self, root: &Rc) -> Option> { - if let QueryExpr::Project { + pub fn candidate(&self, root: &Rc) -> Option> { + if let Some(NonASAPOp::Project { cols, qualifier, child, - } = root.as_ref() + }) = root.non_asap() { let child = self.candidate(child)?; - return Some(Rc::new(SummaryNode { - guarantee: child.guarantee.clone(), - schema: plain(root.output_schema().ok()?), - expr: SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { + let guarantee = child.guarantee.clone(); + return Some(Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Project { cols: cols.clone(), qualifier: qualifier.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - })); + child, + }), + root.schema.clone(), + ) + .with_guarantee(guarantee), + )); } - let (mut population, readout, source) = recognize(root)?; + let (mut population, evaluation, source) = recognize(root)?; let identity = population.clone(); for other in self.roots.iter().chain(std::iter::once(root)) { if let Some((p, r, _)) = recognize(other) { @@ -250,36 +249,39 @@ impl MaintainedPopulationStrategy { } } } - let input_schema = plain(source.output_schema().ok()?); - let scan = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(source), - schema: input_schema.clone(), - guarantee: Some(ResultGuarantee::exact("source samples")), - }); - // Query time is only the initial layout: whether the population is - // retained at ingestion or rebuilt per query is its lifecycle choice - // (`SummaryMaintenanceLifecyclePlan::execution_timed_dag`). The readout - // and projection above it are query-time by construction. - let maintained = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: scan, - operation: ValueOperation::MaintainPopulation { population }, - timing: ExecutionTiming::QueryTime, - }, - schema: input_schema, - guarantee: Some(ResultGuarantee::exact( + let input_schema = source.schema.clone(); + // The source node itself is the maintained input (a non-ASAP node + // keeps its derived schema), kept with its exact guarantee. + let scan = Rc::new( + source + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("source samples"))), + ); + let maintained = std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::MaintainPopulation { + child: scan, + population, + }), + input_schema, + ) + .with_guarantee(Some(ResultGuarantee::exact( "exact members under the declared population semantics", - )), - }); - Some(Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: maintained, - operation: ValueOperation::ReadPopulation { readout }, - timing: ExecutionTiming::QueryTime, - }, - schema: plain(root.output_schema().ok()?), - guarantee: Some(ResultGuarantee::exact("exact current-population readout")), - })) + ))), + ); + Some(std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::EvaluatePopulation { + child: maintained, + evaluation, + }), + root.schema.clone(), + ) + .with_guarantee(Some(ResultGuarantee::exact( + "exact current-population evaluation", + ))), + )) } } impl ReplacementStrategy for MaintainedPopulationStrategy { @@ -290,10 +292,10 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { self.candidate(target.root) .map(|node| ReplacementSubDAG { strategy: "MaintainedPopulationStrategy", - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), provenance: ReplacementProvenance::SummaryRealization, rationale: - "share an exact maintained population across compatible aggregate readouts" + "share an exact maintained population across compatible aggregate evaluations" .into(), }) .into_iter() @@ -305,10 +307,28 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { mod tests { use super::*; use crate::test_support::lower_promql; - use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_sub_dags}; + use asap_types::ir::cse::share_common_sub_dags; + use asap_types::ir::export::compile_physical_asap_dag as export_timed; + use asap_types::ir::properties::timing::{ + apply_materialization_timings, MaterializationAssignment, TimingMemo, + }; + + /// Time `root` under the default materialization assignment (which runs the + /// data-state / population-contract validation) and export it. + fn compile_physical_asap_dag(root: &Rc) -> Result<(), String> { + root.validate_structure().map_err(|e| e.to_string())?; + let timed = apply_materialization_timings( + root, + &MaterializationAssignment::all_query_time(), + &mut TimingMemo::new(), + ) + .map_err(|e| format!("{e:?}"))?; + export_timed(&timed).map_err(|e| format!("{e:?}"))?; + Ok(()) + } - fn lower(q: &str) -> Rc { - Rc::new(lower_promql(q, asap_types::types::AccuracyTarget::Exact)) + fn lower(q: &str) -> Rc { + lower_promql(q, asap_types::types::AccuracyTarget::Exact) } // Instant scalar aggregations share the same retractable series population. @@ -329,11 +349,11 @@ mod tests { let candidate = rule .candidate(&root) .expect("current-series rule candidate"); - compile_post_asap_dag(&candidate).expect("typed post-ASAP DAG"); + compile_physical_asap_dag(&candidate).expect("typed post-ASAP DAG"); } } - // Different readout parameters retain one shared maintenance producer in the DAG. + // Different evaluation parameters retain one shared maintenance producer in the DAG. #[test] fn quantiles_and_topk_share_a_planner_population() { let roots: Vec<_> = [ @@ -357,7 +377,7 @@ mod tests { .target_subdag_candidates() .flat_map(|g| &g.candidates) .any(|c| c.strategy == "MaintainedPopulationStrategy")); - let plans = share_common_summary_sub_dags( + let plans = share_common_sub_dags( roots .iter() .enumerate() @@ -366,19 +386,11 @@ mod tests { ); let mut producers = Vec::new(); for (_, plan) in &plans { - compile_post_asap_dag(plan).unwrap(); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::ReadPopulation { .. }, - .. - } = &plan.expr - else { - panic!("missing typed readout") + compile_physical_asap_dag(plan).unwrap(); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &plan.operator else { + panic!("missing typed evaluation") }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { panic!("missing maintained population") }; @@ -404,14 +416,10 @@ mod tests { let (p, _, _) = recognize(&roots[0]).unwrap(); assert!(matches!(p.input, PopulationInput::CurrentSeries(ref s) if s.grouping.is_empty())); let candidate = strategy.candidate(&roots[0]).unwrap(); - let SummaryExpr::ValueOperation { child, .. } = &candidate.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &candidate.operator else { unreachable!() }; - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &child.expr - else { + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &child.operator else { unreachable!() }; assert_eq!(population.max_k, 5); @@ -437,59 +445,51 @@ mod tests { assert_eq!(p.grouping, ["instance"]); assert_eq!(p.matchers[0].operation, CurrentSeriesMatch::Regex); } - // Population timing is a lifecycle choice: a retained or rebuilt - // population both validate, while its readout must stay at query time. + // Population timing is a materialization choice: a retained or rebuilt + // population both validate, while its evaluation must stay at query time. #[test] fn population_timing_is_not_structural() { let root = lower("topk(5,a)"); let candidate = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)) .candidate(&root) .unwrap(); - let with_timings = |population: ExecutionTiming, readout: ExecutionTiming| { + use asap_types::ir::properties::ExecutionTiming; + let with_timings = |population: ExecutionTiming, evaluation: ExecutionTiming| { let mut node = (*candidate).clone(); - let SummaryExpr::ValueOperation { child, timing, .. } = &mut node.expr else { - unreachable!() - }; - *timing = readout; - let SummaryExpr::ValueOperation { timing, .. } = &mut Rc::make_mut(child).expr else { + node.timing = Some(evaluation); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &mut node.operator + else { unreachable!() }; - *timing = population; - compile_post_asap_dag(&Rc::new(node)) + Rc::make_mut(child).timing = Some(population); + compile_physical_asap_dag(&Rc::new(node)) }; use ExecutionTiming::{IngestionTime, QueryTime}; assert!(with_timings(IngestionTime, QueryTime).is_ok()); assert!(with_timings(QueryTime, QueryTime).is_ok()); assert!(with_timings(IngestionTime, IngestionTime).is_err()); } - // A readout cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. + // A evaluation cannot reinterpret arbitrary rows as maintained state or exceed its producer's contract. #[test] fn malformed_population_dags_fail_closed() { let root = lower("topk(5,a)"); let strategy = MaintainedPopulationStrategy::new(std::slice::from_ref(&root)); let candidate = strategy.candidate(&root).unwrap(); + compile_physical_asap_dag(&candidate).expect("the unmodified candidate is legal"); let mut bad = (*candidate).clone(); - let SummaryExpr::ValueOperation { operation, .. } = &mut bad.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { evaluation, .. }) = &mut bad.operator + else { unreachable!() }; - *operation = ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { k: 6 }, - }; - assert!(compile_post_asap_dag(&Rc::new(bad.clone())).is_err()); - let SummaryExpr::ValueOperation { - child, operation, .. - } = &mut bad.expr + *evaluation = PopulationStatistic::TopK { k: 6 }; + assert!(compile_physical_asap_dag(&Rc::new(bad.clone())).is_err()); + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, evaluation }) = &mut bad.operator else { unreachable!() }; - *operation = ValueOperation::ReadPopulation { - readout: PopulationStatistic::TopK { k: 5 }, - }; + *evaluation = PopulationStatistic::TopK { k: 5 }; let producer = Rc::make_mut(child); - let SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { population }, - .. - } = &mut producer.expr + let Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) = &mut producer.operator else { unreachable!() }; @@ -497,6 +497,6 @@ mod tests { unreachable!() }; spec.metric = "b".into(); - assert!(compile_post_asap_dag(&Rc::new(bad)).is_err()); + assert!(compile_physical_asap_dag(&Rc::new(bad)).is_err()); } } diff --git a/crates/logical-optimizer/src/pass1/mod.rs b/crates/logical-optimizer/src/pass1/mod.rs new file mode 100644 index 000000000..bf32ceee4 --- /dev/null +++ b/crates/logical-optimizer/src/pass1/mod.rs @@ -0,0 +1,13 @@ +//! Pass 1: local logical alternatives for each target sub-DAG. Strategies +//! propose rewrites and summary realizations; a candidate is pruned only when +//! it is provably invalid. + +pub mod exact_composition; +pub mod explanation; +pub(crate) mod function_rules; +pub mod grouping; +pub mod logical_candidates; +pub mod maintained_population; +pub mod replacement; +pub mod rewrite; +pub mod rollup; diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/logical-optimizer/src/pass1/replacement.rs similarity index 56% rename from crates/asap-aware-mapping/src/replacement.rs rename to crates/logical-optimizer/src/pass1/replacement.rs index 44ffef189..c77e3f766 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/logical-optimizer/src/pass1/replacement.rs @@ -3,23 +3,23 @@ //! under "Key concepts (not yet implemented)", implemented for real (issue //! #251, part of #33). //! -//! ## One step, not two: `SketchAlgorithmStrategy::replacements()` decides *and* builds +//! ## One step, not two: `ASAPStrategies::replacements()` decides *and* builds //! -//! For a bindable `Aggregate`, `SketchAlgorithmStrategy::replacements()` is the +//! For a bindable `Aggregate`, `ASAPStrategies::replacements()` is the //! single place this crate both decides what an `AggIntent` may become and //! turns each of those candidates into a real, executable //! [`ReplacementSubDAG`]: //! //! 1. **Decide**: [`realizations_for_intent`] enumerates every valid -//! [`Realization`] for the target's intent — exhaustive, and ranked -//! most-preferred-first via a [`CostModel`] (candidate sketch family/kind, -//! already sized to the target's own accuracy target: `Realization::Sketch`'s -//! `params` are the output of inverting that accuracy target through -//! `CostModel::size_params`, not a placeholder filled in later). +//! [`Realization`] for the target's intent — exhaustive, in a static +//! order (candidate sketch family/kind, already sized to the target's own +//! accuracy target: `Realization::Sketch`'s `params` are the output of +//! inverting that accuracy target through the analytical estimators, not a +//! placeholder filled in later). //! 2. **Build**: for each candidate in that list, [`construct_summary`] //! mechanically turns the already-decided `(kind, params)` into a real -//! [`SummaryNode`] — derives the child schema, resolves the summarized -//! column, builds the readout query, recurses into the child (via +//! [`OperatorNode`] — derives the child schema, resolves the summarized +//! column, builds the evaluation query, recurses into the child (via //! [`realize_child`], so a nested aggregate gets its own //! independent enumeration, never the outer target's forced choice), and //! assembles the `SummaryAgg`/`SummaryEstimate` node. @@ -31,22 +31,22 @@ //! has to run regardless of how `(kind, params)` were chosen, so it lives //! directly inside the one method that needs it. //! -//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`QueryExpr`] node that is a +//! - [`TargetSubDAG`] — a reference to a pre-ASAP [`OperatorNode`] that is a //! candidate for replacement, plus how many places in the workload already //! reference it (its `consumer_count`) — the one piece of cross-node //! context [`SharedSubDAGStrategy`] needs that a bare node reference alone //! doesn't carry. //! - [`ReplacementSubDAG`] — one candidate replacement for a `TargetSubDAG`: -//! either a fully bound [`SummaryNode`] or a pre-ASAP [`QueryExpr`] rewrite +//! either a fully bound summary sub-DAG or a pre-ASAP logical rewrite //! (still logical, structurally different from the target but semantically //! equivalent) — see [`Replacement`] — plus a human-readable `rationale`. //! - [`ReplacementStrategy`] — `matches` + `replacements`, the same -//! extension-point shape [`CostModel`] and [`Matcher`] already use in this +//! extension-point shape `CostModel` and [`Matcher`] already use in this //! crate: a new replacement source is a new `impl ReplacementStrategy`, not //! a restructuring of this trait or of any existing strategy. `replacements` //! is **exhaustive, not ranked, not filtered** — reporting "every valid //! candidate" is core's job; picking the best one is left to the caller. -//! [`crate::explanation`] (issue #257) is this trait's own downstream +//! [`crate::pass1::explanation`] (issue #257) is this trait's own downstream //! consumer, not a second extension point: it explains why a replacement //! exists as a pure view over the candidates strategies registered here //! already produced, rather than re-deriving that explanation with a rule @@ -54,11 +54,9 @@ //! //! A caller may inspect local replacements, but taking the first candidate //! does not establish a compatible workload plan or physical deployability. -//! For Planner-owned logical selection, call [`CandidateLogicalASAPDAGs::global_selection`] +//! For Planner-owned logical selection, call `candidate_selection::global_selection` //! once and [`GlobalSelection::assemble_selected_dag`] for each wanted query -//! root. Alternatively, use the summary-maintenance-lifecycle-aware helpers -//! when Planner should also compare maintenance against raw recomputation. -//! Physical binding, deployment, and execution remain downstream. +//! root. Physical binding, deployment, and execution remain downstream. //! //! Internally, [`realize_child`] and [`realize_one`] may take a preferred local //! realization while constructing or costing a candidate. That local operation @@ -72,21 +70,21 @@ //! //! ## The two strategies, and why these two //! -//! - [`SketchAlgorithmStrategy`] wraps [`realizations_for_intent`]'s exhaustive, +//! - [`ASAPStrategies`] wraps [`realizations_for_intent`]'s exhaustive, //! ranked list directly: for the same bindable-`Aggregate` shape this crate //! binds (single intent, no `HAVING`), every entry becomes its own bound //! candidate. //! - [`SharedSubDAGStrategy`] wraps -//! `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing decision. +//! `asap_types::ir::cse::share_common_sub_dags`'s sharing decision. //! Wherever a [`TargetSubDAG`] already has two or more consumers (i.e. //! `share_common_sub_dags` already collapsed two or more workload -//! locations onto the same `Rc` — [`discover_targets`] below +//! locations onto the same `Rc` — [`discover_targets`] below //! does the identical workload-wide discovery for [`search_workload_with`]; //! this module's own tests reuse the same dedup logic to build realistic //! fixtures), it reports the two-way candidate CSE's own detection pass //! deliberately declines to pick between on its own: build once and share //! the already-interned sub-DAG, or build it independently at each -//! consumer. [`crate::cost_model::CostModel::cse_share_decision`] is where +//! consumer. `cost_model::CostModel::cse_share_decision` is where //! that choice actually gets made *today* (a fixed comparison, not a //! search) — this strategy exposes the same two-way choice as an explicit, //! inspectable pair of candidates instead of a cost model's already-decided @@ -98,7 +96,7 @@ //! unchanged.** Same inputs still produce the same exhaustive, ranked //! list — only its home moved (from a separate `implementation` module //! into this one) and its own visibility dropped to module-private, since -//! [`SketchAlgorithmStrategy`] is now its only caller. +//! [`ASAPStrategies`] is now its only caller. //! //! ## Workload-wide search — merged in from the former `search.rs` (issue #252, part of #33) //! @@ -143,21 +141,21 @@ //! //! 1. **Per-target candidates, not flat plans.** [`TargetSubDAGCandidates`] //! stores the alternatives for one distinct [`TargetSubDAG`] (identified by -//! its own `Rc` pointer identity — the same currency -//! [`asap_types::pre_asap::cse::share_common_sub_dags`] already +//! its own `Rc` pointer identity — the same currency +//! [`asap_types::ir::cse::share_common_sub_dags`] already //! established across the workload) holding every //! [`ReplacementSubDAG`] alternative discovered for it. [`CandidateLogicalASAPDAGs`] is //! a collection of these groups, keyed by `TargetSubDAG` — a candidate -//! "plan" is never materialized as a distinct top-level `Rc` +//! "plan" is never materialized as a distinct top-level `Rc` //! at all; two logically-different overall choices at two different //! targets are just two different entries in two different groups, //! sharing every other node in the workload by construction (they *are* //! the same `Rc`s — nothing was copied to make a second "plan"). -//! 2. **Dedup by structural hash + `PartialEq`, reusing `pre_asap::cse`'s own -//! discipline.** [`asap_types::pre_asap::cse::structural_hash`] (made +//! 2. **Dedup by structural hash + `PartialEq`, reusing `ir::cse`'s own +//! discipline.** [`asap_types::ir::cse::structural_hash`] (made //! `pub` for exactly this reuse) is only ever a candidate-narrowing //! filter; [`TargetSubDAGCandidates::add_candidate`]'s actual duplicate check is -//! `QueryExpr`'s derived `PartialEq` — the same "hash is a filter, +//! `OperatorNode`'s derived `PartialEq` — the same "hash is a filter, //! `PartialEq` is the decision, no exceptions" rule `cse.rs`'s own //! "Correctness" section states and this module inherits rather than //! reinvents. See [`is_duplicate_rewrite`] for the one deliberate @@ -175,7 +173,7 @@ //! line above stands for: every `TargetSubDAG` this pass discovers is one //! iteration of that loop. It walks every workload root's whole DAG (the //! same **relational-skeleton** operator-child scope -//! `asap_types::pre_asap::cse::share_common_sub_dags` itself uses — see +//! `asap_types::ir::cse::share_common_sub_dags` itself uses — see //! that module's "Algorithm" section), discovering one `TargetSubDAG` per //! distinct `Rc` and a *real* `consumer_count`: how many operator-child //! positions anywhere in the workload reference that exact `Rc`, not just @@ -212,10 +210,10 @@ //! an alternative *for* the target just processed, not a new target of its //! own; see [`discover_new_descendant_targets`]) are scanned for pointers //! not already known, and any found become next round's frontier. Both shipped -//! strategies are idempotent in exactly this sense: [`SketchAlgorithmStrategy`] -//! produces terminal [`Replacement::Summary`] candidates (no `QueryExpr` -//! children to scan at all), and [`SharedSubDAGStrategy`]'s two -//! [`Replacement::Rewrite`] candidates both reuse the target's own +//! strategies are idempotent in exactly this sense: [`ASAPStrategies`] +//! produces terminal bound-summary [`Replacement::SubDAG`] candidates (no +//! logical-rewrite children to scan at all), and [`SharedSubDAGStrategy`]'s +//! two logical-rewrite [`Replacement::SubDAG`] candidates both reuse the target's own //! already-known child `Rc`s verbatim (`Rc::clone`/a shallow top-level //! `.clone()` — see that strategy's own doc). So for both, the frontier is //! always empty after round one: real workloads converge in exactly one @@ -235,8 +233,8 @@ //! //! ### Cost-based final selection — reusing `CostModel`, not a second interface //! -//! [`CandidateLogicalASAPDAGs::cost_sorted`] is the `sorted_by(cost_model)` step, and it -//! reuses this crate's existing [`CostModel`] trait rather than inventing a +//! `candidate_selection::cost_sorted` is the `sorted_by(cost_model)` step, and it +//! reuses this crate's existing `CostModel` trait rather than inventing a //! second cost interface (`docs/design_docs/cse-cost-model-decision.md`, //! issue #237, explicitly reasoned about *why* a narrow, direct cost //! comparison was enough for the CSE share/recompute decision alone, and @@ -246,25 +244,25 @@ //! //! - A group whose candidates are the [`SharedSubDAGStrategy`] //! share-vs-recompute pair is ranked by calling -//! [`CostModel::cse_share_decision`] via this module's own +//! `CostModel::cse_share_decision` via this module's own //! [`cse_preference`] — rather than re-deriving a competing comparison. -//! - A group whose candidates are [`SketchAlgorithmStrategy`]'s sketch-family -//! candidates is ranked via [`CostModel::rank_candidates`] (the same hook +//! - A group whose candidates are [`ASAPStrategies`]'s sketch-family +//! candidates is ranked via `CostModel::rank_candidates` (the same hook //! `realizations_for_intent` itself consults), applied to the //! candidates' own [`SketchAlgorithm`]s. //! - Any other shape (a single candidate, or a mix this module doesn't have //! a defined comparison for) keeps discovery order — there is nothing to -//! rank, or no [`CostModel`] hook this module knows how to apply; it never +//! rank, or no `CostModel` hook this module knows how to apply; it never //! invents a comparison `CostModel` doesn't already define. //! //! ## Whole-plan (cross-group) selection — issue #271 //! -//! [`CandidateLogicalASAPDAGs::cost_sorted`] above ranks every group's candidates +//! `candidate_selection::cost_sorted` above ranks every group's candidates //! independently: it never lets one group's choice influence how another //! group is costed. That's the right behavior when groups genuinely don't //! interact — which both shipped strategies' one-round convergence (see //! "Termination" above) makes the common case — but it's the wrong answer -//! whenever they do. Concretely: [`CostModel::cse_share_decision`] costs a +//! whenever they do. Concretely: `CostModel::cse_share_decision` costs a //! [`SharedSubDAGStrategy`] group by comparing a `consumer_count`-scaled //! recompute cost against a fixed maintenance cost — but a **nested** //! `SharedSubDAGStrategy` group's *true* recompute burden isn't its own @@ -277,7 +275,7 @@ //! per-group ranking has no way to see this — it only ever looks at one //! group's own `candidates`, in isolation. //! -//! [`CandidateLogicalASAPDAGs::global_selection`] is that missing step: a single +//! `candidate_selection::global_selection` is that missing step: a single //! **top-down dynamic-programming pass** over the discovered sites, //! processed in the topological order [`topological_order`] computes over a //! small [`ReferenceDAG`] built for exactly this purpose (parent before @@ -286,11 +284,11 @@ //! **effective consumer count** — how many times that site actually runs //! once every ancestor's own selected candidate is accounted for — and, for //! every [`SharedSubDAGStrategy`]-shaped group, re-decides -//! [`CostModel::cse_share_decision`] against *that* corrected count instead +//! `CostModel::cse_share_decision` against *that* corrected count instead //! of the group's raw structural one. When that group also contains a //! non-CSE alternative such as a semantic rewrite, the chosen CSE candidate //! and the cheapest non-CSE candidate additionally compete through -//! [`CostModel::estimate_cost`]; the CSE pair is no longer allowed to hide an +//! `CostModel::estimate_cost`; the CSE pair is no longer allowed to hide an //! otherwise valid logical alternative. See [`multiplier`]'s doc for the //! exact recurrence: a group that chooses `Share` collapses its own //! multiplicity to exactly `1` for everything beneath it (one shared @@ -307,7 +305,7 @@ //! into it) combined via a real recurrence — not just the MEMO-group //! sharing [`CandidateLogicalASAPDAGs`] itself already does for *storing* candidates. That //! distinction is exactly what issue #271 raised: this module already looks -//! like a Cascades/Volcano MEMO, but [`CandidateLogicalASAPDAGs::cost_sorted`] alone never +//! like a Cascades/Volcano MEMO, but `candidate_selection::cost_sorted` alone never //! actually performed this composition step; `global_selection` is that //! step, added alongside `cost_sorted` rather than replacing it (both stay //! available — see [`RankedTargetSubDAGCandidates`] vs. [`TargetSubDAGSelection`]'s own docs for when @@ -316,9 +314,9 @@ //! Two things this deliberately does **not** attempt, both left as //! documented follow-up rather than silently overclaimed: //! -//! - [`CostModel::rank_candidates`]/[`CostModel::size_params`] — the hooks -//! [`SketchAlgorithmStrategy`] groups rank by — take no `consumer_count` -//! parameter at all today, so a `SketchAlgorithmStrategy` group's selection +//! - `CostModel::rank_candidates` — the hook +//! [`ASAPStrategies`] groups rank by — takes no `consumer_count` +//! parameter at all today, so a `ASAPStrategies` group's selection //! here still falls back to [`rank_group`]'s ordinary (consumer-count- //! blind) local ranking, even though its own //! [`TargetSubDAGSelection::effective_consumer_count`] is computed and exposed @@ -329,7 +327,7 @@ //! would need, and this module now computes it for every group, sketch //! groups included. //! - This is not an exhaustive search over combinations of choices for a -//! provably-global optimum in every case. [`CostModel::cse_share_decision`] +//! provably-global optimum in every case. `CostModel::cse_share_decision` //! is still a *local*, pairwise comparison at each `SharedSubDAGStrategy` //! site (recompute-total vs. one fixed maintenance cost) — this module //! just now feeds it a *correct* input instead of an *incorrect* one. Two @@ -343,68 +341,66 @@ use crate::accuracy::estimators::{ cms::{cms_depth, cms_width}, - saturating_ceil, + saturating_ceil, size_params, }; +use asap_types::ir::operator::non_asap::any_measure_filtered; +use asap_types::ir::scalar::resolve_column_ref; use std::cell::RefCell; use std::collections::{HashMap, HashSet, VecDeque}; -use asap_types::post_asap::{ - validate_execution_data_states_at, EntityIdentity, ExactKind, ExactOperation, - ExactOperationSchemaError, ExactParams, ExecutionDataState, ExecutionDataStateError, - ExecutionTiming, Field, FieldDataType, GroupingStrategy, NonNegativeWeightProof, SamplingKind, - SamplingParams, Schema, SketchAlgorithm, SketchKind, SketchParams, - SketchStatistic as PostAsapSketchStatistic, StatModelKind, StatModelParams, SummaryExpr, - SummaryInputExpr, SummaryNode, SummaryUpdate, ValueOperation, WaveletKind, WaveletParams, - WeightDomain, +use asap_types::ir::cse::{share_common_sub_dags, structural_hash, HashCache}; +use asap_types::ir::operator::agg_intent::{agg_is_mergeable, AggIntent}; +use asap_types::ir::operator::operator_properties::{BinaryOpKind, JoinKind, Reduction}; +use asap_types::ir::properties::summary_coverage::{CoverageRegion, SummaryCoverage}; +use asap_types::ir::properties::timing::validate_maintained; +use asap_types::ir::properties::{ + AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee, }; -use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; -use asap_types::pre_asap::agg_intent::{agg_is_mergeable, AggIntent}; -use asap_types::pre_asap::column_resolution::resolve_column_ref; -use asap_types::pre_asap::cse::{share_common_sub_dags, structural_hash, HashCache}; -use asap_types::pre_asap::expr_ir::{ArithmeticOpKind, ColumnRef}; -use asap_types::pre_asap::query_expr::any_measure_filtered; -use asap_types::pre_asap::query_expr::{ - BinaryOpKind, Predicate, QueryExpr, QueryExprError, Reduction, +use asap_types::ir::properties::{ExecutionDataStateError, ExecutionTiming}; +use asap_types::ir::scalar::{ArithmeticOpKind, ColumnRef}; +use asap_types::ir::schema::{ + ColumnId, EntityIdentity, ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, + NonNegativeWeightProof, SamplingKind, SamplingParams, Schema, SketchAlgorithm, SketchKind, + SketchParams, SketchStatistic as PostAsapSketchStatistic, StatModelKind, StatModelParams, + SummaryInputExpr, SummaryUpdate, WaveletKind, WaveletParams, WeightDomain, }; -use asap_types::pre_asap::schema::ColumnId; +use asap_types::ir::SchemaDerivationError; +use asap_types::ir::{ + ASAPOp, BinaryOperator, NonASAPOp, Operator, OperatorNode, Predicate, ProjectItem, ScalarExpr, + SortKey, +}; +use asap_types::physical::ExactOperationSchemaError; use asap_types::types::AccuracyTarget; -use asap_types::workload::{DataWorkload, QueryRecurrence, QueryWorkload, RepeatedDemand}; -use std::rc::Rc; +use std::rc::{Rc, Weak}; use thiserror::Error; -use crate::accuracy::reconciliation::AccuracyReconciliationStrategy; use crate::accuracy::{ AccuracyBudgetAllocator, AccuracyEvidenceProvider, AccuracyModel, CompositionShape, DefaultAccuracyModel, EqualSplitAllocator, NoAccuracyEvidence, }; -use crate::cost_model::{ - raw_recompute_cost_rate, Cost, CostModel, CseCandidate, DefaultCostModel, - ExactCompositionCostInputs, ExactCompositionCostRequest, ShareDecision, -}; -use crate::exact_composition::{ExactComposition, ExactCompositionStrategy, OperationPlacement}; -use crate::grouping::HydraGroupingStrategy; -use crate::recurrence::CostRate; -use crate::recurrence::{ - evaluation_rate_of, Horizon, RecurrenceError, RecurrenceProfile, RootRecurrence, UpdateRate, +use crate::pass1::exact_composition::{ + ExactComposition, ExactCompositionStrategy, OperationPlacement, }; -use crate::rollup::RollupStrategy; -use crate::topk_reuse::TopKLimitReuseStrategy; +use crate::pass1::grouping::HydraGroupingStrategy; +use crate::pass1::rollup::RollupStrategy; +use crate::pass2::reconciliation::AccuracyReconciliationStrategy; +use crate::pass2::topk_reuse::TopKLimitReuseStrategy; /// Errors from the pre-ASAP → post-ASAP replacement/construction path -/// ([`realize_child`] and [`keep_pre_asap`]). Moved here from the former +/// ([`realize_child`] and [`retain_exact`]). Moved here from the former /// `bind.rs` (issue #251): this is what a [`ReplacementStrategy`] /// implementor's own construction path can realistically fail with — -/// schema derivation over a pre-ASAP [`QueryExpr`] — not something specific -/// to workload-wide orchestration. +/// schema derivation over a pre-ASAP [`OperatorNode`] sub-DAG — not +/// something specific to workload-wide orchestration. #[derive(Debug, Error)] pub enum RealizationError { /// Schema derivation failed while lifting an edge to `Schema`. #[error("schema derivation failed during pre-ASAP → post-ASAP binding: {0}")] - Schema(#[from] QueryExprError), + Schema(#[from] SchemaDerivationError), /// The candidate is accuracy-illegal (issue #172): its composed /// guarantee has no sound propagation rule, or misses the applicable /// `AccuracyTarget`. Fail-closed — the candidate is never constructed - /// with the child "treated as exact". [`SketchAlgorithmStrategy::propose`] + /// with the child "treated as exact". [`ASAPStrategies::propose`] /// records it as a [`RejectedCandidate`] instead of a candidate. #[error("accuracy-illegal candidate: {0}")] Accuracy(#[from] AccuracyError), @@ -414,8 +410,8 @@ pub enum RealizationError { /// would change its semantics. #[error("unsupported physical summary realization: {0}")] PhysicalRealization(&'static str), - /// A constructed plan violates the update/readout phase contract - /// (issue #171) — e.g. a summary readout placed beneath a maintained + /// A constructed plan violates the update/evaluation phase contract + /// (issue #171) — e.g. a summary evaluation placed beneath a maintained /// `SummaryAgg`. Detected at construction, never at runtime. #[error("execution-data_state violation in post-ASAP plan: {0}")] ExecutionDataState(#[from] ExecutionDataStateError), @@ -427,10 +423,10 @@ pub enum RealizationError { /// A pre-ASAP sub-DAG a [`ReplacementStrategy`] knows how to replace. /// -/// `root` is a reference into the workload's own [`QueryExpr`] DAG (an -/// `Rc`, the same currency [`search_workload`] and -/// `asap_types::pre_asap::cse::share_common_sub_dags` already thread through -/// this crate's public API — not a bare `&QueryExpr` — so a strategy that +/// `root` is a reference into the workload's own [`OperatorNode`] DAG (an +/// `Rc`, the same currency [`search_workload`] and +/// `asap_types::ir::cse::share_common_sub_dags` already thread through +/// this crate's public API — not a bare `&OperatorNode` — so a strategy that /// needs the node's own `Rc` identity, not just its shape, has it available /// without the caller re-deriving it). /// @@ -439,16 +435,16 @@ pub enum RealizationError { /// [`search_workload_with`] computes the workload-wide value during target /// discovery. [`TargetSubDAG::new`] defaults it to `1` for callers invoking a /// strategy against one node in isolation. A strategy that only cares about -/// `root`'s shape (for example, [`SketchAlgorithmStrategy`]) can ignore the +/// `root`'s shape (for example, [`ASAPStrategies`]) can ignore the /// count; [`SharedSubDAGStrategy`] consults it directly. /// /// `strictest_sibling_accuracy` is the strictest accuracy among workload /// siblings that read the same summary input as `root`, when stricter than -/// `root`'s own. [`search_workload_with`] sets it; [`SketchAlgorithmStrategy`] +/// `root`'s own. [`search_workload_with`] sets it; [`ASAPStrategies`] /// also sizes a candidate to it. #[derive(Debug, Clone, Copy)] pub struct TargetSubDAG<'a> { - pub root: &'a Rc, + pub root: &'a Rc, pub consumer_count: usize, pub strictest_sibling_accuracy: Option<&'a AccuracyTarget>, } @@ -456,7 +452,7 @@ pub struct TargetSubDAG<'a> { impl<'a> TargetSubDAG<'a> { /// A target assumed to have exactly one consumer — the common case for a /// caller that isn't already tracking cross-workload sharing. - pub fn new(root: &'a Rc) -> Self { + pub fn new(root: &'a Rc) -> Self { Self { root, consumer_count: 1, @@ -466,7 +462,7 @@ impl<'a> TargetSubDAG<'a> { /// A target with an explicit `consumer_count`, used by workload discovery /// and by callers that already know how many locations reference `root`. - pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { + pub fn with_consumer_count(root: &'a Rc, consumer_count: usize) -> Self { Self { root, consumer_count, @@ -482,29 +478,39 @@ impl<'a> TargetSubDAG<'a> { /// — into "one candidate among several", each with its own /// [`ReplacementSubDAG`]. #[derive(Debug, Clone)] +#[allow(clippy::large_enum_variant)] // Keep the public strategy API value-based. pub enum Replacement { - /// A fully bound post-ASAP summary decision, for one particular - /// candidate realization of the target. - Summary(Rc), - /// A pre-ASAP rewrite: still a logical [`QueryExpr`], structurally - /// different from the target's own `root` (e.g. sharing vs. not sharing - /// a sub-DAG) but semantically equivalent to it. - Rewrite(Rc), + /// A sub-DAG that replaces the target: either a bound summary decision + /// (a DAG containing ASAP operators, for one particular candidate + /// realization of the target) or a pre-ASAP rewrite (a logical sub-DAG + /// with no ASAP operator, structurally different from the target's own + /// `root` — e.g. sharing vs. not sharing a sub-DAG — but semantically + /// equivalent to it). [`is_logical_rewrite`] tells the two apart. + SubDAG(Rc), /// An exact operator composed over another target's *own* selected - /// decision across an explicit update/readout boundary (issue #171): - /// `ValueOperationAtQueryTime` over a child's summary readout, or + /// decision across an explicit update/evaluation boundary (issue #171): + /// `ValueOperationAtQueryTime` over a child's summary evaluation, or /// `ValueOperationAtIngestionTime` feeding a maintained summary above. Carries only a - /// reference to the child target — [`CandidateLogicalASAPDAGs::global_selection`] + /// reference to the child target — `candidate_selection::global_selection` /// commits the compatible parent/child pair and /// [`GlobalSelection::assemble_selected_dag`] links it into one validated - /// `SummaryNode`. See [`crate::exact_composition`]. + /// `OperatorNode` DAG. See [`crate::pass1::exact_composition`]. ExactComposition(ExactComposition), } +/// Whether a [`Replacement::SubDAG`] is a pure logical rewrite: a sub-DAG +/// with no ASAP operator and no guarantee established yet (the shape every +/// front end emits and every rewrite strategy builds). A bound summary +/// decision contains an ASAP operator, or is a kept pre-ASAP sub-DAG that +/// already carries its exact guarantee. +pub fn is_logical_rewrite(node: &OperatorNode) -> bool { + node.guarantee.is_none() && !node.contains_asap() +} + /// One candidate replacement for a [`TargetSubDAG`], plus a human-readable /// `rationale` explaining why it's a valid candidate (meant for a /// report/log/debugging a search engine's choices, not machine parsing — -/// [`crate::explanation::ReplacementExplanation::reason`] literally reuses +/// [`crate::pass1::explanation::ReplacementExplanation::reason`] literally reuses /// this same string rather than inventing new prose of its own. #[derive(Debug, Clone)] pub struct ReplacementSubDAG { @@ -523,26 +529,14 @@ pub struct ReplacementSubDAG { impl ReplacementSubDAG { /// Whether this summary still needs accuracy/domain evidence before it can /// be treated as certified. A missing guarantee on any summary candidate - /// is unknown; exact `KeepPreAsap` carries an explicit exact guarantee. + /// (a sub-DAG whose root is an ASAP operator) is unknown; a kept + /// pre-ASAP sub-DAG carries an explicit exact guarantee. pub fn has_missing_accuracy_evidence(&self) -> bool { matches!( &self.replacement, - Replacement::Summary(node) if has_missing_accuracy_evidence(node) + Replacement::SubDAG(node) if !is_logical_rewrite(node) && has_missing_accuracy_evidence(node) ) } - - /// Physical feasibility evidence for this candidate. A pure logical - /// rewrite needs no new operator. Unknown support is checked during - /// physical/deployment compilation; explicit rejection prevents selection. - pub fn runtime_support_evidence(&self, cost_model: &dyn CostModel) -> Option { - match &self.replacement { - Replacement::ExactComposition(composition) => { - cost_model.value_operation_support_evidence(&composition.op, composition.placement) - } - Replacement::Summary(node) => cost_model.summary_support_evidence(node), - Replacement::Rewrite(_) => Some(true), - } - } } #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -551,12 +545,12 @@ pub enum ReplacementProvenance { CseShare, CseRecompute, LogicalRewrite, - /// [`crate::accuracy::reconciliation::AccuracyReconciliationStrategy`]'s + /// [`crate::pass2::reconciliation::AccuracyReconciliationStrategy`]'s /// "read a strictly-tighter sibling instead of building an independent, /// looser copy" candidate (issue #273). Kept distinct from /// `LogicalRewrite` — even though both are structurally-different, /// semantically-equivalent rewrites — because - /// [`crate::cost_model::DefaultCostModel::estimate_cost`] needs to price + /// `cost_model::DefaultCostModel::estimate_cost` needs to price /// it differently: `LogicalRewrite` candidates (`RollupStrategy`, /// `TopKLimitReuseStrategy`) still rebuild `target` itself from a /// different source, so pricing them like an independent rebuild is @@ -574,7 +568,7 @@ pub enum ReplacementProvenance { /// A finalized whole-query result over rows carrying the PromQL series /// identity, which the logical root does not expose (see /// [`ReplacementStrategy::propose_for_root`]). Default selection never - /// commits it, because its readout must be validated and priced by + /// commits it, because its evaluation must be validated and priced by /// deployment; otherwise it would silently replace the logical plan. RootPhysicalRealization, } @@ -583,7 +577,7 @@ pub enum ReplacementProvenance { /// accuracy-legality grounds (issue #172) — kept alongside the group's /// legal candidates in [`TargetSubDAGCandidates::rejected`] so a rejection is as /// inspectable (and exportable) as a selection. Never ranked: a -/// [`CostModel`] only ever sees [`TargetSubDAGCandidates::candidates`]. +/// `CostModel` only ever sees [`TargetSubDAGCandidates::candidates`]. #[derive(Debug, Clone)] pub struct RejectedCandidate { /// Name of the [`ReplacementStrategy`] that considered it. @@ -610,12 +604,12 @@ pub struct Proposals { /// replacement (`replacements`)? /// /// The extension point this module exists for — the same shape -/// [`CostModel`] and [`Matcher`] already use elsewhere in this crate: a new +/// `CostModel` and [`Matcher`] already use elsewhere in this crate: a new /// replacement source is a new `impl ReplacementStrategy`, no restructuring /// of this trait or any existing strategy required. /// /// `replacements` is only meaningful when `matches` would return `true` for -/// the same target; both [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] +/// the same target; both [`ASAPStrategies`] and [`SharedSubDAGStrategy`] /// return an empty `Vec` rather than panicking when called on a target they /// don't match, so a caller that skips the `matches` check first still gets a /// safe (merely uninformative) answer instead of a crash. @@ -635,7 +629,7 @@ pub trait ReplacementStrategy { /// Every valid replacement for `target` — not ranked, not filtered. /// Reporting "every valid candidate" is this method's whole job; picking - /// the best one is a [`CostModel`]'s job, out of scope here. + /// the best one is a `CostModel`'s job, out of scope here. fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec; /// [`replacements`](Self::replacements) plus the accuracy-illegal @@ -657,7 +651,7 @@ pub trait ReplacementStrategy { /// (for example, the PromQL series identity), so /// [`search_workload_with_targets`] asks only workload roots, once each. /// They decide what to compute, never placement. Default: none. - fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { + fn propose_for_root(&self, _root: &Rc, _target: &AccuracyTarget) -> Proposals { Proposals::default() } } @@ -674,41 +668,41 @@ pub trait ReplacementStrategy { /// [`realizations_for_intent`] is where every valid realization gets /// enumerated, exhaustive and ranked (most-preferred first) — this crate has /// no separate function that computes just "the one" `Realization` -/// independently of that list. [`SketchAlgorithmStrategy`] is the sole +/// independently of that list. [`ASAPStrategies`] is the sole /// consumer: it wraps every entry of this list into its own bound -/// [`SummaryNode`] and returns all of them, ranked — a caller wanting a +/// [`OperatorNode`] and returns all of them, ranked — a caller wanting a /// single answer keeps the first one itself (see the module docs above). #[derive(Debug, Clone, PartialEq)] pub enum Realization { /// An exact **mergeable** accumulator (partial state ≡ the value /// itself: `Sum` / `Count` / `Min` / `Max` / `Rate` / `Increase`). The - /// built state *is* the answer already — no `SummaryEstimate` readout + /// built state *is* the answer already — no `SummaryEstimate` evaluation /// step. ExactAggregate { kind: ExactKind, params: ExactParams, }, /// An approximate sketch sized to the intent's [`AccuracyTarget`]. - /// Needs a `SummaryEstimate` readout to recover a value. Already + /// Needs a `SummaryEstimate` evaluation to recover a value. Already /// classified into its [`SketchKind`] category (`SketchKind::new` /// having been called) — construction always goes through that /// classifier, never this variant directly. Sketch(SketchKind), /// A sampling-based summary (a retained row subset). Needs a - /// `SummaryEstimate` readout. Not chosen by any core `AggIntent` + /// `SummaryEstimate` evaluation. Not chosen by any core `AggIntent` /// dispatch today — see the module docs. Sample { kind: SamplingKind, params: SamplingParams, }, - /// A wavelet-transform summary. Needs a `SummaryEstimate` readout. Not + /// A wavelet-transform summary. Needs a `SummaryEstimate` evaluation. Not /// chosen by any core `AggIntent` dispatch today — see the module docs. Wavelet { kind: WaveletKind, params: WaveletParams, }, /// A fitted statistical/parametric-model summary. Needs a - /// `SummaryEstimate` readout. Not chosen by any core `AggIntent` + /// `SummaryEstimate` evaluation. Not chosen by any core `AggIntent` /// dispatch today — see the module docs. StatModel { kind: StatModelKind, @@ -823,11 +817,11 @@ pub fn accuracy_target(intent: &AggIntent) -> Option<&AccuracyTarget> { } } -/// Every valid [`Realization`] for `intent`, exhaustive and ranked -/// (most-preferred first via `cost_model`) — the *only* place this crate +/// Every valid [`Realization`] for `intent`, exhaustive, in +/// [`summary_candidates`]' static order — the *only* place this crate /// decides what an `AggIntent` may become. Nothing in this crate computes /// "the one" `Realization` independently of this list: -/// [`SketchAlgorithmStrategy`] keeps every entry as a candidate, and a caller +/// [`ASAPStrategies`] keeps every entry as a candidate, and a caller /// that wants a single executable answer takes the head of *that* strategy's /// output itself. /// @@ -835,15 +829,12 @@ pub fn accuracy_target(intent: &AggIntent) -> Option<&AccuracyTarget> { /// explicit realization is a compile error, and the coverage-matrix test pins /// each variant's category. /// -/// `pub(crate)`: [`SketchAlgorithmStrategy::replacements`] is this module's +/// `pub(crate)`: [`ASAPStrategies::replacements`] is this module's /// own caller; `grouping::HydraGroupingStrategy` (issue #256) is the one /// caller outside it, needing the exact same already-ranked candidate list /// to find the `Realization::Sketch` matching the Hydra-eligible kind it /// is building a candidate for. -pub(crate) fn realizations_for_intent( - intent: &AggIntent, - cost_model: &dyn CostModel, -) -> Vec { +pub(crate) fn realizations_for_intent(intent: &AggIntent) -> Vec { match intent { // ── Approximate-capable intents — the AccuracyTarget decides ──────── AggIntent::Quantile { accuracy, .. } @@ -861,11 +852,11 @@ pub(crate) fn realizations_for_intent( ], AccuracyTarget::Exact => vec![exact_realization(intent)], _ if matches!(intent, AggIntent::Count { .. }) => { - let mut candidates = sketch_realizations(intent, accuracy, cost_model); + let mut candidates = sketch_realizations(intent, accuracy); candidates.push(exact_realization(intent)); candidates } - _ => sketch_realizations(intent, accuracy, cost_model), + _ => sketch_realizations(intent, accuracy), }, // ── Exact mergeable accumulators ───────────────────────────────────── @@ -875,7 +866,7 @@ pub(crate) fn realizations_for_intent( | AggIntent::Rate | AggIntent::IRate | AggIntent::Increase => { - let (kind, params) = crate::function_rules::function_rules(intent) + let (kind, params) = crate::pass1::function_rules::function_rules(intent) .and_then(|rules| rules.accumulator) .expect("exact accumulator intents have registered realizations"); vec![exact_accumulator(intent, kind, params)] @@ -931,17 +922,9 @@ pub(crate) fn realizations_for_intent( AggIntent::Group | AggIntent::CountValues { .. } => vec![Realization::PassThrough], // ── Extension (deployment-model-specific, issue #131) — core has no - // realization opinion for a shape it doesn't know, so it defers - // entirely to the `CostModel` (issue #150): `realize_extension` - // defaults to `PassThrough`, preserving today's behavior for - // every deployment that doesn't override it. Core has no way to - // enumerate alternatives for an opaque deployment-defined shape, - // so this is always exactly one candidate. This is also the only - // path that can currently produce `Realization::Sample`/ - // `Wavelet`/`StatModel` — see the module docs. - AggIntent::Extension { ext_kind, payload } => { - vec![cost_model.realize_extension(ext_kind, payload)] - } + // realization opinion for a shape it doesn't know, so it stays + // logical. + AggIntent::Extension { .. } => vec![Realization::PassThrough], } } @@ -966,8 +949,8 @@ fn exact_accumulator(intent: &AggIntent, kind: ExactKind, params: ExactParams) - Realization::ExactAggregate { kind, params } } -/// Resolve an [`AccuracyTarget`] into the `(eps, delta)` budget -/// [`CostModel::size_params`] needs. Shared by [`sketch_realizations`] and +/// Resolve an [`AccuracyTarget`] into the `(eps, delta)` budget sketch +/// sizing needs. Shared by [`sketch_realizations`] and /// this crate's own sizing — one place this resolution happens, so nothing /// can drift apart on it. /// @@ -983,23 +966,15 @@ pub fn accuracy_budget(accuracy: &AccuracyTarget) -> (f64, f64) { } /// Every candidate sketch [`Realization`] for an approximate-capable -/// intent, sized to `accuracy` and ranked via `cost_model.rank_candidates` -/// (most-preferred first) — [`realizations_for_intent`]'s Sketch branch. -fn sketch_realizations( - intent: &AggIntent, - accuracy: &AccuracyTarget, - cost_model: &dyn CostModel, -) -> Vec { +/// intent, sized analytically to `accuracy`, in [`summary_candidates`]' +/// order — [`realizations_for_intent`]'s Sketch branch. +fn sketch_realizations(intent: &AggIntent, accuracy: &AccuracyTarget) -> Vec { let (eps, delta) = accuracy_budget(accuracy); - let ranked = crate::cost_model::validated_candidate_ranking( - cost_model, - intent, - summary_candidates(intent), - ); - ranked - .into_iter() + summary_candidates(intent) + .iter() + .cloned() .filter_map(|algorithm| { - let params = cost_model.size_params(algorithm.clone(), intent, eps, delta); + let params = size_params(algorithm.clone(), intent, eps, delta); sketch_state_bytes(¶ms) .is_none_or(|bytes| bytes <= DEFAULT_MAX_SKETCH_STATE_BYTES) .then(|| Realization::Sketch(SketchKind::new(algorithm, params))) @@ -1052,10 +1027,7 @@ pub fn sketch_state_bytes(params: &SketchParams) -> Option { } /// `asap-plan`'s built-in `SketchParams` sizing, keyed off the resolved -/// `(eps, delta)` accuracy budget. [`CostModel::size_params`]'s default -/// body — factored out to a free function so a deployment's own -/// `CostModel` impl can still delegate to it for the candidates it -/// doesn't want to resize itself. +/// `(eps, delta)` accuracy budget. /// /// Each formula inverts the sketch family's standard error bound to the /// smallest parameter satisfying the target, clamped to the family's sane @@ -1075,12 +1047,9 @@ pub fn default_size_params( /// [`posterior_aware_size_params`]. /// /// This is **not** derived from Chen et al.'s posterior-error-estimation -/// technique (issue #239, `asap_types::post_asap::query_time::error_estimation`) -/// — that technique computes a tighter bound *at query time* from a -/// sketch's real counter values, and this repo has no sketch runtime yet -/// for a real counter array to size against (see that module's docs, and -/// `asap_types::post_asap::query_time`'s module doc for why it's a -/// deliberately separate folder from this crate's own *plan-time* code). +/// technique (issue #239) — that technique computes a tighter bound *at +/// query time* from a sketch's real counter values, and this repo has no +/// sketch runtime yet for a real counter array to size against. /// This struct is this crate's own *plan-time* analogue of the same /// underlying intuition — an expected-case (skewed / non-adversarial) /// workload needs a smaller sketch than the adversarial worst case — @@ -1108,8 +1077,7 @@ pub struct ExpectedCaseSizing { /// **The tradeoff, spelled out:** [`default_size_params`]'s width guarantees /// `Pr[error > ε·|F|₁] < δ` for *any* input, including an adversarial one /// built to maximize collisions (§3.3 of the posterior-error-estimation -/// paper this issue is about — see -/// `asap_types::post_asap::query_time::error_estimation`'s module docs). +/// paper issue #239 is about). /// Shrinking /// width below that only keeps the same `(ε,δ)` guarantee if the real /// workload's collision load stays within `width_relaxation` of the @@ -1190,36 +1158,28 @@ pub fn posterior_aware_size_params( } } -// ── SketchAlgorithmStrategy ───────────────────────────────────────────────── +// ── ASAPStrategies ───────────────────────────────────────────────── -/// A single static instance so [`SketchAlgorithmStrategy::default_cost_model`] -/// can hand out a `&'static dyn CostModel` without heap-allocating one — -/// `DefaultCostModel` is a unit struct with no state, so one instance serves -/// every caller. -static DEFAULT_COST_MODEL: DefaultCostModel = DefaultCostModel; static DEFAULT_ACCURACY_MODEL: DefaultAccuracyModel = DefaultAccuracyModel; static DEFAULT_ALLOCATOR: EqualSplitAllocator = EqualSplitAllocator; static NO_ACCURACY_EVIDENCE: NoAccuracyEvidence = NoAccuracyEvidence; -/// The cost, accuracy, allocation, and evidence inputs consulted during -/// candidate construction, bundled so the construction path threads one argument. `cost` ranks and sizes; `accuracy` and `allocator` decide -/// legality (issue #172) — see [`crate::accuracy`]'s module docs for why -/// those are separate from `cost` and run before it. +/// The accuracy, allocation, and evidence inputs consulted during +/// candidate construction, bundled so the construction path threads one +/// argument. `accuracy` and `allocator` decide legality (issue #172) — see +/// [`crate::accuracy`]'s module docs. #[derive(Clone, Copy)] pub(crate) struct CandidatePlanningInputs<'a> { - pub cost: &'a dyn CostModel, pub accuracy: &'a dyn AccuracyModel, pub allocator: &'a dyn AccuracyBudgetAllocator, pub evidence: &'a dyn AccuracyEvidenceProvider, } -impl<'a> CandidatePlanningInputs<'a> { - /// `cost` with the built-in [`DefaultAccuracyModel`]/ - /// [`EqualSplitAllocator`] — what every entry point that only takes a - /// `CostModel` uses. - pub(crate) fn with_default_accuracy(cost: &'a dyn CostModel) -> Self { +impl CandidatePlanningInputs<'static> { + /// The built-in [`DefaultAccuracyModel`]/[`EqualSplitAllocator`], with no + /// planning-time evidence. + pub(crate) fn with_default_accuracy() -> Self { Self { - cost, accuracy: &DEFAULT_ACCURACY_MODEL, allocator: &DEFAULT_ALLOCATOR, evidence: &NO_ACCURACY_EVIDENCE, @@ -1227,62 +1187,45 @@ impl<'a> CandidatePlanningInputs<'a> { } } -/// Wraps [`realizations_for_intent`]'s exhaustive, ranked list directly: for -/// a bindable `Aggregate`, every valid candidate summary realization as its -/// own [`ReplacementSubDAG`]. +/// Proposes the supported ASAP realizations for a bindable aggregate, including +/// exact accumulators, approximate sketches, and supported maintained populations. +/// Each valid realization becomes its own [`ReplacementSubDAG`]. /// -/// Ranked (only to *order the enumeration*, never to drop a candidate) via a -/// [`CostModel`] — [`DefaultCostModel`] unless constructed with -/// [`SketchAlgorithmStrategy::new`] — so a deployment-specific cost model's -/// other hooks (`size_params`, `realize_extension`, `readout_extension`) are -/// still consulted while binding each candidate. +/// [`realizations_for_intent`] enumerates summary families. This is not +/// limited to sketch algorithms. Nothing here ranks candidates by cost; that +/// is plan selection's job. /// -/// The one thing that *does* drop a candidate is accuracy legality (issue -/// #172), decided by the [`AccuracyModel`] — never by the cost model: a +/// The one thing that drops a candidate is accuracy legality (issue +/// #172), decided by the [`AccuracyModel`]: a /// sketch over an approximate child is proposed only if its composed /// guarantee has a sound propagation rule and satisfies the node's own /// `AccuracyTarget`; otherwise it is reported through /// [`ReplacementStrategy::propose`] as a [`RejectedCandidate`]. See /// [`crate::accuracy`]'s module docs for the rules and the precedence /// between root and per-node targets. -pub struct SketchAlgorithmStrategy<'a> { +pub struct ASAPStrategies<'a> { planning_inputs: CandidatePlanningInputs<'a>, } -impl SketchAlgorithmStrategy<'static> { - /// A strategy that ranks/binds via the built-in [`DefaultCostModel`] — - /// what a deployment gets with no custom cost model plugged in. - pub fn default_cost_model() -> Self { +impl Default for ASAPStrategies<'static> { + /// The built-in [`DefaultAccuracyModel`]/[`EqualSplitAllocator`]. + fn default() -> Self { Self { - planning_inputs: CandidatePlanningInputs::with_default_accuracy(&DEFAULT_COST_MODEL), + planning_inputs: CandidatePlanningInputs::with_default_accuracy(), } } } -impl<'a> SketchAlgorithmStrategy<'a> { - /// A strategy that ranks/binds via `cost_model` instead of the built-in - /// static preference order — the same customization point - /// [`realizations_for_intent`] already offers. Accuracy legality stays - /// with the built-in [`DefaultAccuracyModel`]/[`EqualSplitAllocator`]. - pub fn new(cost_model: &'a dyn CostModel) -> Self { - Self { - planning_inputs: CandidatePlanningInputs::with_default_accuracy(cost_model), - } - } - - /// A strategy with every model plugged in explicitly: `cost_model` for - /// ranking/sizing, `accuracy_model` for guarantee derivation/propagation/ - /// satisfaction, `allocator` for end-to-end budget splits. One model - /// never overrides another: legality is settled by `accuracy_model` - /// before `cost_model` ranks what is left. +impl<'a> ASAPStrategies<'a> { + /// A strategy with every model plugged in explicitly: `accuracy_model` + /// for guarantee derivation/propagation/satisfaction, `allocator` for + /// end-to-end budget splits. pub fn new_with_planning_inputs( - cost_model: &'a dyn CostModel, accuracy_model: &'a dyn AccuracyModel, allocator: &'a dyn AccuracyBudgetAllocator, ) -> Self { Self { planning_inputs: CandidatePlanningInputs { - cost: cost_model, accuracy: accuracy_model, allocator, evidence: &NO_ACCURACY_EVIDENCE, @@ -1293,14 +1236,12 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// Like [`Self::new_with_planning_inputs`], with typed planning-time evidence for /// rules such as TopK membership and Hydra shared-grid composition. pub fn new_with_planning_inputs_and_evidence( - cost_model: &'a dyn CostModel, accuracy_model: &'a dyn AccuracyModel, allocator: &'a dyn AccuracyBudgetAllocator, evidence: &'a dyn AccuracyEvidenceProvider, ) -> Self { Self { planning_inputs: CandidatePlanningInputs { - cost: cost_model, accuracy: accuracy_model, allocator, evidence, @@ -1314,34 +1255,33 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// treats a range of historical samples as the instant vector. pub fn current_series_topk_candidates( &self, - root: &Rc, + root: &Rc, accuracy: &AccuracyTarget, ) -> Proposals { - let QueryExpr::Limit { - n, + let Some(NonASAPOp::Limit { + n: Some(n), offset: 0, child, - } = root.as_ref() + .. + }) = root.non_asap() else { return Proposals::default(); }; - let QueryExpr::Sort { + let Some(NonASAPOp::Sort { keys, partition_by, child, - } = child.as_ref() + }) = child.non_asap() else { return Proposals::default(); }; let [key] = keys.as_slice() else { return Proposals::default(); }; - let QueryExpr::Column(value) = key.expr else { - return Proposals::default(); - }; - let Ok(schema) = child.output_schema() else { + let ScalarExpr::Column(value) = key.expr else { return Proposals::default(); }; + let schema = &child.schema; if key.ascending || key.nulls_first || partition_by.is_without() @@ -1354,20 +1294,103 @@ impl<'a> SketchAlgorithmStrategy<'a> { { return Proposals::default(); } - let ranked = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(partition_by.clone()), - measures: vec![AggIntent::TopK { - k: *n, - accuracy: accuracy.clone(), - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::clone(child), - }); + let Ok(ranked) = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(partition_by.clone()), + measures: vec![AggIntent::TopK { + k: *n, + accuracy: accuracy.clone(), + }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::clone(child), + })) + else { + return Proposals::default(); + }; self.propose_with(&ranked, None, None) } + /// Fixed-window maintenance can finalize each series' counter state and + /// build a fresh heap or grouped Sum for that evaluation window. Deployment must provide + /// a complete, synchronized population and bind the matching window; this + /// candidate never incrementally adds one window's rates to another. + /// + pub fn fixed_window_rate_candidates(&self, root: &Rc) -> Proposals { + fn place(node: &Rc) -> Option> { + retime_rate_finalize(node, ExecutionTiming::IngestionTime, true) + } + // Legal only if the candidate stays executable with its states maintained. + let timed = |node: &Rc| { + asap_types::ir::properties::timing::apply_materialization_timings( + node, + &asap_types::ir::properties::timing::MaterializationAssignment::all_ingestion_time( + ), + &mut asap_types::ir::properties::timing::TimingMemo::new(), + ) + .ok() + .and_then(|timed| asap_types::ir::export::compile_physical_asap_dag(&timed).ok()) + }; + let mut proposals = self.propose_with(root, None, None); + proposals.candidates.retain_mut(|candidate| { + let Replacement::SubDAG(node) = &candidate.replacement else { + return false; + }; + let Some(dag) = timed(node) else { + return false; + }; + if !dag.nodes.iter().any(|node| match &node.payload { + asap_types::ir::export::PhysicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + } => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + asap_types::ir::export::PhysicalASAPOperatorPayload::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Sum, _), + .. + } => true, + _ => false, + }) { + return false; + } + let Some(placed) = place(node) else { + return false; + }; + if timed(&placed).is_none() { + return false; + } + let Ok(placed) = finalize_query_candidate(placed, root) else { + return false; + }; + candidate.replacement = Replacement::SubDAG(placed); + candidate + .rationale + .push_str("; fixed-window precompute over complete per-series counter states"); + true + }); + proposals + } + + /// Retain grouped Sum after a per-series Rate evaluation as a query-time + /// candidate alongside its complete-window maintenance placement. + /// + pub fn query_time_rate_aggregation_candidates(&self, root: &Rc) -> Proposals { + let mut proposals = self.fixed_window_rate_candidates(root); + proposals.candidates.retain_mut(|candidate| { + let Replacement::SubDAG(node) = &candidate.replacement else { return false }; + if !matches!(&node.operator, Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!(&child.operator, Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(ExactKind::Sum, _), .. }))) { return false; } + let Some(query_time) = retime_rate_finalize(node, ExecutionTiming::QueryTime, false) else { return false }; + candidate.replacement = Replacement::SubDAG(query_time); + candidate.rationale = "query-time grouped Sum over complete per-series Rate evaluations".into(); + true + }); + proposals + } + pub(crate) fn from_planning_inputs(planning_inputs: CandidatePlanningInputs<'a>) -> Self { Self { planning_inputs } } @@ -1380,20 +1403,21 @@ impl<'a> SketchAlgorithmStrategy<'a> { /// with the sibling that needs it. fn propose_with( &self, - root: &Rc, + root: &Rc, intent_override: Option<&AggIntent>, strictest_sibling: Option<&AccuracyTarget>, ) -> Proposals { let mut proposals = Proposals::default(); - // A selected logical rewrite otherwise remains KeepPreAsap during DAG - // assembly. Also expose its concrete summary realization for selection. + // A selected logical rewrite otherwise stays a kept pre-ASAP sub-DAG + // during DAG assembly. Also expose its concrete summary realization + // for selection. if intent_override.is_none() { - if let Some(rewritten) = crate::rewrite::composed_aggregate_rewrite(root) { + if let Some(rewritten) = crate::pass1::rewrite::composed_aggregate_rewrite(root) { if let Ok(node) = realize_child_with(&rewritten, self.planning_inputs, None) { - if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { + if node.contains_asap() { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "realize a schema-preserving composition of temporal and grouped accumulators".into(), }); @@ -1403,8 +1427,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { } if let Ok(Some(node)) = exact_topk_over_temporal_values(root, self.planning_inputs) { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "select exact Top-K from independently maintained temporal values" .into(), @@ -1413,8 +1437,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { if intent_override.is_none() { if let Ok(Some(node)) = realize_temporal_average(root, self.planning_inputs, None) { proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: "read temporal average from sum/count only within the finite arithmetic domain; otherwise execute the original average".into(), }); @@ -1428,8 +1452,8 @@ impl<'a> SketchAlgorithmStrategy<'a> { "preserve exact PromQL arithmetic over independently realized summary operands" }; proposals.candidates.push(ReplacementSubDAG { - replacement: Replacement::Summary(node), - strategy: "SketchAlgorithmStrategy", + replacement: Replacement::SubDAG(node), + strategy: "ASAPStrategies", provenance: ReplacementProvenance::SummaryRealization, rationale: rationale.into(), }); @@ -1460,7 +1484,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { // candidate in practice (every other variant's own dispatch produces // exactly one `Realization`), but this loop doesn't need to know // that; it just constructs whatever the list contains. - for realization in realizations_for_intent(intent, planning_inputs.cost) { + for realization in realizations_for_intent(intent) { let rationale = describe_realization(intent, &realization); // The as-declared composition: every layer sized to its own // declared `AccuracyTarget`. Legal iff the composed guarantee @@ -1483,10 +1507,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { if let (Some(stricter), Realization::Sketch(kind)) = (strictest_sibling, &realization) { let (eps, delta) = accuracy_budget(stricter); let algorithm = kind.algorithm().clone(); - let params = - planning_inputs - .cost - .size_params(algorithm.clone(), intent, eps, delta); + let params = size_params(algorithm.clone(), intent, eps, delta); if params != *kind.params() { proposals.record( format!( @@ -1516,17 +1537,17 @@ impl<'a> SketchAlgorithmStrategy<'a> { let Some(child) = aggregate_child(root) else { continue; }; - let QueryExpr::Aggregate { reduction, .. } = root.as_ref() else { + let Some(NonASAPOp::Aggregate { reduction, .. }) = root.non_asap() else { continue; }; let Ok(input) = realize_physical_summary_input(intent, &family, reduction, child) else { continue; }; - let readout_query = readout(intent, &input.input, planning_inputs.cost); + let evaluation_query = evaluation(intent, &input.input); let Some(local) = planning_inputs .accuracy - .local_guarantee(&family, &readout_query) + .local_guarantee(&family, &evaluation_query) else { continue; }; @@ -1537,7 +1558,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { let allocations = planning_inputs.allocator.allocations(target, &shape); if allocations.is_empty() { proposals.rejected.push(RejectedCandidate { - strategy: "SketchAlgorithmStrategy", + strategy: "ASAPStrategies", description: rationale.clone(), error: AccuracyError::NoLegalAllocation { target: target.clone(), @@ -1555,9 +1576,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { let (eps, delta) = accuracy_budget(outer_target); let resized = Realization::Sketch(SketchKind::new( kind.algorithm().clone(), - planning_inputs - .cost - .size_params(kind.algorithm().clone(), intent, eps, delta), + size_params(kind.algorithm().clone(), intent, eps, delta), )); // Identical to the as-declared composition already recorded // above — nothing new to propose. @@ -1591,10 +1610,10 @@ impl<'a> SketchAlgorithmStrategy<'a> { } if proposals.candidates.is_empty() { if let Some(error) = &proposals.domain_error { - if let Ok(node) = keep_pre_asap(root) { + if let Ok(node) = retain_exact(root) { proposals.candidates.push(ReplacementSubDAG { - strategy: "SketchAlgorithmStrategy", - replacement: Replacement::Summary(node), + strategy: "ASAPStrategies", + replacement: Replacement::SubDAG(node), provenance: ReplacementProvenance::SummaryRealization, rationale: format!( "{} stays pre-ASAP because summary construction crosses an illegal \ @@ -1613,16 +1632,16 @@ impl Proposals { /// File one construction attempt: a legal node becomes a candidate, an /// [`RealizationError::Accuracy`] becomes a [`RejectedCandidate`], and a /// schema-derivation failure is skipped exactly as it always was. - fn record(&mut self, rationale: String, built: Result, RealizationError>) { + fn record(&mut self, rationale: String, built: Result, RealizationError>) { match built { Ok(node) => self.candidates.push(ReplacementSubDAG { - strategy: "SketchAlgorithmStrategy", - replacement: Replacement::Summary(node), + strategy: "ASAPStrategies", + replacement: Replacement::SubDAG(node), provenance: ReplacementProvenance::SummaryRealization, rationale, }), Err(RealizationError::Accuracy(error)) => self.rejected.push(RejectedCandidate { - strategy: "SketchAlgorithmStrategy", + strategy: "ASAPStrategies", description: rationale, error, }), @@ -1639,14 +1658,14 @@ impl Proposals { } /// The `child` of a [`bindable_intent`]-shaped `Aggregate`. -fn aggregate_child(node: &QueryExpr) -> Option<&Rc> { - match node { - QueryExpr::Aggregate { child, .. } => Some(child), +fn aggregate_child(node: &OperatorNode) -> Option<&Rc> { + match node.non_asap() { + Some(NonASAPOp::Aggregate { child, .. }) => Some(child), _ => None, } } -impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { +impl ReplacementStrategy for ASAPStrategies<'_> { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { bindable_intent(target.root).is_some() || is_supported_exact_binary(target.root) } @@ -1664,25 +1683,25 @@ impl ReplacementStrategy for SketchAlgorithmStrategy<'_> { /// the logical root does not expose, so each is a finalized query result /// for the identity-carrying root. Placement variants (for example, /// fixed-window or query-time Rate aggregation) are not listed here: the - /// lifecycle assigns timing and the physical compiler reads it. - fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { - let Ok(typed) = asap_types::pre_asap::schema::with_promql_series_identity(root) else { + /// materialization assigns timing and the physical compiler reads it. + fn propose_for_root(&self, root: &Rc, target: &AccuracyTarget) -> Proposals { + let Ok(typed) = asap_types::ir::schema_support::with_promql_series_identity(root) else { return Proposals::default(); }; - let typed = Rc::new(typed); + let mut proposals = self.current_series_topk_candidates(&typed, target); for mut candidate in std::mem::take(&mut proposals.candidates) { - let Replacement::Summary(node) = candidate.replacement else { + let Replacement::SubDAG(node) = candidate.replacement else { continue; }; let Ok(node) = finalize_query_candidate(node, &typed) else { continue; }; let duplicate = proposals.candidates.iter().any(|existing| { - matches!(&existing.replacement, Replacement::Summary(other) if *other == node) + matches!(&existing.replacement, Replacement::SubDAG(other) if *other == node) }); if !duplicate { - candidate.replacement = Replacement::Summary(node); + candidate.replacement = Replacement::SubDAG(node); candidate.provenance = ReplacementProvenance::RootPhysicalRealization; proposals.candidates.push(candidate); } @@ -1697,7 +1716,7 @@ fn describe_realization(intent: &AggIntent, realization: &Realization) -> String match realization { Realization::Sketch(kind) => format!( "{} realizes as a {:?} sketch — one of summary_candidates' \ - candidates for this intent (asap_aware_mapping::replacement::realizations_for_intent)", + candidates for this intent (asap_logical_optimizer::pass1::replacement::realizations_for_intent)", describe_intent(intent), kind.algorithm() ), @@ -1713,18 +1732,18 @@ fn describe_realization(intent: &AggIntent, realization: &Realization) -> String describe_intent(intent) ), Realization::Sample { kind, .. } => format!( - "{} realizes as a {kind:?} sample — the only realization the plugged-in \ - CostModel produced for this intent", + "{} realizes as a {kind:?} sample — the only realization produced for \ + this intent", describe_intent(intent) ), Realization::Wavelet { kind, .. } => format!( - "{} realizes as a {kind:?} wavelet transform — the only realization the \ - plugged-in CostModel produced for this intent", + "{} realizes as a {kind:?} wavelet transform — the only realization \ + produced for this intent", describe_intent(intent) ), Realization::StatModel { kind, .. } => format!( - "{} realizes as a {kind:?} statistical model — the only realization the \ - plugged-in CostModel produced for this intent", + "{} realizes as a {kind:?} statistical model — the only realization \ + produced for this intent", describe_intent(intent) ), } @@ -1735,7 +1754,7 @@ fn describe_realization(intent: &AggIntent, realization: &Realization) -> String /// this crate's other `AggIntent` matches, e.g. [`realizations_for_intent`]'s) /// — this is prose for a rationale string, not a decision, so an unlisted /// variant just falls back to its `Debug` tag rather than forcing every -/// future intent to be named here too. [`crate::explanation`] needs no +/// future intent to be named here too. [`crate::pass1::explanation`] needs no /// counterpart of its own: it reads a candidate's `rationale` — built from /// this text — straight off [`ReplacementSubDAG`], rather than re-describing /// the same intent a second time. @@ -1755,37 +1774,24 @@ pub(crate) fn describe_intent(intent: &AggIntent) -> String { } } -// ── realize_child / keep_pre_asap: rank-and-take-first, and its fallback ── +// ── realize_child / retain_exact: rank-and-take-first, and its fallback ── -/// Rank-and-take-first selector for a single [`QueryExpr`] node: enumerate -/// every candidate via [`SketchAlgorithmStrategy::replacements`], keep the -/// `cost_model`-preferred (first) one, and fall back to [`keep_pre_asap`] +/// Rank-and-take-first selector for a single [`OperatorNode`]: enumerate +/// every candidate via [`ASAPStrategies::replacements`], keep the +/// `cost_model`-preferred (first) one, and fall back to [`retain_exact`] /// when there's no candidate at all — **not** a general single-answer API -/// for a whole workload. Use [`CandidateLogicalASAPDAGs::global_selection`] and DAG assembly +/// for a whole workload. Use `candidate_selection::global_selection` and DAG assembly /// for coordinated logical selection; physical deployment remains downstream. /// `root` must already be the caller's own /// `Rc`, never fabricated per call, so this never allocates beyond what the /// caller already held. /// -/// `pub(crate)`: reachable from this module's own construction helper -/// ([`construct_summary_agg`], so a nested aggregate gets its own -/// independent enumeration instead of inheriting the parent's forced -/// candidate), from this module's own [`realize_one`] (the representative -/// bound `SummaryNode` [`cse_preference`] needs for a -/// [`CostModel::cse_share_decision`] comparison), and from -/// [`crate::cost_model::DefaultCostModel::estimate_cost`] (the same -/// representative-node need, for a [`Replacement::Rewrite`] candidate's own -/// cost estimate). Every other caller goes through -/// [`SketchAlgorithmStrategy::replacements`] directly and decides for itself. -pub(crate) fn realize_child( - root: &Rc, - cost_model: &dyn CostModel, -) -> Result, RealizationError> { - realize_child_with( - root, - CandidatePlanningInputs::with_default_accuracy(cost_model), - None, - ) +/// Public because Stage 3 cost models need one representative bound node for +/// a target: `DefaultCostModel::estimate_cost` and the legacy CSE ranking in +/// `plan_selection::candidate_selection`. Every other caller goes through +/// [`ASAPStrategies::replacements`] directly and decides for itself. +pub fn realize_child(root: &Rc) -> Result, RealizationError> { + realize_child_with(root, CandidatePlanningInputs::with_default_accuracy(), None) } /// [`realize_child`] with every model explicit, plus an optional @@ -1797,17 +1803,17 @@ pub(crate) fn realize_child( /// budget. A child whose declared target is `Exact` keeps it: an allocation /// never approximates something the caller declared exact. fn exact_topk_over_temporal_values( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, -) -> Result>, RealizationError> { - let QueryExpr::Aggregate { +) -> Result>, RealizationError> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names: _, filters, having: None, child, - } = root.as_ref() + }) = root.non_asap() else { return Ok(None); }; @@ -1817,19 +1823,19 @@ fn exact_topk_over_temporal_values( let [AggIntent::TopK { k, .. }] = measures.as_slice() else { return Ok(None); }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, child: input, .. - } = child.as_ref() + }) = child.non_asap() else { return Ok(None); }; - if !matches!(input.as_ref(), QueryExpr::TimeRange { .. }) { + if !matches!(input.non_asap(), Some(NonASAPOp::TimeRange { .. })) { return Ok(None); } let values = realize_child_with(child, planning_inputs, Some(&AccuracyTarget::Exact))?; - if matches!(values.expr, SummaryExpr::KeepPreAsap(_)) + if !values.contains_asap() || !values .guarantee .as_ref() @@ -1845,61 +1851,63 @@ fn exact_topk_over_temporal_values( ))? .clone(); let score = ranking_score_index(child, &values.schema)?; - let sorted = Rc::new(SummaryNode { - guarantee: values.guarantee.clone(), - schema: values.schema.clone(), - expr: SummaryExpr::ValueOperation { - child: values, - operation: ValueOperation::Sort { - keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + let guarantee = values.guarantee.clone(); + let schema = values.schema.clone(); + let sorted = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Sort { + keys: vec![SortKey { + expr: ScalarExpr::Column(score), ascending: false, nulls_first: false, }], partition_by: partition_by.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - }); - let node = Rc::new(SummaryNode { - guarantee: sorted.guarantee.clone(), - schema: sorted.schema.clone(), - expr: SummaryExpr::ValueOperation { - child: sorted, - operation: ValueOperation::Limit { - n: *k, + child: values, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let node = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Limit { + n: Some(*k), offset: 0, partition_by, - }, - timing: ExecutionTiming::QueryTime, - }, - }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + child: sorted, + }), + schema, + ) + .with_guarantee(guarantee), + ); + validate_maintained(&node, ExecutionTiming::QueryTime)?; Ok(Some(node)) } fn realize_temporal_average( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { - let Some(components) = crate::rewrite::temporal_average_components(root) else { +) -> Result>, RealizationError> { + let Some(components) = crate::pass1::rewrite::temporal_average_components(root) else { return Ok(None); }; let mut node = realize_child_with(&components, planning_inputs, target)?; - let SummaryExpr::BinaryOp { operator, .. } = &mut Rc::make_mut(&mut node).expr else { + let Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) = + &mut Rc::make_mut(&mut node).operator + else { return Ok(None); }; operator.checked_finite_division = true; - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + validate_maintained(&node, ExecutionTiming::QueryTime)?; Ok(Some(node)) } pub(crate) fn realize_child_with( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { +) -> Result, RealizationError> { if let Some(node) = realize_temporal_average(root, planning_inputs, end_to_end_target)? { return Ok(node); } @@ -1913,29 +1921,29 @@ pub(crate) fn realize_child_with( Some(_) => Some(override_accuracy(declared, target)), } }); - match SketchAlgorithmStrategy::from_planning_inputs(planning_inputs) + match ASAPStrategies::from_planning_inputs(planning_inputs) .propose_with(root, overridden.as_ref(), None) .candidates .into_iter() .next() { Some(ReplacementSubDAG { - replacement: Replacement::Summary(node), + replacement: Replacement::SubDAG(node), .. }) => Ok(node), Some(ReplacementSubDAG { - replacement: Replacement::Rewrite(_) | Replacement::ExactComposition(_), + replacement: Replacement::ExactComposition(_), .. }) => { - unreachable!("SketchAlgorithmStrategy never returns a Rewrite/composition candidate") + unreachable!("ASAPStrategies never returns a composition candidate") } // No candidate at all: `root` isn't `bindable_intent` shape (or its // intent has no realization `realizations_for_intent` can't // produce — never happens, that match is exhaustive), or every // candidate was accuracy-illegal — either way the same conservative - // fallback `SketchAlgorithmStrategy::matches` uses: keep the + // fallback `ASAPStrategies::matches` uses: keep the // pre-ASAP sub-DAG, executed exactly. - None => keep_pre_asap(root), + None => retain_exact(root), } } @@ -1944,29 +1952,24 @@ pub(crate) fn realize_child_with( /// accelerated, return `None` so the caller keeps the whole query exact; /// mixed raw/summary snapshots are never constructed. fn realize_binary( - root: &Rc, + root: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result>, RealizationError> { - let QueryExpr::BinaryOp { - op, +) -> Result>, RealizationError> { + let Some(NonASAPOp::BinaryOp { + operator, + return_bool, lhs, rhs, - vector_match, - } = root.as_ref() + }) = root.non_asap() else { return Ok(None); }; + let (op, vector_match) = (&operator.kind, &operator.vector_match); if !matches!(op, BinaryOpKind::Arithmetic(_)) || vector_match.is_some() { return Ok(None); } - let lhs_scalar = is_promql_scalar(lhs); - let rhs_scalar = is_promql_scalar(rhs); - if lhs_scalar && rhs_scalar { - return Ok(None); - } - let mut lhs_node = realize_binary_operand(lhs, planning_inputs, None)?; let mut rhs_node = realize_binary_operand(rhs, planning_inputs, None)?; @@ -2085,8 +2088,8 @@ fn realize_binary( return Ok(None); } - let lhs_accelerated = lhs_scalar || !matches!(lhs_node.expr, SummaryExpr::KeepPreAsap(_)); - let rhs_accelerated = rhs_scalar || !matches!(rhs_node.expr, SummaryExpr::KeepPreAsap(_)); + let lhs_accelerated = lhs_node.contains_asap(); + let rhs_accelerated = rhs_node.contains_asap(); if !lhs_accelerated || !rhs_accelerated { return Ok(None); } @@ -2133,47 +2136,130 @@ fn realize_binary( return Ok(None); } - Ok(Some(Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: ExecutionTiming::QueryTime, - lhs: lhs_node, - rhs: rhs_node, - operator: asap_types::post_asap::BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: op.clone(), - vector_match: vector_match.clone(), - }, - }, - schema: lift(&root.output_schema()?), + Ok(Some(Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: op.clone(), + vector_match: vector_match.clone(), + }, + return_bool: *return_bool, + lhs: lhs_node, + rhs: rhs_node, + }), + root.schema.clone(), + ) // Exact arithmetic does not erase approximation error. Until the // accuracy algebra has an operator-specific rule (and any value-range // evidence needed by multiplication/division), unknown stays unknown. - guarantee, - }))) + .with_guarantee(guarantee), + ))) +} + +/// Coverage of a summary built over `child`: every observation of the one +/// source scanned beneath it. Today's planner proves no time or population +/// restriction, so this whole-source declaration is trusted, not derived from +/// the scan (#570). `None` when `child` does not read exactly one source. +pub(crate) fn whole_source_coverage(child: &Rc) -> Option { + let mut sources = OperatorNode::reachable(child) + .into_iter() + .filter_map(|node| match node.non_asap() { + Some(NonASAPOp::Scan { source, .. }) => Some(source.clone()), + _ => None, + }); + let source = sources.next()?; + sources + .all(|other| other == source) + .then(|| SummaryCoverage { + source, + regions: vec![CoverageRegion { + time_ms: None, + population: Default::default(), + }], + }) +} + +/// Rebuild the summary chain above a per-series `Rate` accumulator with its +/// `FinalizeExactAccumulator` placed at `timing`. `strict` additionally +/// requires the fixed-window shape (a `PerEntity` Rate over a `TimeRange`); +/// `None` when no such boundary exists (strict only). +fn retime_rate_finalize( + node: &Rc, + timing: ExecutionTiming, + strict: bool, +) -> Option> { + let is_rate_boundary = |child: &OperatorNode| match &child.operator { + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(ExactKind::Rate, _), + reduction, + child: source, + .. + }) => { + !strict + || (matches!(reduction, Reduction::PerEntity) + && matches!(source.non_asap(), Some(NonASAPOp::TimeRange { .. }))) + } + _ => false, + }; + let rebuilt = |operator: Operator, timing: Option| { + Rc::new(OperatorNode { + operator, + timing, + ..node.as_ref().clone() + }) + }; + match &node.operator { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_rate_boundary(child) => { + Some(rebuilt(node.operator.clone(), Some(timing))) + } + Operator::ASAP( + ASAPOp::FinalizeExactAccumulator { child } + | ASAPOp::SummaryAgg { child, .. } + | ASAPOp::SummaryEstimate { + summary_input: child, + .. + }, + ) => { + let placed = match retime_rate_finalize(child, timing, strict) { + Some(placed) => placed, + None if strict => return None, + None => return Some(Rc::clone(node)), + }; + let operator = node.operator.map_children(|_| Rc::clone(&placed)); + Some(rebuilt(operator, node.timing)) + } + _ if strict => None, + _ => Some(Rc::clone(node)), + } } /// Put an explicit read boundary between maintained exact state and a /// query-time value consumer. Approximate summaries must already carry a /// `SummaryEstimate`, so they deliberately do not pass this predicate. pub fn finalize_query_candidate( - node: Rc, - logical_output: &QueryExpr, -) -> Result, RealizationError> { - finalize_exact_accumulator_at(node, logical_output, ExecutionTiming::QueryTime) -} - -fn finalize_exact_accumulator_at( - node: Rc, - logical_output: &QueryExpr, - timing: ExecutionTiming, -) -> Result, RealizationError> { + node: Rc, + logical_output: &OperatorNode, +) -> Result, RealizationError> { + finalize_exact_accumulator(node, logical_output, ExecutionTiming::QueryTime) +} + +/// The read boundary's placement is fixed here, where the candidate's +/// semantics decide it (a fresh query-time summary over this evaluation's +/// finalized values vs. finalized values feeding maintenance); the materialization +/// timing pass honors it. +fn finalize_exact_accumulator( + node: Rc, + logical_output: &OperatorNode, + placement: ExecutionTiming, +) -> Result, RealizationError> { let is_exact_state = matches!( - node.expr, - SummaryExpr::SummaryAgg { + node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::ExactAggregate(..), .. - } + }) ); if !is_exact_state { return Ok(node); @@ -2182,41 +2268,36 @@ fn finalize_exact_accumulator_at( // boundary produces the logical operator's ordinary values. Preserve the // canonical pre-ASAP output types instead of leaking ExactAggregate into // query-time operators that follow this node. - let schema = lift(&logical_output.output_schema()?); + let schema = logical_output.schema.clone(); let guarantee = node.guarantee.clone(); - Ok(Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: node, - operation: ValueOperation::FinalizeExactAccumulator, - timing, - }, - schema, - guarantee, - })) + Ok(Rc::new( + OperatorNode::with_schema( + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: node }), + schema, + ) + .with_guarantee(guarantee) + .with_timing(Some(placement)), + )) } -fn is_supported_exact_binary(root: &QueryExpr) -> bool { +fn is_supported_exact_binary(root: &OperatorNode) -> bool { matches!( - root, - QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(_), - vector_match: None, + root.non_asap(), + Some(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(_), + vector_match: None, + .. + }, .. - } - ) -} - -fn is_promql_scalar(expr: &QueryExpr) -> bool { - matches!( - expr, - QueryExpr::PromqlScalarBridge(_) | QueryExpr::Literal(_) + }) ) } /// Quantile operands inherit one workload target. A temporal mean is exact /// on its checked finite domain and needs no approximation budget. -fn shared_quantile_target(lhs: &QueryExpr, rhs: &QueryExpr) -> Option { - let quantile_target = |expr: &QueryExpr| match bindable_intent(expr) { +fn shared_quantile_target(lhs: &OperatorNode, rhs: &OperatorNode) -> Option { + let quantile_target = |expr: &OperatorNode| match bindable_intent(expr) { Some(AggIntent::Quantile { accuracy, q, .. }) if q.is_finite() && (0.0..=1.0).contains(q) => { @@ -2254,18 +2335,18 @@ fn ddsketch_ratio_operand_target(target: &AccuracyTarget) -> Option Option { - let SummaryExpr::SummaryEstimate { +fn ddsketch_quantile_alpha(node: &OperatorNode) -> Option { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query: PostAsapSketchStatistic::Quantile { .. }, - } = &node.expr + }) = &node.operator else { return None; }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = &summary_input.operator else { return None; }; @@ -2275,7 +2356,7 @@ fn ddsketch_quantile_alpha(node: &SummaryNode) -> Option { } } -fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { +fn has_missing_accuracy_evidence(node: &OperatorNode) -> bool { node.guarantee .as_ref() .is_none_or(ResultGuarantee::has_unknown) @@ -2284,10 +2365,10 @@ fn has_missing_accuracy_evidence(node: &SummaryNode) -> bool { /// A direct ratio has an operator-specific DDSketch proof, so it must select /// DDSketch rather than the cost model's generally preferred KLL candidate. fn realize_ddsketch_quantile_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, target: &AccuracyTarget, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let intent = bindable_intent(operand).and_then(|intent| match intent { AggIntent::Quantile { .. } => Some(override_accuracy(intent, target)), _ => None, @@ -2298,25 +2379,16 @@ fn realize_ddsketch_quantile_operand( let (epsilon, delta) = accuracy_budget(target); let realization = Realization::Sketch(SketchKind::new( SketchAlgorithm::DDSketch, - planning_inputs - .cost - .size_params(SketchAlgorithm::DDSketch, &intent, epsilon, delta), + size_params(SketchAlgorithm::DDSketch, &intent, epsilon, delta), )); construct_summary_with(operand, &intent, realization, planning_inputs, None, None) } fn realize_binary_operand( - operand: &Rc, + operand: &Rc, planning_inputs: CandidatePlanningInputs<'_>, end_to_end_target: Option<&AccuracyTarget>, -) -> Result, RealizationError> { - if is_promql_scalar(operand) { - return Ok(Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::clone(operand)), - schema: Schema::lifted(Vec::new(), None), - guarantee: Some(ResultGuarantee::exact("PromQL scalar")), - })); - } +) -> Result, RealizationError> { realize_child_with(operand, planning_inputs, end_to_end_target) } @@ -2334,43 +2406,77 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { out } -/// Wrap an unrewritten pre-ASAP sub-DAG, lifting its schema with every column -/// `FieldDataType::Plain`. `pub` so a caller can fall back to this -/// explicitly — e.g. when `SketchAlgorithmStrategy::replacements()` returns no -/// candidate for a target, or a deployment wants to force a node its own -/// runtime can't actually implement — through the same fallback this -/// crate's own dispatch uses, without duplicating the schema-lift logic. -pub fn keep_pre_asap(expr: &Rc) -> Result, RealizationError> { - keep_pre_asap_rc(Rc::clone(expr)) -} - -fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationError> { - let schema = expr.output_schema()?; - Ok(Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(expr), - schema: lift(&schema), - // A kept pre-ASAP sub-DAG is executed exactly by the runtime - // (`Realization::PassThrough`'s contract) — zero error. - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), - })) +/// Keep an unrewritten pre-ASAP sub-DAG as it is. There is no wrapper node: +/// the sub-DAG itself is the plan, carrying an exact guarantee. The same +/// `Rc` is returned when the node already has a guarantee; otherwise a copy +/// with `guarantee = exact("RetainedExact")` — only for a sub-DAG with no +/// ASAP operator (a sub-DAG containing one keeps whatever its construction +/// established). `pub` so a caller can fall back to this explicitly — e.g. +/// when `ASAPStrategies::replacements()` returns no candidate for a +/// target, or a deployment wants to force a node its own runtime can't +/// actually implement — through the same fallback this crate's own dispatch +/// uses. +pub fn retain_exact(expr: &Rc) -> Result, RealizationError> { + retain_exact_rc(Rc::clone(expr)) +} + +fn retain_exact_rc(expr: Rc) -> Result, RealizationError> { + if expr.guarantee.is_some() || expr.contains_asap() { + return Ok(expr); + } + // Keeping the same sub-DAG twice (e.g. one `Scan` read by an exact + // aggregate and by a sketch, or by two candidates) must yield one node: + // sharing is pointer identity. Memoize the kept copy per input node while + // both are alive; weak references keep the memo from extending lifetimes + // or matching a reused address. + type KeptMemo = HashMap<*const OperatorNode, (Weak, Weak)>; + thread_local! { + static KEPT: RefCell = RefCell::new(HashMap::new()); + } + let key = Rc::as_ptr(&expr); + if let Some(kept) = KEPT.with(|memo| { + memo.borrow().get(&key).and_then(|(input, kept)| { + input + .upgrade() + .filter(|input| Rc::ptr_eq(input, &expr)) + .and_then(|_| kept.upgrade()) + }) + }) { + return Ok(kept); + } + let kept = Rc::new( + expr.as_ref() + .clone() + // A kept pre-ASAP sub-DAG is executed exactly by the runtime + // (`Realization::PassThrough`'s contract) — zero error. + .with_guarantee(Some(ResultGuarantee::exact("RetainedExact"))), + ); + KEPT.with(|memo| { + let mut memo = memo.borrow_mut(); + if memo.len() > 4096 { + memo.retain(|_, (input, kept)| input.strong_count() > 0 && kept.strong_count() > 0); + } + memo.insert(key, (Rc::downgrade(&expr), Rc::downgrade(&kept))); + }); + Ok(kept) } -// ── Construction: turn one already-decided Realization into a SummaryNode ─ +// ── Construction: turn one already-decided Realization into an OperatorNode ─ -/// The bindable shape [`SketchAlgorithmStrategy`] targets: a single intent, no +/// The bindable shape [`ASAPStrategies`] targets: a single intent, no /// `HAVING`. A multi-intent node (SQL `SELECT SUM(a), AVG(b)`), or one with a /// `HAVING` predicate (the filter would need the estimate first), stays -/// logical. Unsupported logical parents still conservatively become one -/// [`SummaryExpr::KeepPreAsap`] sub-DAG. Composable query-time value -/// operators (`Project`, `Filter`, `Sort`, and `Limit`) are retained during final -/// DAG assembly so their independently planned children remain visible. -pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { - if let QueryExpr::Aggregate { +/// logical. Unsupported logical parents are conservatively kept as pre-ASAP +/// sub-DAGs ([`retain_exact`]). Relational operators are retained during +/// final DAG assembly so their independently planned children remain +/// visible. +pub fn bindable_intent(node: &OperatorNode) -> Option<&AggIntent> { + if let Some(NonASAPOp::Aggregate { measures, filters, having, .. - } = node + }) = node.non_asap() { if let ([intent], None) = (measures.as_slice(), having) { if !any_measure_filtered(filters) { @@ -2382,7 +2488,7 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { } /// `expr` must still be the [`bindable_intent`] shape for `realization` to -/// have any effect; anything else falls back to [`keep_pre_asap`]. +/// have any effect; anything else falls back to [`retain_exact`]. /// Only `expr`'s own top-level decision is forced — recursion into `expr`'s /// child goes back through [`realize_child`] (fresh candidate /// enumeration, not a forced pick), so choosing one candidate for a target @@ -2390,10 +2496,10 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// /// `pub(crate)`: `grouping::HydraGroupingStrategy` (issue #256) is the one /// caller outside this module — the same first-class, -/// one-candidate-at-a-time primitive [`SketchAlgorithmStrategy`] itself +/// one-candidate-at-a-time primitive [`ASAPStrategies`] itself /// calls once per candidate, reused rather than duplicated so a Hydra /// candidate gets exactly the same schema derivation/column -/// resolution/readout construction as every other candidate, patching only +/// resolution/evaluation construction as every other candidate, patching only /// the `grouping` field this axis owns. /// Construct a summary with every model explicit (issue #172). `intent` /// is `expr`'s own [`bindable_intent`], or a copy of it with an allocated @@ -2404,13 +2510,13 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// fail-closed answer for a composition with no sound rule or one that /// misses `intent`'s target. pub(crate) fn construct_summary_with( - expr: &QueryExpr, + expr: &OperatorNode, intent: &AggIntent, realization: Realization, planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let local_target = match allocation.as_ref() { Some(GuaranteeSource::BudgetAllocation { local_target, .. }) => Some(local_target), _ => accuracy_target(intent), @@ -2427,9 +2533,9 @@ pub(crate) fn construct_summary_with( }, other => other, }; - if let QueryExpr::Aggregate { + if let Some(NonASAPOp::Aggregate { reduction, child, .. - } = expr + }) = expr.non_asap() { // `bindable_intent` already established the shape: exactly one // intent, no HAVING. (Multi-intent nodes and HAVING stay logical.) @@ -2454,28 +2560,28 @@ pub(crate) fn construct_summary_with( } } } - keep_pre_asap_rc(Rc::new(expr.clone())) + retain_exact_rc(Rc::new(expr.clone())) } fn finish_weighted_topk( - candidate: Rc, - logical: &QueryExpr, + candidate: Rc, + logical: &OperatorNode, intent: &AggIntent, -) -> Result, RealizationError> { +) -> Result, RealizationError> { let AggIntent::TopK { k, .. } = intent else { unreachable!() }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(groups), child, .. - } = logical + }) = logical.non_asap() else { return Err(RealizationError::PhysicalRealization( "TopK requires explicit grouping", )); }; - let schema = lift(&child.output_schema()?); + let schema = child.schema.clone(); let score = ranking_score_index(child, &schema)?; let cols = schema .fields @@ -2502,84 +2608,81 @@ fn finish_weighted_topk( } } }; - Ok(asap_types::pre_asap::query_expr::ProjectItem { + Ok(ProjectItem { alias: Some(field.name.clone()), - expr: QueryExpr::Column(source), + expr: ScalarExpr::Column(source), }) }) .collect::, _>>()?; let guarantee = candidate.guarantee.clone(); - let projected = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: candidate, - operation: ValueOperation::Project { + let projected = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Project { cols, qualifier: None, - }, - timing: ExecutionTiming::QueryTime, - }, - schema: schema.clone(), - guarantee: guarantee.clone(), - }); - let sorted = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: projected, - operation: ValueOperation::Sort { - keys: vec![asap_types::pre_asap::SortKey { - expr: QueryExpr::Column(score), + child: candidate, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let sorted = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Sort { + keys: vec![SortKey { + expr: ScalarExpr::Column(score), ascending: false, nulls_first: false, }], partition_by: groups.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema: schema.clone(), - guarantee: guarantee.clone(), - }); - let result = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: sorted, - operation: ValueOperation::Limit { - n: *k, + child: projected, + }), + schema.clone(), + ) + .with_guarantee(guarantee.clone()), + ); + let result = Rc::new( + OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Limit { + n: Some(*k), offset: 0, partition_by: groups.clone(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema, - guarantee, - }); - validate_execution_data_states_at(&result, ExecutionDataState::QUERY_ROWS)?; + child: sorted, + }), + schema, + ) + .with_guarantee(guarantee), + ); + validate_maintained(&result, ExecutionTiming::QueryTime)?; Ok(result) } -fn is_current_series_source(child: &QueryExpr) -> bool { - let source = match child { - QueryExpr::TimeRange { child, .. } => child.as_ref(), - source => source, +fn is_current_series_source(child: &OperatorNode) -> bool { + let source = match child.non_asap() { + Some(NonASAPOp::TimeRange { child, .. }) => child.as_ref(), + _ => child, }; - matches!(source, QueryExpr::Scan { - source: asap_types::pre_asap::Source::TimeSeries { .. }, schema, .. - } if schema.has_promql_series_identity()) + matches!(source.non_asap(), Some(NonASAPOp::Scan { + source: asap_types::ir::operator::Source::TimeSeries { .. }, schema, .. + }) if schema.has_promql_series_identity()) } -fn is_snapshot_weighted_topk(intent: &AggIntent, child: &QueryExpr) -> bool { +fn is_snapshot_weighted_topk(intent: &AggIntent, child: &OperatorNode) -> bool { matches!(intent, AggIntent::TopK { .. }) && (is_current_series_source(child) - || matches!(child, - QueryExpr::Aggregate { measures, child, .. } + || matches!(child.non_asap(), + Some(NonASAPOp::Aggregate { measures, child, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]) || (matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(child.non_asap(), Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase]))))) } /// Translate an [`Realization`] into the `(family, needs a -/// SummaryEstimate readout)` pair [`construct_summary_agg`] needs, or `None` -/// for `PassThrough` (the caller falls back to [`keep_pre_asap`]). +/// SummaryEstimate evaluation)` pair [`construct_summary_agg`] needs, or `None` +/// for `PassThrough` (the caller falls back to [`retain_exact`]). /// -/// Every family's partial state needs a readout to recover a value, except +/// Every family's partial state needs a evaluation to recover a value, except /// `ExactAggregate` — its partial state *is* the value already, so no /// estimate step follows it. fn summary_family(realization: Realization) -> Option<(FieldDataType, bool)> { @@ -2602,19 +2705,19 @@ fn summary_family(realization: Realization) -> Option<(FieldDataType, bool)> { /// consume the logical aggregate's immediate child and summarize its declared /// input value. Composite realizations can instead consume a larger /// logical sub-DAG and bind a different key or value. -struct PhysicalSummaryInput { - child: Rc, - input: SummaryUpdate, +pub(crate) struct PhysicalSummaryInput { + pub(crate) child: Rc, + pub(crate) input: SummaryUpdate, } -enum PhysicalSummaryInputRuleResult { +pub(crate) enum PhysicalSummaryInputRuleResult { NotApplicable, Realized(PhysicalSummaryInput), Unsupported(&'static str), } type PhysicalSummaryInputRule = - fn(&AggIntent, &FieldDataType, &Reduction, &Rc) -> PhysicalSummaryInputRuleResult; + fn(&AggIntent, &FieldDataType, &Reduction, &Rc) -> PhysicalSummaryInputRuleResult; /// Ordered physical-realization rules for realizations that consume more /// than the immediate logical input. New composite primitives add a rule here @@ -2631,7 +2734,7 @@ fn realize_value_frequency_summary_input( intent: &AggIntent, family: &FieldDataType, _reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { // Frequency counts hash sample values as items but add one per observation. // Using the sample as a weight would turn counts into sums and admit signed CMS updates. @@ -2642,11 +2745,7 @@ fn realize_value_frequency_summary_input( { return PhysicalSummaryInputRuleResult::NotApplicable; } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "value frequency input needs a valid schema", - ); - }; + let schema = &child.schema; // One item per observation is a single value stream. `summary_candidates` // already withholds UnivMon from a distinct-tuple count; refused here too // so the invariant does not rest on that table alone. @@ -2658,7 +2757,7 @@ fn realize_value_frequency_summary_input( PhysicalSummaryInputRuleResult::Realized(PhysicalSummaryInput { child: Rc::clone(child), input: SummaryUpdate { - item: Some(SummaryInputExpr::Column(summarised_column(intent, &schema))), + item: Some(SummaryInputExpr::Column(summarised_column(intent, schema))), weight: SummaryInputExpr::Constant(1.0), weight_domain: WeightDomain::NonNegative { proof: NonNegativeWeightProof::UnitCount, @@ -2671,7 +2770,7 @@ fn realize_physical_summary_input( intent: &AggIntent, family: &FieldDataType, reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> Result { for rule in PHYSICAL_SUMMARY_INPUT_RULES { match rule(intent, family, reduction, child) { @@ -2683,7 +2782,7 @@ fn realize_physical_summary_input( } } - let child_schema = child.output_schema()?; + let child_schema = &child.schema; if matches!(intent, AggIntent::TopK { .. }) { return Err(RealizationError::PhysicalRealization( "Top-K needs an explicit item identity and additive update input", @@ -2693,76 +2792,70 @@ fn realize_physical_summary_input( child: Rc::clone(child), input: SummaryUpdate { item: None, - weight: summarised_input(intent, &child_schema)?, + weight: summarised_input(intent, child_schema)?, weight_domain: WeightDomain::UnknownOrSigned, }, }) } /// Emit `SummaryAgg` (recursively binding the child), plus the -/// `SummaryEstimate` readout when `estimate` is set. +/// `SummaryEstimate` evaluation when `estimate` is set. // Retain the exact expression and schema while placing its value production -// on the update path. This is the initial layout for values feeding a summary; -// lifecycle timing is authoritative. Read-time consumers keep their original -// shared nodes. -fn maintenance_exact_values(node: Rc) -> Option> { - let expr = match &node.expr { +// on the update path (a node runs when its consumer runs, so beneath a +// maintained summary this value production is ingestion-time work). +// Read-time consumers keep their original shared nodes. +fn maintenance_exact_values(node: Rc) -> Option> { + let operator = match &node.operator { // These guards can fall back at read time, but cannot recover a parent // sketch after an invalid value has entered its maintained state. - SummaryExpr::BinaryOp { operator, .. } + Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) if operator.checked_finite_division || operator.checked_relative_division => { return None; } - SummaryExpr::BinaryOp { - lhs, rhs, operator, .. - } if operator.vector_match.is_none() - && matches!( - operator.kind, - asap_types::pre_asap::BinaryOpKind::Arithmetic(_) - ) + Operator::NonASAP(NonASAPOp::BinaryOp { + lhs, + rhs, + operator, + return_bool, + }) if operator.vector_match.is_none() + && matches!(operator.kind, BinaryOpKind::Arithmetic(_)) && node .guarantee .as_ref() .is_some_and(ResultGuarantee::is_exact) => { - SummaryExpr::BinaryOp { + Operator::NonASAP(NonASAPOp::BinaryOp { lhs: maintenance_exact_values(lhs.clone())?, rhs: maintenance_exact_values(rhs.clone())?, operator: operator.clone(), - timing: ExecutionTiming::IngestionTime, - } + return_bool: *return_bool, + }) } - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } if matches!( - child.expr, - SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(..), - .. - } - ) => + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) + if matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(..), + .. + }) + ) => { - SummaryExpr::ValueOperation { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: child.clone(), - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } + }) } _ => return Some(node), }; - Some(Rc::new(SummaryNode { - expr, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - })) + Some(Rc::new( + OperatorNode::with_schema(operator, node.schema.clone()) + .with_guarantee(node.guarantee.clone()), + )) } #[allow(clippy::too_many_arguments)] fn construct_summary_agg( - node: &QueryExpr, + node: &OperatorNode, reduction: &Reduction, intent: &AggIntent, input: PhysicalSummaryInput, @@ -2771,7 +2864,7 @@ fn construct_summary_agg( planning_inputs: CandidatePlanningInputs<'_>, child_target: Option<&AccuracyTarget>, allocation: Option, -) -> Result, RealizationError> { +) -> Result, RealizationError> { // The single canonical pre-ASAP derivation (per-series vs cross-series, // name overrides) already computes the row shape; binding only retypes // the summary state column. @@ -2781,7 +2874,7 @@ fn construct_summary_agg( FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap) ); - let snapshot_weighted = matches!(node, QueryExpr::Aggregate { child, .. } + let snapshot_weighted = matches!(node.non_asap(), Some(NonASAPOp::Aggregate { child, .. }) if is_snapshot_weighted_topk(intent, child)); let mut family = family; let score_population = if snapshot_weighted { @@ -2809,10 +2902,10 @@ fn construct_summary_agg( None }; let physical_reduction = if snapshot_weighted { - let QueryExpr::Aggregate { child, .. } = node else { + let Some(NonASAPOp::Aggregate { child, .. }) = node.non_asap() else { unreachable!() }; - let source = input.child.output_schema()?; + let source = &input.child.schema; let Reduction::Reduce(keys) = reduction else { return Err(RealizationError::PhysicalRealization( "TopK requires explicit partitions", @@ -2850,19 +2943,19 @@ fn construct_summary_agg( } else { reduction.clone() }; - let out_schema = node.output_schema()?; - let measures = match node { - QueryExpr::Aggregate { measures, .. } => measures.len(), + let out_schema = &node.schema; + let measures = match node.non_asap() { + Some(NonASAPOp::Aggregate { measures, .. }) => measures.len(), _ => 1, }; - let state_idx = summary_col_index(&out_schema, reduction, measures); + let state_idx = summary_col_index(out_schema, reduction, measures); - let readout_schema = if keyed_heap - && matches!(node, QueryExpr::Aggregate { child, .. } if is_snapshot_weighted_topk(intent, child)) + let evaluation_schema = if keyed_heap + && matches!(node.non_asap(), Some(NonASAPOp::Aggregate { child, .. }) if is_snapshot_weighted_topk(intent, child)) { - keyed_heap_readout_schema(&input, node)? + keyed_heap_evaluation_schema(&input, node)? } else { - lift(&out_schema) + out_schema.clone() }; let summary_input = input.input; @@ -2879,15 +2972,21 @@ fn construct_summary_agg( }; } } - readout(intent, &summary_input, planning_inputs.cost) + evaluation(intent, &summary_input) }); - let mut state_schema = lift(&out_schema); + let mut state_schema = out_schema.clone(); if keyed_heap { let mut state = state_schema.fields[state_idx].clone(); state.dtype = family.clone(); + // A top-k's output row holds the ranked item at `state_idx`; the + // state column is the heap itself. + if let AggIntent::TopK { k, .. } = intent { + state.name = format!("topk_{k}"); + state.nullable = false; + } let mut fields = if snapshot_weighted { - readout_schema.fields[..reduction.group_keys().map_or(0, |keys| keys.len())].to_vec() + evaluation_schema.fields[..reduction.group_keys().map_or(0, |keys| keys.len())].to_vec() } else { Vec::new() }; @@ -2895,6 +2994,7 @@ fn construct_summary_agg( state_schema = Schema::lifted(fields, None); } else if let Some(field) = state_schema.fields.get_mut(state_idx) { field.dtype = family.clone(); + field.nullable = false; if matches!(&family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::UnivMon) { // State identity is independent of which statistic reads it. @@ -2902,9 +3002,9 @@ fn construct_summary_agg( } else if let (AggIntent::Quantile { .. }, SummaryInputExpr::Column(col)) = (intent, &summary_input.weight) { - // The quantile is a readout parameter: name the state after the + // The quantile is a evaluation parameter: name the state after the // column it summarizes, not after the query's output column. - let child_schema = input.child.output_schema()?; + let child_schema = input.child.schema.clone(); if let Ok(i) = resolve_column_ref(col, &child_schema) { field.name = child_schema.fields[i].name.clone(); } @@ -2924,46 +3024,33 @@ fn construct_summary_agg( // Explicit snapshot selection prevents historical observations from // becoming repeated weights in an instant-vector heap. let root = Rc::new(node.clone()); - let population = crate::maintained_population::MaintainedPopulationStrategy::new( + let population = crate::pass1::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&root), ) .candidate(&root) .ok_or(RealizationError::PhysicalRealization( "snapshot ranking requires a supported current-series population", ))?; - let SummaryExpr::ValueOperation { child, .. } = &population.expr else { + let Operator::ASAP(ASAPOp::EvaluatePopulation { child, .. }) = &population.operator else { return Err(RealizationError::PhysicalRealization( - "missing population readout", + "missing population evaluation", )); }; Rc::clone(child) } else if snapshot_weighted { // Each evaluation's finalized rates feed a fresh summary; rate snapshots // must never accumulate across evaluations. Query time is only the - // initial layout; a retained summary's lifecycle moves it to ingestion. + // initial layout; a maintained summary's materialization moves it to ingestion. finalize_query_candidate(bound_child, &input.child)? } else { - let child = finalize_exact_accumulator_at( - bound_child, - &input.child, - ExecutionTiming::IngestionTime, - )?; - let child = maintenance_exact_values(child).unwrap_or(keep_pre_asap(&input.child)?); - // Maintenance arithmetic must satisfy the ingestion contract; e.g. a - // per-series sum over different selectors has no exact aligned - // layout, so this candidate fails closed and exact execution remains. - // Unlike checked division, it does not fall back to `keep_pre_asap`: - // that retains the range expression at ingestion time, where range - // functions cannot run (they need a query evaluation time). - if matches!(child.expr, SummaryExpr::BinaryOp { .. }) { - validate_execution_data_states_at(&child, ExecutionDataState::INGESTION_ROWS)?; - } - child + let child = + finalize_exact_accumulator(bound_child, &input.child, ExecutionTiming::IngestionTime)?; + maintenance_exact_values(child).unwrap_or(retain_exact(&input.child)?) }; // ── Guarantee (issue #172) ────────────────────────────────────────── // Derived *before* the node exists, so an illegal composition is never - // materialized: the local guarantee of this family's readout (or exact + // materialized: the local guarantee of this family's evaluation (or exact // accumulator) composed over the child's, under the operator this // family applies to the child's values. let local_target = match allocation.as_ref() { @@ -2976,7 +3063,7 @@ fn construct_summary_agg( local_target, ); let membership_query = if snapshot_weighted { - Some(readout(intent, &summary_input, planning_inputs.cost)) + Some(evaluation(intent, &summary_input)) } else { query.clone() }; @@ -2991,7 +3078,7 @@ fn construct_summary_agg( )?; if snapshot_weighted { - use asap_types::post_asap::{BoundExpr, ProbabilityExpr}; + use asap_types::ir::properties::{BoundExpr, ProbabilityExpr}; let target = accuracy_target(intent).expect("TopK target"); guarantee = if let Some(mut score) = estimator.local_guarantee(&family, query.as_ref().unwrap()) @@ -3054,56 +3141,65 @@ fn construct_summary_agg( // a genuine empty-`by` reduction apart from a per-entity shape with no // grouping concept at all (issue #163). `construct_summary_agg` is the // single place that decides this; nothing downstream re-derives it. - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { + let coverage = whole_source_coverage(&bound_child); + let agg = OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { child: bound_child, family, input: summary_input, reduction: physical_reduction, grouping: GroupingStrategy::default(), filter: None, - }, - schema: state_schema, + }), + state_schema, + ) + .with_guarantee( // Summary *state* carries no caller-visible guarantee; only a // finalized value does. An exact accumulator's state is its value. - guarantee: if estimate { None } else { guarantee.clone() }, + if estimate { None } else { guarantee.clone() }, + ); + let agg = std::rc::Rc::new(match coverage { + Some(coverage) => agg.with_coverage(coverage)?, + None => agg, }); match query { - // The readout: downstream of the estimate the schema is the plain + // The evaluation: downstream of the estimate the schema is the plain // pre-ASAP row shape again (the summary-state type does not // propagate). - Some(query) => Ok(Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: agg, - query, - }, - schema: readout_schema, - guarantee, - })), + Some(query) => Ok(std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: agg, + query, + }), + evaluation_schema, + ) + .with_guarantee(guarantee), + )), None => Ok(agg), } } -// Heap readout rows contain the encoded item identity, subpopulation keys, +// Heap evaluation rows contain the encoded item identity, subpopulation keys, // and an estimated score. They never inherit the exact-value producer's schema. -fn keyed_heap_readout_schema( +fn keyed_heap_evaluation_schema( input: &PhysicalSummaryInput, - node: &QueryExpr, + node: &OperatorNode, ) -> Result { - let source = input.child.output_schema()?; + let source = &input.child.schema; let mut refs = Vec::new(); - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, child, .. - } = node + }) = node.non_asap() else { return Err(RealizationError::PhysicalRealization( - "heap readout requires an aggregate", + "heap evaluation requires an aggregate", )); }; if let Reduction::Reduce(groups) = reduction { if groups.is_without() { return Err(RealizationError::PhysicalRealization( - "heap readout requires explicit grouping", + "heap evaluation requires explicit grouping", )); } for index in groups.iter() { @@ -3161,10 +3257,10 @@ fn keyed_heap_readout_schema( .ok_or(RealizationError::PhysicalRealization( "heap item identity is missing", ))?, - &source, + source, &mut refs, )?; - let mut fields = Vec::::new(); + let mut fields = Vec::::new(); for reference in refs { let matches: Vec<_> = source .fields @@ -3197,35 +3293,36 @@ fn keyed_heap_readout_schema( } if fields.is_empty() { return Err(RealizationError::PhysicalRealization( - "heap readout has no identity columns", + "heap evaluation has no identity columns", )); } fields.push(Field::new( "__asap_estimate", - FieldDataType::Plain(asap_types::pre_asap::DataType::Float64), + FieldDataType::Plain(asap_types::ir::schema::DataType::Float64), false, )); Ok(Schema::lifted(fields, None)) } -fn ranking_score_index(logical: &QueryExpr, values: &Schema) -> Result { +fn ranking_score_index(logical: &OperatorNode, values: &Schema) -> Result { if is_current_series_source(logical) { return values .fields .iter() .position(|field| { field.name == "value" - && field.dtype == FieldDataType::Plain(asap_types::pre_asap::DataType::Float64) + && field.dtype + == FieldDataType::Plain(asap_types::ir::schema::DataType::Float64) }) .ok_or(RealizationError::PhysicalRealization( "snapshot ranking requires the sample value column", )); } - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, .. - } = logical + }) = logical.non_asap() else { return Err(RealizationError::PhysicalRealization( "ranking requires an explicit aggregate score", @@ -3256,7 +3353,8 @@ fn ranking_score_index(logical: &QueryExpr, values: &Schema) -> Result, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !matches!(family, FieldDataType::Sketch(kind, _) if matches!(kind.algorithm(), SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap)) - || !matches!(child.as_ref(), QueryExpr::Aggregate { reduction: Reduction::PerEntity, measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) + || !matches!(child.non_asap(), Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "counter ranking needs a valid value schema", - ); - }; + let schema = &child.schema; if !schema.closed { return PhysicalSummaryInputRuleResult::Unsupported( "counter ranking needs the complete resolved series identity", @@ -3332,7 +3426,7 @@ fn realize_current_series_summary_input( intent: &AggIntent, family: &FieldDataType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) || !is_current_series_source(child) { return PhysicalSummaryInputRuleResult::NotApplicable; @@ -3359,11 +3453,7 @@ fn realize_current_series_summary_input( "snapshot ranking requires resolved partitions", ); } - let Ok(schema) = child.output_schema() else { - return PhysicalSummaryInputRuleResult::Unsupported( - "snapshot ranking requires a valid source schema", - ); - }; + let schema = &child.schema; let items = schema .fields .iter() @@ -3389,11 +3479,11 @@ fn realize_current_series_summary_input( /// Realize the composite heavy-hitter realization for /// `TopK(Count GROUP BY key)`. The heap sketch consumes the raw keyed stream; /// it does not consume an independently materialized Count result. -fn realize_keyed_additive_summary_input( +pub(crate) fn realize_keyed_additive_summary_input( intent: &AggIntent, family: &FieldDataType, output_reduction: &Reduction, - child: &Rc, + child: &Rc, ) -> PhysicalSummaryInputRuleResult { if !matches!(intent, AggIntent::TopK { .. }) { return PhysicalSummaryInputRuleResult::NotApplicable; @@ -3408,18 +3498,18 @@ fn realize_keyed_additive_summary_input( ) { return PhysicalSummaryInputRuleResult::NotApplicable; } - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, having: None, child: raw_child, .. - } = child.as_ref() + }) = child.non_asap() else { return PhysicalSummaryInputRuleResult::NotApplicable; }; let counter_input = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(raw_child.as_ref(), QueryExpr::Aggregate { measures, .. } + && matches!(raw_child.non_asap(), Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])); let weight = match measures.as_slice() { [AggIntent::Count { .. }] => SummaryInputExpr::Constant(1.0), @@ -3511,9 +3601,8 @@ fn realize_keyed_additive_summary_input( }) } -fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { - let schema = child.output_schema().ok()?; - let column = schema.fields.get(index)?; +fn schema_column_ref(child: &OperatorNode, index: usize) -> Option { + let column = child.schema.fields.get(index)?; Some(match &column.table { Some(table) => ColumnRef::Qualified { table: table.clone(), @@ -3535,7 +3624,7 @@ fn schema_column_ref(child: &QueryExpr, index: usize) -> Option { fn compose_guarantee( family: &FieldDataType, query: Option<&PostAsapSketchStatistic>, - child: &SummaryNode, + child: &OperatorNode, intent: &AggIntent, accuracy: &dyn AccuracyModel, evidence: &dyn AccuracyEvidenceProvider, @@ -3555,7 +3644,7 @@ fn compose_guarantee( // Lipschitz constant — over an approximate child this is a // deterministic transform with no registered rule. _ => { - crate::function_rules::function_rules(intent) + crate::pass1::function_rules::function_rules(intent) .expect("exact accumulator intents have registered accuracy rules") .accuracy } @@ -3644,7 +3733,7 @@ fn summarised_column(intent: &AggIntent, child_schema: &Schema) -> ColumnRef { } } -fn column_ref(column: &Field) -> ColumnRef { +pub(crate) fn column_ref(column: &Field) -> ColumnRef { match &column.table { Some(t) => ColumnRef::Qualified { table: t.clone(), @@ -3662,7 +3751,7 @@ fn column_ref(column: &Field) -> ColumnRef { /// A tuple leg outside the child schema is an error rather than /// [`summarised_column`]'s sample-value fallback: a leg has no sample-value /// reading, and silently dropping one would under-count. -fn summarised_input( +pub(crate) fn summarised_input( intent: &AggIntent, child_schema: &Schema, ) -> Result { @@ -3685,12 +3774,8 @@ fn summarised_input( )) } -/// The `SummaryEstimate` readout for a summary-bound intent. -fn readout( - intent: &AggIntent, - input: &SummaryUpdate, - cost_model: &dyn CostModel, -) -> PostAsapSketchStatistic { +/// The `SummaryEstimate` evaluation for a summary-bound intent. +fn evaluation(intent: &AggIntent, input: &SummaryUpdate) -> PostAsapSketchStatistic { match intent { AggIntent::Quantile { q, .. } => PostAsapSketchStatistic::Quantile { q: *q }, AggIntent::Cardinality { .. } => PostAsapSketchStatistic::Cardinality, @@ -3705,31 +3790,15 @@ fn readout( }, value: None, }, - // Core doesn't know the shape of a deployment-specific `Extension` - // intent, so it can't build its readout either — delegate to the - // same `CostModel` that decided (via `realize_extension`) this - // intent gets a summary realization at all. See `readout_extension`'s - // doc for the invariant this depends on. - AggIntent::Extension { ext_kind, payload } => match &input.weight { - SummaryInputExpr::Column(col) => cost_model.readout_extension(ext_kind, payload, col), - _ => unreachable!("extension readout requires one column"), - }, other => { unreachable!("no summary realization for {other:?} (realizations_for_intent)") } } } -/// Lift a pre-ASAP [`Schema`] to a [`Schema`] with every column -/// `FieldDataType::Plain` — shared by [`construct_summary_agg`] and -/// [`keep_pre_asap`], both in this module. -fn lift(schema: &Schema) -> Schema { - Schema::lifted(schema.fields.clone(), schema.time_index) -} - // ── SharedSubDAGStrategy ──────────────────────────────────────────────── -/// Wraps `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing +/// Wraps `asap_types::ir::cse::share_common_sub_dags`'s sharing /// decision as an explicit candidate pair, wherever a [`TargetSubDAG`] /// already has two or more consumers. /// @@ -3743,7 +3812,7 @@ fn lift(schema: &Schema) -> Schema { /// "Non-goals" on why that traversal isn't itself part of this strategy). /// This strategy only reframes "two or more consumers already share this /// `Rc`" as the two-way choice a downstream cost model (today, -/// [`CostModel::cse_share_decision`]) picks between: build once and share, or +/// `CostModel::cse_share_decision`) picks between: build once and share, or /// build independently at each consumer. pub struct SharedSubDAGStrategy; @@ -3762,11 +3831,11 @@ impl ReplacementStrategy for SharedSubDAGStrategy { strategy: "SharedSubDAGStrategy", // The already-interned `Rc` itself: reusing it verbatim *is* // "build once and share" — no new node to construct. - replacement: Replacement::Rewrite(Rc::clone(target.root)), + replacement: Replacement::SubDAG(Rc::clone(target.root)), provenance: ReplacementProvenance::CseShare, rationale: format!( "build once and share: share_common_sub_dags already interned this \ - sub-DAG once and reused it across {count} consumers — one build can \ + sub_dag once and reused it across {count} consumers — one build can \ answer all of them instead of computing it {count} times" ), }, @@ -3775,11 +3844,11 @@ impl ReplacementStrategy for SharedSubDAGStrategy { // A structurally-identical but freshly-allocated `Rc`: same // value (`PartialEq`), deliberately *not* the same pointer, // representing "undo the sharing and recompute independently". - replacement: Replacement::Rewrite(Rc::new((**target.root).clone())), + replacement: Replacement::SubDAG(Rc::new((**target.root).clone())), provenance: ReplacementProvenance::CseRecompute, rationale: format!( "build independently: undo the sharing share_common_sub_dags found and \ - recompute this sub-DAG separately at each of its {count} consumers — \ + recompute this sub_dag separately at each of its {count} consumers — \ worth it only when independence outweighs the shared-maintenance cost, \ a CostModel's call (e.g. CostModel::cse_share_decision) and not this \ strategy's" @@ -3798,14 +3867,14 @@ impl ReplacementStrategy for SharedSubDAGStrategy { /// A generous, documented backstop against a hypothetically ill-behaved /// future [`ReplacementStrategy`] (see the module docs' "Termination" /// section) — not a bound either shipped strategy could ever approach. -/// [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] both converge in +/// [`ASAPStrategies`] and [`SharedSubDAGStrategy`] both converge in /// exactly 2 passes over a fixed target set, regardless of workload size. pub const MAX_SEARCH_ITERATIONS: usize = 1_000; // ── TargetSubDAGCandidates ────────────────────────────────────────────── /// Candidates for one distinct [`TargetSubDAG`] (its -/// own `target` `Rc`, keyed by pointer identity in +/// own `target` `Rc`, keyed by pointer identity in /// [`CandidateLogicalASAPDAGs`]'s internal map — never re-derived by value) plus every /// [`ReplacementSubDAG`] alternative any registered [`ReplacementStrategy`] /// proposed for it. @@ -3818,24 +3887,24 @@ pub const MAX_SEARCH_ITERATIONS: usize = 1_000; #[derive(Debug, Clone)] pub struct TargetSubDAGCandidates { /// The target sub-DAG this group is for. - pub target: Rc, + pub target: Rc, /// How many operator-child positions across the whole workload /// reference this exact `Rc` — see [`discover_targets`]. pub consumer_count: usize, /// Every distinct alternative discovered for `target`, in discovery - /// order (not ranked — see [`CandidateLogicalASAPDAGs::cost_sorted`] for the ranked + /// order (not ranked — see `candidate_selection::cost_sorted` for the ranked /// view). pub candidates: Vec, /// Every candidate a strategy considered for `target` but refused on /// accuracy-legality grounds (issue #172), plus any `candidates` entry /// the root-target check ([`search_workload_with_targets`]) moved here. - /// Never ranked — [`CandidateLogicalASAPDAGs::cost_sorted`]/[`CandidateLogicalASAPDAGs::global_selection`] - /// read only `candidates`, so a [`CostModel`] cannot resurrect one. + /// Never ranked — `candidate_selection::cost_sorted`/`candidate_selection::global_selection` + /// read only `candidates`, so a `CostModel` cannot resurrect one. pub rejected: Vec, } impl TargetSubDAGCandidates { - fn new(target: Rc, consumer_count: usize) -> Self { + pub(crate) fn new(target: Rc, consumer_count: usize) -> Self { Self { target, consumer_count, @@ -3852,10 +3921,12 @@ impl TargetSubDAGCandidates { fn add_candidate(&mut self, candidate: ReplacementSubDAG) -> bool { let is_duplicate = self.candidates.iter().any(|existing| { match (&existing.replacement, &candidate.replacement) { - (Replacement::Rewrite(existing_rc), Replacement::Rewrite(rc)) => { + (Replacement::SubDAG(existing_rc), Replacement::SubDAG(rc)) + if is_logical_rewrite(existing_rc) && is_logical_rewrite(rc) => + { is_duplicate_rewrite(existing_rc, rc, &self.target) } - (Replacement::Summary(existing_node), Replacement::Summary(node)) => { + (Replacement::SubDAG(existing_node), Replacement::SubDAG(node)) => { is_duplicate_summary(existing_node, node) } ( @@ -3876,11 +3947,11 @@ impl TargetSubDAGCandidates { } } -/// Are `existing` and `candidate` the same [`Replacement::Rewrite`] -/// candidate for a group targeting `target`? +/// Are `existing` and `candidate` the same logical-rewrite +/// [`Replacement::SubDAG`] candidate for a group targeting `target`? /// -/// Structural (`QueryExpr`) value equality alone is *not* enough here: this -/// module's one shipped multi-candidate `Replacement::Rewrite` source, +/// Structural (`OperatorNode`) value equality alone is *not* enough here: +/// this module's one shipped multi-candidate logical-rewrite source, /// [`SharedSubDAGStrategy`], deliberately returns **two** candidates that /// are value-equal to each other (`build once and share` vs. `build /// independently` — see that strategy's own doc) but represent genuinely @@ -3898,7 +3969,7 @@ impl TargetSubDAGCandidates { /// So: two candidates whose "is this the target's own `Rc`?" bit disagrees /// are never duplicates of each other, full stop. Only when that bit /// *agrees* does this fall through to the real dedup discipline — -/// [`structural_hash`] as a candidate-narrowing filter, `QueryExpr`'s +/// [`structural_hash`] as a candidate-narrowing filter, `OperatorNode`'s /// derived `PartialEq` as the actual decision — protecting against the /// (currently hypothetical, since neither shipped strategy causes it) /// case of the exact same alternative being proposed twice. A fresh @@ -3907,9 +3978,9 @@ impl TargetSubDAGCandidates { /// wider traversal to amortize the cache across the way `InternTable`'s own /// use of `structural_hash` does. fn is_duplicate_rewrite( - existing: &Rc, - candidate: &Rc, - target: &Rc, + existing: &Rc, + candidate: &Rc, + target: &Rc, ) -> bool { let existing_is_target = Rc::ptr_eq(existing, target); let candidate_is_target = Rc::ptr_eq(candidate, target); @@ -3921,16 +3992,16 @@ fn is_duplicate_rewrite( && existing == candidate } -/// Are `existing` and `candidate` the same [`Replacement::Summary`] -/// candidate? +/// Are `existing` and `candidate` the same bound-summary +/// [`Replacement::SubDAG`] candidate? /// -/// [`SummaryNode`] derives neither `PartialEq` nor `Hash` (it embeds -/// `SketchParams`/`f64`-bearing accuracy targets deep inside `SummaryExpr`, -/// the same reason `QueryExpr` can't derive `Hash` either — see -/// [`structural_hash`]'s own doc). Per this module's inherited "hash is a -/// filter, `PartialEq` is the decision, no exceptions" rule, there is no -/// real equality check to back a dedup *decision* here — and skipping the -/// check is the only choice that rule permits: never merging two candidates +/// A bound summary embeds `SketchParams`/`f64`-bearing accuracy targets and +/// guarantees, so value equality is not a dedup decision this module is +/// willing to make (see [`structural_hash`]'s own doc on `f64` hashing). +/// Per this module's inherited "hash is a filter, `PartialEq` is the +/// decision, no exceptions" rule, there is no real equality check to back a +/// dedup *decision* here — and skipping the check is the only choice that +/// rule permits: never merging two candidates /// is harmless (at worst, a redundant entry in a group's candidate list), /// while comparing by some proxy this module can't actually verify (e.g. /// `Debug` text, or `ReplacementSubDAG::rationale` — documented elsewhere in @@ -3939,7 +4010,7 @@ fn is_duplicate_rewrite( /// shipped today already return a structurally distinct candidate for every /// entry of one `replacements()` call, so this is future-proofing against a /// hypothetical repeat call, not a gap either strategy's own tests exercise. -fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { +fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc) -> bool { false } @@ -3948,34 +4019,36 @@ fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc` whose -/// group holds its alternatives. +/// caller can still map a `Root`'s `Id` back to the `Rc` whose +/// group holds its alternatives. Memos are keyed by `*const OperatorNode`. pub struct CandidateLogicalASAPDAGs { /// The workload's roots, after the one `share_common_sub_dags` pass /// [`search_workload_with`] runs up front — the same post-CSE roots /// every `TargetSubDAG` in `groups` was discovered from. - pub roots: Vec<(Id, Rc)>, - groups: HashMap<*const QueryExpr, TargetSubDAGCandidates>, + pub roots: Vec<(Id, Rc)>, + pub(crate) groups: HashMap<*const OperatorNode, TargetSubDAGCandidates>, /// Discovery order — stable iteration for [`CandidateLogicalASAPDAGs::target_subdag_candidates`]/ - /// [`CandidateLogicalASAPDAGs::cost_sorted`], since `HashMap` iteration order isn't. - order: Vec<*const QueryExpr>, + /// `candidate_selection::cost_sorted`, since `HashMap` iteration order isn't. + pub(crate) order: Vec<*const OperatorNode>, /// Composition proofs are computed with the search model, then retained /// through costing and DAG assembly so no later default can replace it. - composition_plans: Vec, + pub(crate) composition_plans: Vec, } -struct PreparedComposition { - target: *const QueryExpr, - operation: ExactComposition, - child: Rc, - plan: Rc, +/// One exact composition validated during search: `operation` at `target`, +/// over the child candidate `child`, giving `plan`. +pub struct PreparedComposition { + pub target: *const OperatorNode, + pub operation: ExactComposition, + pub child: Rc, + pub plan: Rc, } impl CandidateLogicalASAPDAGs { fn prepare_compositions( &mut self, accuracy: &dyn AccuracyModel, - targets: &HashMap<*const QueryExpr, Vec>, + targets: &HashMap<*const OperatorNode, Vec>, ) { self.composition_plans.clear(); for group in self.groups.values() { @@ -3990,14 +4063,16 @@ impl CandidateLogicalASAPDAGs { .into_iter() .flat_map(|g| &g.candidates) .filter_map(|c| match &c.replacement { - Replacement::Summary(child) if operation.accepts_child(child) => { + Replacement::SubDAG(child) + if !is_logical_rewrite(child) && operation.accepts_child(child) => + { Some(Rc::clone(child)) } _ => None, }) .collect(), OperationPlacement::Maintenance => { - keep_pre_asap(&operation.child_target).into_iter().collect() + retain_exact(&operation.child_target).into_iter().collect() } }; for child in children { @@ -4029,17 +4104,17 @@ impl CandidateLogicalASAPDAGs { } /// DAG candidates assembled from an unpriced search space. -/// This is an internal planning stage: callers must still validate lifecycle +/// This is an internal planning stage: callers must still validate materialization /// requirements and compile supported physical operators before deployment. /// The caller supplies a finite expansion budget; exceeding it is an error, /// never a silently truncated inventory presented as exhaustive. #[derive(Debug)] pub struct CandidateDAGInventory { - pub candidates: Vec)>>, + pub candidates: Vec)>>, pub rejected_assemblies: Vec, } -type CandidateDAGChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); +type CandidateDAGChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); impl CandidateLogicalASAPDAGs { pub fn enumerate_candidate_dags( @@ -4074,7 +4149,7 @@ impl CandidateLogicalASAPDAGs { fn enumerate_candidate_roots( &self, - roots: &[(Id, Rc)], + roots: &[(Id, Rc)], expansion_limit: usize, ) -> Result, RealizationError> { let mut reachable = Vec::new(); @@ -4090,8 +4165,10 @@ impl CandidateLogicalASAPDAGs { cursor += 1; if let Some(group) = self.groups.get(&ptr) { for candidate in &group.candidates { - if let Replacement::Rewrite(rewritten) = &candidate.replacement { - walk(rewritten, &mut reachable, &mut nodes, &mut counts); + if let Replacement::SubDAG(rewritten) = &candidate.replacement { + if is_logical_rewrite(rewritten) { + walk(rewritten, &mut reachable, &mut nodes, &mut counts); + } } } } @@ -4163,13 +4240,13 @@ impl CandidateLogicalASAPDAGs { consumer_count: group.consumer_count, effective_consumer_count: group.consumer_count, chosen: *chosen, - composition: None, }, ); } let assembly = GlobalSelection { order: order.clone(), groups, + composition_plans: HashMap::new(), assembled_nodes: RefCell::new(assembled_nodes), }; let roots = roots @@ -4185,77 +4262,12 @@ impl CandidateLogicalASAPDAGs { .collect::, _>>(); match roots { Ok(roots) => { - let roots = asap_types::post_asap::share_common_summary_sub_dags(roots); + let roots = share_common_sub_dags(roots); use std::hash::{Hash, Hasher}; let mut hash = std::collections::hash_map::DefaultHasher::new(); - let mut pending = roots - .iter() - .map(|(_, node)| node.as_ref()) - .collect::>(); - while let Some(node) = pending.pop() { - std::mem::discriminant(&node.expr).hash(&mut hash); - let raw = match &node.expr { - SummaryExpr::KeepPreAsap(raw) => Some(raw.as_ref()), - _ => None, - }; - let operation = match &node.expr { - SummaryExpr::ValueOperation { - timing, operation, .. - } => serde_json::json!((timing, operation)), - SummaryExpr::BinaryOp { - timing, operator, .. - } => serde_json::json!((timing, operator)), - SummaryExpr::SummaryMerge { timing, .. } => serde_json::json!(timing), - _ => serde_json::Value::Null, - }; - let mut value = - serde_json::to_value((&node.schema, &node.guarantee, raw, operation)) - .map_err(|_| { - RealizationError::PhysicalRealization( - "candidate identity serialization failed", - ) - })?; - fn normalize(value: &mut serde_json::Value) { - match value { - serde_json::Value::Number(number) - if number.as_f64() == Some(0.0) => - { - *value = serde_json::json!(0); - } - serde_json::Value::Array(values) => { - values.iter_mut().for_each(normalize) - } - serde_json::Value::Object(values) => { - values.values_mut().for_each(normalize) - } - _ => {} - } - } - normalize(&mut value); - value.sort_all_objects(); - value.to_string().hash(&mut hash); - match &node.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - pending.extend([lhs.as_ref(), rhs.as_ref()]) - } - SummaryExpr::RelationalJoin { left, right, .. } - | SummaryExpr::SummarySubtract { left, right } => { - pending.extend([left.as_ref(), right.as_ref()]) - } - SummaryExpr::ValueOperation { child, .. } - | SummaryExpr::SummaryAgg { child, .. } => pending.push(child.as_ref()), - SummaryExpr::SummaryJoin { outer, inner, .. } => { - pending.extend([outer.as_ref(), inner.as_ref()]) - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - pending.push(summary_input.as_ref()) - } - SummaryExpr::SummaryMerge { children, .. } => { - pending.extend(children.iter().map(|child| child.as_ref())) - } - } + let mut cache = HashCache::new(); + for (_, node) in &roots { + structural_hash(node, &mut cache).hash(&mut hash); } let bucket = seen.entry(hash.finish()).or_default(); if !bucket @@ -4278,54 +4290,25 @@ impl CandidateLogicalASAPDAGs { } } -/// Lifecycle-aware whole-subplan costs keyed by target and candidate identity. -#[derive(Default, Clone)] -pub(crate) struct CandidateCostOverrides { - costs: HashMap<(*const QueryExpr, *const ReplacementSubDAG), Cost>, - raw_costs: HashMap<*const QueryExpr, Cost>, - /// Targets for which the caller requested an atomic raw-vs-summary - /// decision. Other memo groups continue through ordinary CSE selection. - finalized_targets: HashSet<*const QueryExpr>, -} - -impl CandidateCostOverrides { - pub(crate) fn finalize_target(&mut self, target: &Rc) { - self.finalized_targets.insert(Rc::as_ptr(target)); - } - - fn finalizes(&self, target: &Rc) -> bool { - self.finalized_targets.contains(&Rc::as_ptr(target)) - } - - pub(crate) fn insert( - &mut self, - target: &Rc, - candidate: &ReplacementSubDAG, - cost: Cost, - ) { - self.costs - .insert((Rc::as_ptr(target), candidate as *const _), cost); - } - - fn get(&self, target: &Rc, candidate: &ReplacementSubDAG) -> Option { - self.costs - .get(&(Rc::as_ptr(target), candidate as *const _)) - .copied() +impl CandidateLogicalASAPDAGs { + /// One candidate set per discovered target sub-DAG, in discovery order. + pub fn target_subdag_candidates(&self) -> impl Iterator { + self.order.iter().map(move |ptr| &self.groups[ptr]) } - pub(crate) fn insert_raw(&mut self, target: &Rc, cost: Cost) { - self.raw_costs.insert(Rc::as_ptr(target), cost); + /// Every discovered target's candidate set, keyed by target identity. + pub fn groups(&self) -> &HashMap<*const OperatorNode, TargetSubDAGCandidates> { + &self.groups } - fn raw(&self, target: &Rc) -> Option { - self.raw_costs.get(&Rc::as_ptr(target)).copied() + /// Target identities in discovery order. + pub fn order(&self) -> &[*const OperatorNode] { + &self.order } -} -impl CandidateLogicalASAPDAGs { - /// One candidate set per discovered target sub-DAG, in discovery order. - pub fn target_subdag_candidates(&self) -> impl Iterator { - self.order.iter().map(move |ptr| &self.groups[ptr]) + /// The exact compositions validated during search. + pub fn composition_plans(&self) -> &[PreparedComposition] { + &self.composition_plans } /// How many distinct targets were discovered. @@ -4334,7 +4317,7 @@ impl CandidateLogicalASAPDAGs { } /// Whether no targets were discovered at all (an empty workload, or one - /// with no `QueryExpr` nodes reachable from any root — never true for a + /// with no `OperatorNode`s reachable from any root — never true for a /// non-empty `roots`, since every root is itself a target). pub fn is_empty(&self) -> bool { self.groups.is_empty() @@ -4343,707 +4326,98 @@ impl CandidateLogicalASAPDAGs { /// The candidate set for `target`, if `target`'s own `Rc` is a discovered /// `TargetSubDAG` (i.e. `Rc::ptr_eq` to some node reachable from /// `roots`). - pub fn candidates_for_target(&self, target: &Rc) -> Option<&TargetSubDAGCandidates> { + pub fn candidates_for_target( + &self, + target: &Rc, + ) -> Option<&TargetSubDAGCandidates> { self.groups.get(&Rc::as_ptr(target)) } +} - /// The `sorted_by(cost_model)` step: every group, each with its own - /// candidates ranked best-first under `cost_model` where this module - /// knows how (see the module docs' "Cost-based final selection" - /// section) — groups themselves stay in discovery order, since targets - /// are independent decision points, not alternatives competing with - /// each other. - /// - /// Ranking itself is decided entirely by [`rank_group`] before - /// [`RankedTargetSubDAGCandidates::costs`] is ever computed — pairing each candidate with - /// [`CostModel::grouping_state_cost`] for grouping alternatives, or - /// [`CostModel::estimate_cost`] otherwise, is an additive annotation - /// for a caller that wants to *display* a cost (e.g. a - /// DAG-visualization view), not a second ranking signal, so plugging in - /// a `CostModel` whose `estimate_cost` disagrees with its own - /// `rank_candidates`/`cse_share_decision` (a deployment bug, not - /// something this method tries to protect against) would show a - /// `RankedTargetSubDAGCandidates` whose `costs` aren't monotonically non-decreasing — - /// `cost_sorted`'s own ordering guarantee is unaffected either way. - pub fn cost_sorted(&self, cost_model: &dyn CostModel) -> Vec> { - self.order - .iter() - .map(|ptr| { - let group = &self.groups[ptr]; - let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); - let mut candidates = rank_group(group, cost_model); - // Availability is candidate-specific and cannot be expressed - // by `rank_candidates`' exhaustive permutation contract. - // Keep unavailable alternatives for explanation, but place - // them after every selectable candidate. - candidates.sort_by_key(|candidate| { - cost_model.candidate_cost(candidate, &target).is_none() - }); - let costs = candidates - .iter() - .map(|c| { - cost_model - .grouping_state_cost(c, &target) - .map_or_else(|| cost_model.estimate_cost(c, &target), |cost| cost.0) - }) - .collect(); - RankedTargetSubDAGCandidates { - target: &group.target, - consumer_count: group.consumer_count, - candidates, - costs, +/// Find the explicitly-tagged CSE share/recompute pair inside `group`, even +/// when other strategies contributed additional alternatives to the same +/// memo group. Provenance makes these two orthogonal choices identifiable +/// without inferring semantics from pointer or expression shape. +pub fn cse_candidate_pair( + group: &TargetSubDAGCandidates, +) -> Option<(&ReplacementSubDAG, &ReplacementSubDAG)> { + let mut share = None; + let mut recompute = None; + for candidate in &group.candidates { + match candidate.provenance { + ReplacementProvenance::CseShare => { + let Replacement::SubDAG(rc) = &candidate.replacement else { + return None; + }; + if !Rc::ptr_eq(rc, &group.target) || share.replace(candidate).is_some() { + return None; } - }) - .collect() - } - - /// Recurrence-aware counterpart to [`Self::cost_sorted`]. CSE - /// share/recompute pairs are ordered with the target's recurrence - /// profile; all other candidate shapes retain their existing ranking. - pub fn cost_sorted_with_recurrence( - &self, - cost_model: &dyn CostModel, - profiles: &RecurrenceProfileMap, - horizon: Option, - ) -> Result>, RecurrenceError> { - self.order - .iter() - .map(|ptr| { - let group = &self.groups[ptr]; - let mut candidates = rank_group(group, cost_model); - if cse_candidate_pair(group).is_some() { - if let Some(decision) = decide_group_with_recurrence( - group, - group.consumer_count, - profiles.for_target(&group.target), - horizon, - cost_model, - )? { - candidates.sort_by_key(|candidate| match candidate.provenance { - ReplacementProvenance::CseShare if decision == ShareDecision::Share => { - 0 - } - ReplacementProvenance::CseRecompute - if decision == ShareDecision::RecomputeIndependently => - { - 0 - } - ReplacementProvenance::CseShare - | ReplacementProvenance::CseRecompute => 2, - _ => 1, - }); - } - } - let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); - let costs = candidates - .iter() - .map(|candidate| { - cost_model - .grouping_state_cost(candidate, &target) - .map_or_else( - || cost_model.estimate_cost(candidate, &target), - |cost| cost.0, - ) - }) - .collect(); - Ok(RankedTargetSubDAGCandidates { - target: &group.target, - consumer_count: group.consumer_count, - candidates, - costs, - }) - }) - .collect() - } -} - -// ── Recurrence-aware cost context (issue #287) ────────────────────────── - -/// One [`RecurrenceProfile`] per discovered [`TargetSubDAGCandidates`] target, built by -/// [`CandidateLogicalASAPDAGs::recurrence_profiles`] — the "carry `RepeatingEntry.demand` -/// and relevant `DataWorkload` into ASAP-aware search/cost context" -/// half of issue #287. Looked up by `Rc` pointer identity, the same -/// currency [`CandidateLogicalASAPDAGs::candidates_for_target`]/[`GlobalSelection::for_target`] already -/// use. -/// Holds an owned `Rc` clone alongside each profile (not just its -/// raw pointer) so this map keeps every node it describes alive for as long -/// as the map itself lives — a `RecurrenceProfileMap` is safe to outlive the -/// `CandidateLogicalASAPDAGs` it was built from. Without this, a raw `*const QueryExpr` key -/// could, after the originating `CandidateLogicalASAPDAGs` (the only other owner of those -/// `Rc`s) is dropped, collide with an unrelated, later allocation that -/// happens to reuse the same freed address — silently returning a stale -/// profile for the wrong node (issue #287 review, bug 4). -#[derive(Debug, Clone)] -pub struct RecurrenceProfileMap { - profiles: HashMap<*const QueryExpr, (Rc, RecurrenceProfile)>, -} - -impl RecurrenceProfileMap { - /// The [`RecurrenceProfile`] for `target`, or - /// [`RecurrenceProfile::EMPTY`] when `target` wasn't a discovered site - /// in the [`CandidateLogicalASAPDAGs`] this map was built from (or carried no - /// recurring/one-shot/update-rate metadata at all) — always a valid, - /// "no metadata" answer, never a panic. - pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { - self.profiles - .get(&Rc::as_ptr(target)) - .map(|(_, profile)| *profile) - .unwrap_or(RecurrenceProfile::EMPTY) - } -} - -impl CandidateLogicalASAPDAGs { - /// Build one [`RecurrenceProfile`] per discovered site, by walking every - /// root's whole reachable sub-DAG (the same relational-skeleton - /// traversal [`discover_targets`] itself used to discover those sites) - /// and folding each root's own recurrence tag - /// (a normalized repeating rate or a one-time invocation count) into every - /// site reachable from it. - /// - /// `root_recurrence` is positional: `root_recurrence[i]` describes - /// `self.roots[i]` — the same order [`search_workload`]/ - /// [`search_workload_with`] were originally called with (post-CSE - /// dedup preserves both root count and order — see - /// `asap_types::pre_asap::cse::share_common_sub_dags`'s own - /// `.map(...).collect()` body). This keeps `Id` fully opaque (no `Eq`/ - /// `Hash`/`Clone` bound needed on it at all — issue #287's "keep - /// caller/query identifiers opaque" requirement) at the cost of the - /// caller keeping the two slices in step; `root_recurrence.len()` must - /// equal `self.roots.len()`. - /// - /// A shared sub-DAG reachable from more than one root aggregates every - /// reaching root's contribution — repeating roots' rates are summed and - /// one-shot roots - /// increment [`RecurrenceProfile::one_shot_consumers`] — so a summary - /// consumed by queries with different intervals gets one profile - /// reflecting all of them, per issue #287's "support a shared sub-DAG - /// consumed by queries with different intervals". - /// - /// `update_rate` is applied uniformly to every discovered site *that - /// this walk actually reached from some root* (see the "unreachable - /// sites" note below): today's - /// [`asap_types::workload::DataWorkload`] is a single - /// workload-level value (applies to every query in a `QueryWorkload`), - /// not per-target, so there is no finer-grained source to attach - /// instead. `None` when no `DataWorkload` evidence was available — - /// preserves "missing metadata" behavior for the update-rate term alone - /// even when repeating/one-shot consumer information is present. - /// - /// A parent that structurally references the same child more than once - /// (e.g. `BinaryOp{lhs: X, rhs: X}`) credits that child with one - /// contribution per reference, not one contribution per distinct node — - /// matching how [`TargetSubDAGCandidates::consumer_count`] counts that occurrence. - /// Multiplicity is propagated through the full descendant path: if the - /// repeated parent is independently evaluated twice, its child is also - /// evaluated twice. This supplies recurrence-aware selection with the - /// effective structural execution rate rather than mere reachability. - /// - /// **Unreachable sites**: [`CandidateLogicalASAPDAGs`] can contain a site no root's own - /// structural DAG actually reaches — e.g. one only ever produced by a - /// [`Replacement::Rewrite`] candidate a [`ReplacementStrategy`] invented - /// (this walk only follows [`TargetSubDAGCandidates::target`]'s own structural - /// children, the same scope [`discover_targets`] uses for the original - /// roots, never a candidate's rewritten value). Such a site gets - /// [`RecurrenceProfile::EMPTY`] — in particular, `update_rate` is - /// **not** stamped onto it — so it falls back to the ordinary - /// structural decision instead of being charged an ingest-driven - /// maintenance cost against a real evaluation/one-shot signal of - /// exactly zero, which previously made `RecomputeIndependently` win - /// there unconditionally, regardless of the site's actual - /// `consumer_count` (issue #287 review, bug 2). - /// - /// Returns [`RecurrenceError::InvalidEvaluationRate`] if any repeating - /// rate is non-finite or negative, - /// [`RecurrenceError::InvalidUpdateRate`] if `update_rate` is non-finite - /// or negative, or [`RecurrenceError::RootCountMismatch`] if - /// `root_recurrence.len() != self.roots.len()`. - pub fn recurrence_profiles( - &self, - root_recurrence: &[RootRecurrence], - update_rate: Option, - ) -> Result { - if root_recurrence.len() != self.roots.len() { - return Err(crate::recurrence::RecurrenceError::RootCountMismatch { - expected: self.roots.len(), - got: root_recurrence.len(), - }); - } - if let Some(rate) = update_rate { - crate::recurrence::validate_update_rate(rate)?; - } - for recurrence in root_recurrence { - if let RootRecurrence::Repeating(rate) = recurrence { - if !rate.0.is_finite() || rate.0 < 0.0 { - return Err(crate::recurrence::RecurrenceError::InvalidEvaluationRate( - *rate, - )); - } - } - } - - let mut rates: HashMap<*const QueryExpr, f64> = HashMap::new(); - let mut one_shot_counts: HashMap<*const QueryExpr, usize> = HashMap::new(); - // Sites actually reached by at least one root's own recurrence tag - // during the walk below — see this method's own "Unreachable - // sites" doc. - let mut reached: HashSet<*const QueryExpr> = HashSet::new(); - - for ((_, root), recurrence) in self.roots.iter().zip(root_recurrence) { - let recurrence = *recurrence; - let root_ptr = Rc::as_ptr(root); - // Carry path multiplicity transitively. If a shared ancestor is - // referenced twice, every descendant below an independently - // recomputed occurrence is evaluated twice as well; stopping - // expansion after the first pointer visit undercounts exactly - // the effective-consumer rate recurrence-aware costing needs. - let mut queue: VecDeque<(*const QueryExpr, usize)> = VecDeque::new(); - queue.push_back((root_ptr, 1)); - - while let Some((ptr, path_count)) = queue.pop_front() { - contribute( - ptr, - path_count, - recurrence, - &mut rates, - &mut one_shot_counts, - &mut reached, - ); - // Every reachable node was itself discovered as its own - // `TargetSubDAGCandidates` (`discover_targets` walks the identical - // relational-skeleton scope) — its own `target` is the - // canonical `Rc` to read children off. - if let Some(group) = self.groups.get(&ptr) { - for (child, edge_count) in direct_child_counts(&group.target) { - queue.push_back(( - child, - path_count - .checked_mul(edge_count) - .expect("query DAG path multiplicity overflowed usize"), - )); - } - } - } - } - - let mut profiles = HashMap::with_capacity(self.order.len()); - for ptr in &self.order { - let rate = rates.get(ptr).copied().unwrap_or(0.0); - let evaluation_rate = (rate > 0.0).then_some(crate::recurrence::EvaluationRate(rate)); - let one_shot_consumers = one_shot_counts.get(ptr).copied().unwrap_or(0); - // Bug 2 fix (see "Unreachable sites" above): only a reached - // site carries the caller-supplied `update_rate`. - let site_update_rate = if reached.contains(ptr) { - update_rate - } else { - None - }; - let node = Rc::clone(&self.groups[ptr].target); - profiles.insert( - *ptr, - ( - node, - RecurrenceProfile { - evaluation_rate, - one_shot_consumers, - update_rate: site_update_rate, - }, - ), - ); - } - - Ok(RecurrenceProfileMap { profiles }) - } - - /// Derive per-target recurrence profiles directly from the normalized - /// query and data workloads. This is the authoritative bridge from the - /// public workload model into recurrence-aware candidate costing. - /// `root_workload_entries[i]` explicitly identifies the normalized - /// workload entry for `self.roots[i]`; callers need not arrange roots in - /// the batch-then-repeating storage order. - pub fn recurrence_profiles_from_workload( - &self, - workload: &QueryWorkload, - data_workload: Option<&DataWorkload>, - // For each `CandidateLogicalASAPDAGs::roots[i]`, the explicit index of its - // corresponding normalized workload entry. - root_workload_entries: &[usize], - now_ms: u64, - horizon: Option, - ) -> Result { - workload.validate()?; - if let Some(data) = data_workload { - data.validate()?; - } - if let Some(horizon) = horizon { - if !horizon.0.is_finite() || horizon.0 <= 0.0 { - return Err(crate::recurrence::RecurrenceError::InvalidHorizon(horizon)); - } - } - if root_workload_entries.len() != self.roots.len() { - return Err(crate::recurrence::RecurrenceError::RootCountMismatch { - expected: self.roots.len(), - got: root_workload_entries.len(), - }); - } - let entries: Vec<_> = workload.entries().collect(); - let mut recurrences = Vec::with_capacity(root_workload_entries.len()); - for &index in root_workload_entries { - let entry = entries.get(index).ok_or( - crate::recurrence::RecurrenceError::InvalidWorkloadEntry { - index, - entry_count: entries.len(), - }, - )?; - let recurrence = match &entry.recurrence { - QueryRecurrence::OneTime { invocations, .. } => RootRecurrence::OneShotCount( - usize::try_from(*invocations).unwrap_or(usize::MAX), - ), - QueryRecurrence::Repeated(RepeatedDemand::FixedInterval(interval)) - | QueryRecurrence::Repeated(RepeatedDemand::FixedIntervalAt { interval, .. }) => { - RootRecurrence::Repeating(evaluation_rate_of([*interval])?.unwrap()) - } - QueryRecurrence::Repeated(RepeatedDemand::Scheduled(schedule)) => { - let Some(horizon) = horizon else { - return Err(crate::recurrence::RecurrenceError::MissingHorizon); - }; - let end_ms = now_ms.saturating_add((horizon.0 * 1000.0) as u64); - let count = schedule - .iter() - .filter(|at| at.0 >= now_ms && at.0 <= end_ms) - .count(); - RootRecurrence::Repeating(crate::recurrence::EvaluationRate( - count as f64 / horizon.0, - )) - } - QueryRecurrence::Repeated(RepeatedDemand::EstimatedRate(estimate)) => { - if !estimate.is_fresh_at(now_ms) { - RootRecurrence::Unknown - } else { - RootRecurrence::Repeating(crate::recurrence::EvaluationRate( - estimate.expected_rate.0, - )) - } - } - QueryRecurrence::Unknown => RootRecurrence::Unknown, - }; - recurrences.push(recurrence); - } - let update_rate = data_workload - .and_then(|data| data.ingestion_rate.value_at(now_ms)) - .map(|rate| UpdateRate(rate.0)); - self.recurrence_profiles(&recurrences, update_rate) - } - - /// Associate every discovered target with the normalized workload entries - /// whose roots can reach it. - pub(crate) fn workload_entries_by_target( - &self, - workload: &QueryWorkload, - root_workload_entries: &[usize], - ) -> Result>, RecurrenceError> { - let entry_count = workload.entries().count(); - if root_workload_entries.len() != self.roots.len() { - return Err(RecurrenceError::RootCountMismatch { - expected: self.roots.len(), - got: root_workload_entries.len(), - }); - } - let mut bindings: HashMap<*const QueryExpr, HashSet> = HashMap::new(); - for ((_, root), &entry_index) in self.roots.iter().zip(root_workload_entries) { - if entry_index >= entry_count { - return Err(RecurrenceError::InvalidWorkloadEntry { - index: entry_index, - entry_count, - }); } - let mut seen = HashSet::new(); - let mut queue = VecDeque::from([Rc::as_ptr(root)]); - while let Some(ptr) = queue.pop_front() { - if !seen.insert(ptr) { - continue; - } - bindings.entry(ptr).or_default().insert(entry_index); - if let Some(group) = self.groups.get(&ptr) { - queue.extend( - direct_child_counts(&group.target) - .into_iter() - .map(|(child, _)| child), - ); + ReplacementProvenance::CseRecompute => { + let Replacement::SubDAG(rc) = &candidate.replacement else { + return None; + }; + if Rc::ptr_eq(rc, &group.target) + || rc.as_ref() != group.target.as_ref() + || recompute.replace(candidate).is_some() + { + return None; } } + _ => {} } - Ok(bindings - .into_iter() - .map(|(ptr, entries)| { - let mut entries: Vec<_> = entries.into_iter().collect(); - entries.sort_unstable(); - (ptr, entries) - }) - .collect()) - } -} - -/// Record `times` occurrences of `recurrence` against `ptr` — `times > 1` -/// when a single parent structurally references `ptr` more than once (see -/// [`CandidateLogicalASAPDAGs::recurrence_profiles`]'s own doc on edge multiplicity). -/// A no-op for `times == 0` (an `Rc` returned as a `direct_child_counts` -/// child always has `edge_count >= 1` in practice, but this keeps the -/// helper correct regardless). -fn contribute( - ptr: *const QueryExpr, - times: usize, - recurrence: RootRecurrence, - rates: &mut HashMap<*const QueryExpr, f64>, - one_shot_counts: &mut HashMap<*const QueryExpr, usize>, - reached: &mut HashSet<*const QueryExpr>, -) { - if times == 0 { - return; - } - reached.insert(ptr); - match recurrence { - RootRecurrence::Repeating(rate) => { - *rates.entry(ptr).or_insert(0.0) += rate.0 * times as f64; - } - RootRecurrence::OneShotCount(count) => { - *one_shot_counts.entry(ptr).or_insert(0) += count.saturating_mul(times); - } - RootRecurrence::Unknown => {} } + Some((share?, recompute?)) } - -/// One [`TargetSubDAGCandidates`]'s candidates, ranked best-first by -/// [`CandidateLogicalASAPDAGs::cost_sorted`]. -#[derive(Debug)] -pub struct RankedTargetSubDAGCandidates<'a> { - pub target: &'a Rc, - pub consumer_count: usize, - pub candidates: Vec<&'a ReplacementSubDAG>, - /// `costs[i]` is `candidates[i]`'s own grouping-state cost when available, - /// and its [`CostModel::estimate_cost`] otherwise - /// estimate — aligned index-for-index with `candidates`, one number per - /// candidate, for a caller that wants an actual `f64` next to each - /// candidate (e.g. "candidate A costs ≈ X, candidate B costs ≈ Y") and - /// not just `candidates`' own relative order. `f64::NAN` throughout - /// unless `cost_model` overrides `estimate_cost` — see that method's own - /// doc. - pub costs: Vec, -} - -/// Rank `group`'s candidates best-first under `cost_model`, per the module -/// docs' "Cost-based final selection" section. Falls back to discovery -/// order whenever there's nothing to rank (0 or 1 candidates) or this -/// module doesn't have a defined `CostModel` comparison for the shape it -/// sees — it never invents one. -fn rank_group<'a>( - group: &'a TargetSubDAGCandidates, - cost_model: &dyn CostModel, -) -> Vec<&'a ReplacementSubDAG> { - let mut ranked: Vec<&ReplacementSubDAG> = group.candidates.iter().collect(); - if ranked.len() <= 1 { - return ranked; - } - - // Shape 1: the exact `SharedSubDAGStrategy` share-vs-recompute pair — - // rank via `CostModel::cse_share_decision`, the same comparison - // the local CSE ranking path already uses. - if cse_candidate_pair(group).is_some() { - if let Some(prefer_target) = cse_preference(group, cost_model) { - ranked.sort_by_key(|c| match c.provenance { - ReplacementProvenance::CseShare if prefer_target => 0, - ReplacementProvenance::CseRecompute if !prefer_target => 0, - ReplacementProvenance::CseShare | ReplacementProvenance::CseRecompute => 2, - _ => 1, - }); +/// Direct relational-skeleton children and their edge multiplicities. +/// `Concat` is transparent, matching [`walk_children`]'s site scope. +pub fn direct_child_counts(node: &OperatorNode) -> Vec<(*const OperatorNode, usize)> { + fn push(children: &mut Vec<(*const OperatorNode, usize)>, child: &Rc) { + let ptr = Rc::as_ptr(child); + match children.iter_mut().find(|(existing, _)| *existing == ptr) { + Some((_, count)) => *count += 1, + None => children.push((ptr, 1)), } - return ranked; - } - - // Shape 2: independent and Hydra grouping alternatives for the same - // sketch algorithms. When deployment statistics provide a subpopulation - // estimate, compare N independent states with the shared grid directly. - let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); - let has_hydra = ranked.iter().any(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { - return false; - }; - summary_grouping(node).is_some_and(|grouping| { - matches!(grouping, GroupingStrategy::SharedMultiSubpopulation { .. }) - }) - }); - let grouping_costs: Option> = if has_hydra { - ranked - .iter() - .map(|candidate| { - cost_model - .grouping_state_cost(candidate, &target) - .map(|cost| cost.0) - }) - .collect() - } else { - None - }; - if let Some(costs) = grouping_costs { - let by_ptr: HashMap<*const ReplacementSubDAG, f64> = ranked - .iter() - .zip(costs) - .map(|(candidate, cost)| (*candidate as *const ReplacementSubDAG, cost)) - .collect(); - ranked.sort_by(|a, b| { - by_ptr[&(*a as *const ReplacementSubDAG)] - .total_cmp(&by_ptr[&(*b as *const ReplacementSubDAG)]) - }); - return ranked; } - // Shape 3: `SketchAlgorithmStrategy`'s sketch-family candidates (every - // candidate is a `Summary` that realizes a `SketchAlgorithm`) — rank via - // `CostModel::rank_candidates`, the same hook `realizations_for_intent` - // itself consults. - if let Some(intent) = bindable_intent(&group.target) { - let kinds: Option> = ranked - .iter() - .map(|c| match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, - }) - .collect(); - if let Some(kinds) = kinds { - let order = crate::cost_model::validated_candidate_ranking(cost_model, intent, &kinds); - ranked.sort_by_key(|c| { - let kind = match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, - }; - kind.and_then(|k| order.iter().position(|o| *o == k)) - .unwrap_or(usize::MAX) - }); - return ranked; + fn collect(node: &OperatorNode, children: &mut Vec<(*const OperatorNode, usize)>) { + if let Some(NonASAPOp::Concat { + children: concat_children, + .. + }) = node.non_asap() + { + for c in concat_children { + collect(c, children); + } + return; } - } - - // A target may be handled by more than one strategy (for example, a - // shared aggregate has both bound-summary and share/recompute rewrite - // candidates). No shape-specific hook spans those different candidate - // types, so compare the numeric estimates the CostModel exposes for that - // purpose. `total_cmp` gives deterministic placement to a model's NaN - // placeholders without dropping any candidate. - ranked.sort_by(|a, b| { - match ( - cost_model.candidate_cost(a, &target), - cost_model.candidate_cost(b, &target), - ) { - (Some(a), Some(b)) => a.0.total_cmp(&b.0), - (Some(_), None) => std::cmp::Ordering::Less, - (None, Some(_)) => std::cmp::Ordering::Greater, - (None, None) => cost_model - .estimate_cost(a, &target) - .total_cmp(&cost_model.estimate_cost(b, &target)), + for child in node.children() { + push(children, child); } - }); - ranked -} - -/// For a group whose candidates are all [`Replacement::Rewrite`] (the -/// [`SharedSubDAGStrategy`] shape): does [`CostModel::cse_share_decision`] -/// prefer the candidate that shares `group.target`'s own `Rc` (`true`), or -/// the one that recomputes independently (`false`)? `None` when there's no -/// real comparison to make — fewer than 2 consumers (mirrors -/// [`SharedSubDAGStrategy::matches`]'s own gate), or `group.target` can't -/// actually be bound at all (no candidate and no logical fallback — never -/// expected in practice for a target that's already part of a legitimate -/// workload DAG, but this degrades to "keep discovery order" rather than -/// panicking). -fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> Option { - if group.consumer_count < 2 { - return None; - } - let bound = realize_one(&group.target, cost_model)?; - let candidate = CseCandidate { - sub_dag: &group.target, - bound_summary: &bound, - consumer_count: group.consumer_count, - }; - Some(match cost_model.cse_share_decision(&candidate) { - ShareDecision::Share => true, - ShareDecision::RecomputeIndependently => false, - }) -} - -/// [`cse_preference`] only needs one representative bound [`SummaryNode`] -/// for `target` (to build a [`CseCandidate`] for -/// [`CostModel::cse_share_decision`]), not the full ranked candidate list -/// [`SketchAlgorithmStrategy::replacements`] returns — so this just reuses -/// [`realize_child`], the same rank-and-take-first helper -/// `construct_summary_agg`'s own recursion and -/// [`crate::cost_model::DefaultCostModel::estimate_cost`] already use, -/// wrapped to swallow the (here, uninteresting) error into `None`. -fn realize_one(target: &Rc, cost_model: &dyn CostModel) -> Option> { - realize_child(target, cost_model).ok() -} - -/// The `SketchAlgorithm` a bound [`Replacement::Summary`] candidate ultimately -/// realizes, if any (`None` for an `ExactAggregate`/pass-through -/// `Summary` — nothing to rank against another `SketchAlgorithm`). -/// -/// Mirrors this module's own `#[cfg(test)]`-only `summary_family_algorithm` -/// helper (in the test module below), which does the identical -/// `SummaryEstimate`-unwrap-then-match for that module's own tests; that -/// copy is test-only, so this needs its own for real (non-test) ranking -/// code — the same "duplicate a small, self-contained traversal rather than -/// restructure a test helper" call this file's own top doc already makes -/// for [`discover_targets`]. -fn sketch_kind_of(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_kind_of(summary_input), - SummaryExpr::SummaryAgg { - family: FieldDataType::Sketch(kind, _), - .. - } => Some(kind.algorithm().clone()), - _ => None, } -} -/// The grouping strategy used by a bound summary candidate, unwrapping its -/// readout node when necessary. -fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => summary_grouping(summary_input), - SummaryExpr::SummaryAgg { grouping, .. } => Some(grouping), - _ => None, - } + let mut children = Vec::new(); + collect(node, &mut children); + children } +// ── GlobalSelection: assemble a DAG from given choices ─────────────────── -// ── global_selection ───────────────────────────────────────────────────── - -/// One target sub-DAG's selected choice and usage information — the answer -/// [`CandidateLogicalASAPDAGs::global_selection`] commits to for one site, after folding in -/// every ancestor [`SharedSubDAGStrategy`] decision on the path from a -/// workload root to this site. See the module docs' "Whole-plan -/// (cross-group) selection" section for the full recurrence. -/// -/// Contrast with [`RankedTargetSubDAGCandidates`] ([`CandidateLogicalASAPDAGs::cost_sorted`]'s output): -/// that ranks every candidate for one target in isolation and never commits -/// to just one; this commits to exactly one (or none), and the count it -/// ranks against — [`Self::effective_consumer_count`] — can differ from the -/// target's own raw structural [`TargetSubDAGCandidates::consumer_count`] whenever an -/// ancestor's choice changes how many times this site truly runs. Use -/// `cost_sorted` to inspect every alternative for a site; use -/// `global_selection` when you need this module's best single answer, -/// accounting for cross-target interaction where it knows how to. +/// One target sub-DAG's chosen candidate and usage counts: the input +/// [`GlobalSelection`] assembles a DAG from. Building a DAG from given +/// choices needs no cost model; whoever makes the choices (enumeration here, +/// or the legacy cost-based selection in plan selection) fills these in. #[derive(Debug)] pub struct TargetSubDAGSelection<'a> { /// The target sub-DAG this selection is for. - pub target: &'a Rc, + pub target: &'a Rc, /// [`TargetSubDAGCandidates::consumer_count`] — how many operator-child positions /// directly reference `target`, ignoring every ancestor's own choice. pub consumer_count: usize, /// How many times `target`'s computation actually runs once every - /// ancestor's own selected candidate is accounted for — see - /// [`multiplier`]'s doc for the exact recurrence. Equal to + /// ancestor's own selected candidate is accounted for. Equal to /// `consumer_count` unless some ancestor on a path from a root to this - /// site has a [`SharedSubDAGStrategy`] alternative that chose - /// [`ShareDecision::RecomputeIndependently`]. + /// site has a [`SharedSubDAGStrategy`] alternative that chose to + /// recompute independently. pub effective_consumer_count: usize, /// The candidate chosen for this target, or `None` when no replacement /// is selected. The candidate set need not be empty: an unproven DDSketch @@ -5052,47 +4426,23 @@ pub struct TargetSubDAGSelection<'a> { /// DAG assembly then preserves exact computation at this target where /// supported, while independently selected children may remain visible. pub chosen: Option<&'a ReplacementSubDAG>, - /// When `chosen` is a [`Replacement::ExactComposition`]: the child - /// decision it was committed together with, and the cost comparison - /// that justified it — the explicit target-to-decision provenance - /// chain (issue #171). - pub composition: Option>, } -/// Why [`CandidateLogicalASAPDAGs::global_selection`] committed an exact composition at a -/// site: which child candidate it composes with, and the -/// cost-units-per-second comparison against the raw fallback that it won. -#[derive(Debug)] -pub struct CompositionDecision<'a> { - /// The exact child/operation pair validated by the search accuracy model. - pub plan: Rc, - /// The child target the composed operator consumes. - pub child_target: &'a Rc, - /// For a read-time operation: the child's own candidate committed alongside - /// (the summary readout the operator folds). `None` for an update-path - /// transform, whose input is raw update data — its cost is charged to - /// the maintained summary *above* it instead. - pub child_candidate: Option<&'a ReplacementSubDAG>, - /// The composed plan's recurring rate — `read_operation_plan_cost_rate` - /// or `maintenance_operation_plan_cost_rate`. - pub cost_rate: CostRate, - /// `raw_recompute_cost_rate` — the `KeepPreAsap` baseline it beat. - pub baseline_rate: CostRate, - /// The statistics (and their provenance) both rates were computed from. - pub inputs: ExactCompositionCostInputs, -} - -/// [`CandidateLogicalASAPDAGs::global_selection`]'s result: one [`TargetSubDAGSelection`] per -/// discovered site, in the same discovery order [`CandidateLogicalASAPDAGs::target_subdag_candidates`]/ -/// [`CandidateLogicalASAPDAGs::cost_sorted`] use. +/// One [`TargetSubDAGSelection`] per discovered site, in the same discovery +/// order [`CandidateLogicalASAPDAGs::target_subdag_candidates`] uses, plus the +/// DAG assembly over those choices. #[derive(Debug)] pub struct GlobalSelection<'a> { - order: Vec<*const QueryExpr>, - groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'a>>, + order: Vec<*const OperatorNode>, + groups: HashMap<*const OperatorNode, TargetSubDAGSelection<'a>>, + /// The validated plan of each site whose chosen candidate is a + /// [`Replacement::ExactComposition`]. + composition_plans: HashMap<*const OperatorNode, Rc>, /// [`Self::assemble_selected_dag`]'s memo — one bound node per target for the /// life of this selection, so two parents composing over one shared - /// child get the *same* `Rc`. - assembled_nodes: RefCell>>, + /// child get the *same* `Rc` (a kept pre-ASAP sub-DAG + /// shared by two parents stays one `Rc` the same way). + assembled_nodes: RefCell>>, } fn normalize_cross_input_equi_predicate( @@ -5100,15 +4450,17 @@ fn normalize_cross_input_equi_predicate( left_width: usize, total_width: usize, ) -> Option { - let QueryExpr::Compare { + let ScalarExpr::Compare { left, - op: asap_types::pre_asap::CompareOpKind::Eq, + op: asap_types::ir::scalar::CompareOpKind::Eq, right, - } = pred.0.as_ref() + semantics, + } = &pred.0 else { return None; }; - let (QueryExpr::Column(left_id), QueryExpr::Column(right_id)) = (left.as_ref(), right.as_ref()) + let (ScalarExpr::Column(left_id), ScalarExpr::Column(right_id)) = + (left.as_ref(), right.as_ref()) else { return None; }; @@ -5121,23 +4473,30 @@ fn normalize_cross_input_equi_predicate( } else { return None; }; - Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left_id)), - op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right_id)), - }))) -} - -fn relational_join_guarantee( - left: Option<&ResultGuarantee>, - right: Option<&ResultGuarantee>, -) -> Option { - left.zip(right) - .filter(|(left, right)| left.is_exact() && right.is_exact()) - .map(|_| ResultGuarantee::exact("RelationalJoin over exact inputs")) + Some(Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(left_id)), + op: asap_types::ir::scalar::CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(right_id)), + semantics: *semantics, + })) } impl<'a> GlobalSelection<'a> { + /// A selection over `groups`, listed in `order`. `composition_plans` + /// holds the validated plan of every site that chose an exact composition. + pub fn new( + order: Vec<*const OperatorNode>, + groups: HashMap<*const OperatorNode, TargetSubDAGSelection<'a>>, + composition_plans: HashMap<*const OperatorNode, Rc>, + ) -> Self { + Self { + order, + groups, + composition_plans, + assembled_nodes: RefCell::new(HashMap::new()), + } + } + /// One selection per discovered target sub-DAG, in discovery order. pub fn target_selections(&self) -> impl Iterator> { self.order.iter().map(move |ptr| &self.groups[ptr]) @@ -5146,62 +4505,69 @@ impl<'a> GlobalSelection<'a> { /// The selection for `target`, if `target`'s own `Rc` is a discovered /// site (i.e. `Rc::ptr_eq` to some node reachable from the workload's /// roots). - pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { + pub fn for_target(&self, target: &Rc) -> Option<&TargetSubDAGSelection<'a>> { self.groups.get(&Rc::as_ptr(target)) } /// Link this selection's per-site decisions into one data_state-validated /// post-ASAP DAG rooted at `target` — the one place a committed - /// composition's child *reference* becomes an actual `Rc` + /// composition's child *reference* becomes an actual `Rc` /// edge (issue #171). `None` if `target` is not a discovered site. /// /// Per site: a [`Replacement::ExactComposition`] uses its validated /// operation/child plan, retaining the search model's guarantee; - /// a [`Replacement::Summary`] is + /// a bound-summary [`Replacement::SubDAG`] is /// re-linked so its `SummaryAgg` child is the child target's own /// DAG assembly whenever that is phase-legal beneath maintenance /// (so a child that chose an `ValueOperationAtIngestionTime` actually ends up under - /// the summary); a [`Replacement::Rewrite`] or an unmatched site stays - /// the conservative `KeepPreAsap`. Memoized by target identity, so a - /// shared inner summary is one `Rc` no matter how many roots reach it. + /// the summary); a logical-rewrite [`Replacement::SubDAG`] is kept + /// as it is (exact); an unmatched site keeps its own operator with each + /// child assembled independently ([`Self::assemble_residual`]). + /// Memoized by target identity, so a shared inner summary is one `Rc` + /// no matter how many roots reach it. pub fn assemble_selected_dag( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { if !self.groups.contains_key(&Rc::as_ptr(target)) { return Ok(None); } self.assemble_target(target).map(Some) } - /// Assemble a complete query result, including an exact-state readout when + /// Assemble a complete query result, including an exact-state evaluation when /// needed. `assemble_selected_dag` also serves internal state frontiers; /// callers exposing query results must use this boundary instead. pub fn assemble_selected_query( &self, - target: &Rc, - ) -> Result>, RealizationError> { + target: &Rc, + ) -> Result>, RealizationError> { self.assemble_selected_dag(target)? .map(|node| finalize_query_candidate(node, target)) .transpose() } - fn assemble_target(&self, target: &Rc) -> Result, RealizationError> { + fn assemble_target( + &self, + target: &Rc, + ) -> Result, RealizationError> { let ptr = Rc::as_ptr(target); if let Some(node) = self.assembled_nodes.borrow().get(&ptr) { return Ok(Rc::clone(node)); } // A selected summary that realizes its inner aggregate, instead of - // hiding it in `KeepPreAsap`, is kept; lifecycle assignment decides + // hiding it in `KeepPreAsap`, is kept; materialization assignment decides // whether it runs in precompute or at query time. let selected_composed_summary = self .groups .get(&ptr) .and_then(|sel| sel.chosen) - .is_some_and(|candidate| matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { child, .. } - if !matches!(&child.expr, SummaryExpr::KeepPreAsap(raw) if contains_aggregate(raw))))); + .is_some_and(|candidate| { + matches!(&candidate.replacement, + Replacement::SubDAG(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) + if child.contains_asap() || !contains_aggregate(child))) + }); let node = if query_time_nested_sum(target) && !selected_composed_summary { self.assemble_residual(target)? } else { @@ -5212,14 +4578,14 @@ impl<'a> GlobalSelection<'a> { .map(|c| &c.replacement) { None => self.assemble_residual(target)?, - Some(Replacement::Rewrite(rewritten)) => keep_pre_asap(rewritten)?, - Some(Replacement::Summary(node)) => self.relink_summary(node, target)?, + Some(Replacement::SubDAG(node)) if node.contains_asap() => { + self.relink_summary(node, target)? + } + Some(Replacement::SubDAG(kept)) => retain_exact(kept)?, Some(Replacement::ExactComposition(_)) => Rc::clone( - &self.groups[&ptr] - .composition - .as_ref() - .expect("selected compositions have a validated decision") - .plan, + self.composition_plans + .get(&ptr) + .expect("selected compositions have a validated plan"), ), } }; @@ -5229,130 +4595,124 @@ impl<'a> GlobalSelection<'a> { Ok(node) } - /// Preserve composable query-time value operators in post-ASAP form even - /// when the operator itself has no summary realization. Its child is - /// assembled independently, so a selected summary remains visible - /// beneath `Project`/`Filter`/`Sort`/`Limit` instead of being swallowed by - /// one opaque `KeepPreAsap` sub-DAG. + /// Keep `target`'s own operator and assemble each child independently, + /// so a selected summary remains visible beneath a relational operator + /// that has no summary realization of its own instead of being + /// swallowed by one opaque kept sub-DAG. Every child that is a + /// discovered target is assembled (and finalized to query-time values); + /// any other child is kept as it is. The guarantee is composed from the + /// assembled children: all exact → exact; exactly one child → that + /// child's guarantee; otherwise unknown. An inner `Join` first has its + /// cross-input equi-predicate normalized; any other join is kept whole. fn assemble_residual( &self, - target: &Rc, - ) -> Result, RealizationError> { - if let QueryExpr::Join { + target: &Rc, + ) -> Result, RealizationError> { + if target.children().is_empty() { + // A leaf has nothing to assemble beneath it: keep it as it is. + return retain_exact(target); + } + let mut operator = target.operator.clone(); + if let Operator::NonASAP(NonASAPOp::Join { left, right, kind, pred, - } = target.as_ref() + }) = &mut operator { - let left_width = left.output_schema()?.fields.len(); - let total_width = left_width + right.output_schema()?.fields.len(); - let normalized_pred = matches!(kind, asap_types::pre_asap::JoinKind::Inner) + let left_width = left.schema.fields.len(); + let total_width = left_width + right.schema.fields.len(); + let normalized_pred = matches!(kind, JoinKind::Inner) .then(|| normalize_cross_input_equi_predicate(pred, left_width, total_width)) .flatten(); - let Some(pred) = normalized_pred else { - return keep_pre_asap(target); + let Some(normalized) = normalized_pred else { + return retain_exact(target); }; - let left = finalize_query_candidate(self.assemble_target(left)?, left)?; - let right = finalize_query_candidate(self.assemble_target(right)?, right)?; - let guarantee = - relational_join_guarantee(left.guarantee.as_ref(), right.guarantee.as_ref()); - let node = Rc::new(SummaryNode { - expr: SummaryExpr::RelationalJoin { - left, - right, - kind: kind.clone(), - pred, - pruning: None, - }, - schema: lift(&target.output_schema()?), - guarantee, - }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; - return Ok(node); + *pred = normalized; } - let (child_target, operation) = match target.as_ref() { - QueryExpr::Project { - cols, - qualifier, - child, - } => ( - child, - ValueOperation::Project { - cols: cols.clone(), - qualifier: qualifier.clone(), - }, - ), - QueryExpr::Filter { pred, child } => { - (child, ValueOperation::Filter { pred: pred.clone() }) + let mut failure = None; + let mut children = Vec::new(); + let operator = operator.map_children(|child| { + if failure.is_some() { + return Rc::clone(child); + } + let assembled = if self.groups.contains_key(&Rc::as_ptr(child)) { + self.assemble_target(child) + .and_then(|node| finalize_query_candidate(node, child)) + } else { + Ok(Rc::clone(child)) + }; + match assembled { + Ok(node) => { + children.push(Rc::clone(&node)); + node + } + Err(error) => { + failure = Some(error); + Rc::clone(child) + } } - QueryExpr::Sort { - keys, - partition_by, - child, - } => ( - child, - ValueOperation::Sort { - keys: keys.clone(), - partition_by: partition_by.clone(), - }, - ), - QueryExpr::Limit { n, offset, child } => ( - child, - ValueOperation::Limit { - n: *n, - offset: *offset, - partition_by: match child.as_ref() { - QueryExpr::Sort { partition_by, .. } => partition_by.clone(), - _ => Default::default(), - }, - }, - ), - QueryExpr::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } if query_time_nested_sum(target) => ( - child, - ValueOperation::Exact(ExactOperation::Aggregate { - reduction: reduction.clone(), - measures: measures.clone(), - output_names: output_names.clone(), - filters: filters.clone(), - having: having.clone(), - }), - ), - _ => return keep_pre_asap(target), - }; - let child = finalize_query_candidate(self.assemble_target(child_target)?, child_target)?; - let guarantee = child.guarantee.clone(); - let node = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child, - operation, - timing: ExecutionTiming::QueryTime, - }, - schema: lift(&target.output_schema()?), - guarantee, }); - validate_execution_data_states_at(&node, ExecutionDataState::QUERY_ROWS)?; + if let Some(error) = failure { + return Err(error); + } + // An operator that computes new values from its input rows has no + // sound accuracy composition over an approximate input (e.g. `max` + // over a quantile evaluation's rank error). Without a selected + // composition such a node stays an exact pre-ASAP sub-DAG; only the + // read-time nested SUM keeps its assembled children. + let computes_values = matches!( + target.non_asap(), + Some( + NonASAPOp::Aggregate { .. } + | NonASAPOp::BinaryOp { .. } + | NonASAPOp::SQLWindowFunc { .. } + ) + ) && !query_time_nested_sum(target); + let approximate_input = children.iter().any(|child| { + !child + .guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }); + if computes_values && approximate_input { + return retain_exact(target); + } + let guarantee = match children.as_slice() { + [child] => child.guarantee.clone(), + children + if children.iter().all(|child| { + child + .guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }) => + { + Some(ResultGuarantee::exact(format!( + "{} over exact inputs", + target.operator.kind_name() + ))) + } + _ => None, + }; + let node = Rc::new( + OperatorNode::with_schema(operator, target.schema.clone()).with_guarantee(guarantee), + ); + validate_maintained(&node, ExecutionTiming::QueryTime)?; Ok(node) } - /// Re-link a bound `Summary` candidate's `SummaryAgg` child to the + /// Re-link a bound summary candidate's `SummaryAgg` child to the /// child target's own DAG assembly when that is legal beneath /// maintenance; otherwise keep the candidate exactly as constructed. fn relink_summary( &self, - node: &Rc, - target: &Rc, - ) -> Result, RealizationError> { - let QueryExpr::Aggregate { + node: &Rc, + target: &Rc, + ) -> Result, RealizationError> { + let Some(NonASAPOp::Aggregate { child: pre_child, .. - } = target.as_ref() + }) = target.non_asap() else { return Ok(Rc::clone(node)); }; @@ -5377,16 +4737,16 @@ impl<'a> GlobalSelection<'a> { /// A mergeable outer SUM over a relationally wrapped aggregate is a read-time /// reduction of the inner summary values. Maintaining the outer SUM directly -/// would hide that inner temporal aggregate inside `KeepPreAsap` and lose its -/// independently selected summary. -fn query_time_nested_sum(target: &QueryExpr) -> bool { - let QueryExpr::Aggregate { +/// would hide that inner temporal aggregate inside one kept sub-DAG and lose +/// its independently selected summary. +fn query_time_nested_sum(target: &OperatorNode) -> bool { + let Some(NonASAPOp::Aggregate { measures, filters, having: None, child, .. - } = target + }) = target.non_asap() else { return false; }; @@ -5395,13 +4755,15 @@ fn query_time_nested_sum(target: &QueryExpr) -> bool { && contains_aggregate(child) } -fn contains_aggregate(expr: &QueryExpr) -> bool { - match expr { - QueryExpr::Aggregate { .. } => true, - QueryExpr::Project { child, .. } - | QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } => contains_aggregate(child), +fn contains_aggregate(expr: &OperatorNode) -> bool { + match expr.non_asap() { + Some(NonASAPOp::Aggregate { .. }) => true, + Some( + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. }, + ) => contains_aggregate(child), _ => false, } } @@ -5409,50 +4771,55 @@ fn contains_aggregate(expr: &QueryExpr) -> bool { /// Rebuild `node` (a `SummaryAgg`, possibly under a `SummaryEstimate`) with /// `new_child` as the `SummaryAgg`'s child, if the result still validates /// as maintained state; otherwise return `node` unchanged. -fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { +fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } => { + }) => { let inner = relink_agg_child(summary_input, new_child); if Rc::ptr_eq(&inner, summary_input) { return Rc::clone(node); } - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: inner, - query: query.clone(), - }, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - }) + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: inner, + query: query.clone(), + }), + node.schema.clone(), + ) + .with_guarantee(node.guarantee.clone()), + ) } - SummaryExpr::SummaryAgg { + Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, grouping, filter, - } => { + }) => { if Rc::ptr_eq(child, new_child) { return Rc::clone(node); } - let rebuilt = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::clone(new_child), - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: filter.clone(), - }, - schema: node.schema.clone(), - guarantee: node.guarantee.clone(), + // The same summary over a re-placed input keeps its coverage. + let rebuilt = std::rc::Rc::new(OperatorNode { + coverage: node.coverage.clone(), + ..OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: Rc::clone(new_child), + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: filter.clone(), + }), + node.schema.clone(), + ) + .with_guarantee(node.guarantee.clone()) }); - match validate_execution_data_states_at(&rebuilt, ExecutionDataState::INGESTION_SUMMARY) - { + match validate_maintained(&rebuilt, ExecutionTiming::IngestionTime) { Ok(_) => rebuilt, Err(_) => Rc::clone(node), } @@ -5461,4271 +4828,2225 @@ fn relink_agg_child(node: &Rc, new_child: &Rc) -> Rc) -> Option<&Rc> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => maintained_summary(summary_input), - SummaryExpr::SummaryAgg { .. } => Some(node), - _ => None, - } -} - -fn is_composition_candidate(candidate: &ReplacementSubDAG) -> bool { - matches!(candidate.replacement, Replacement::ExactComposition(_)) -} - -/// Everything [`CandidateLogicalASAPDAGs::global_selection`] threads between sites for -/// exact compositions (issue #171): child candidates already committed by -/// an earlier parent, and the maintained summary above each site. -#[derive(Default)] -struct CompositionContext { - /// child target ptr → the child's candidate an ancestor's composition - /// already committed to (a later parent must compose with the *same* - /// one, and the child's own selection is forced to it). - committed_child: HashMap<*const QueryExpr, *const ReplacementSubDAG>, - /// site ptr → the maintained `SummaryAgg` directly above it, when its - /// parent chose a bound `Summary` — what an `ValueOperationAtIngestionTime` here feeds. - maintaining_parent: HashMap<*const QueryExpr, Rc>, -} +// ── default_strategies ────────────────────────────────────────────────── -/// One eligible composed alternative at a site, before the cheapest wins. -struct CompositionOption<'a> { - candidate: &'a ReplacementSubDAG, - decision: CompositionDecision<'a>, +/// The context-free strategies [`search_workload`] runs with the built-in +/// accuracy models. Workload-dependent strategies such as +/// [`RollupStrategy`] and [`AccuracyReconciliationStrategy`] (issue #273, +/// cross-consumer accuracy reconciliation for CSE sharing — see that +/// module's own docs) are added by [`search_workload`] after CSE and target +/// discovery, when their sibling context exists. +/// [`crate::pass1::explanation::explain_replacements`] (issue #257) uses +/// this same set (via [`search_workload`]) rather than keeping a second, +/// explanation-specific list to stay in sync with. +/// +/// `AvgToSumOverCountStrategy` is +/// included here (issue #253) even though it's a +/// [`Replacement::Rewrite`]-only strategy — it's context-free (`matches`/`replacements` need nothing beyond +/// the target itself) exactly like [`SharedSubDAGStrategy`], so it belongs +/// in this list rather than being derived per-workload the way +/// [`RollupStrategy`] is. Rewriting `avg` into `sum`/`count` upfront is what +/// lets [`ASAPStrategies`] and [`SharedSubDAGStrategy`] see a +/// mergeable accumulator to sketch or share at all — see that module's own +/// doc comment for why a bare `avg` node otherwise never becomes a +/// [`ReplacementStrategy`] target for anything. +pub fn default_strategies() -> Vec> { + vec![ + Box::new(ASAPStrategies::default()), + Box::new(HydraGroupingStrategy::default()), + Box::new(SharedSubDAGStrategy), + Box::new(crate::pass1::rewrite::AvgToSumOverCountStrategy), + Box::new(ExactCompositionStrategy), + ] } -/// Every [`Replacement::ExactComposition`] candidate of `group` whose -/// composed-plan rate is *known* and beats the raw-recompute baseline — -/// costed against each compatible child candidate already in `CandidateLogicalASAPDAGs` -/// (or the one an earlier parent committed). Unknown statistics yield no -/// option at all: the conservative `KeepPreAsap` path stays. -fn composition_options<'a>( - group: &'a TargetSubDAGCandidates, - groups: &'a HashMap<*const QueryExpr, TargetSubDAGCandidates>, - effective: usize, - cost_model: &dyn CostModel, - context: &CompositionContext, - plans: &[PreparedComposition], -) -> Vec> { - let mut options = Vec::new(); - for candidate in &group.candidates { - let Replacement::ExactComposition(composition) = &candidate.replacement else { - continue; - }; - if candidate.runtime_support_evidence(cost_model) != Some(true) { - continue; - } - let child_ptr = Rc::as_ptr(&composition.child_target); - let Some(child_group) = groups.get(&child_ptr) else { - continue; - }; - let already_committed = context.committed_child.get(&child_ptr).copied(); - let cost = |summary: &SummaryNode, shared: bool| { - let request = ExactCompositionCostRequest { - target: &group.target, - composition, - summary, - effective_consumer_count: effective, - }; - let mut inputs = cost_model.exact_composition_cost_inputs(&request); - if shared { - // Shared state is counted once: an earlier parent already - // pays this child's maintenance, so the marginal cost here - // is zero — a *known* zero, unlike an unknown input. - if let Some(maintenance) = inputs.summary_maintenance_cost_per_update.as_mut() { - *maintenance = 0.0; - } - } - let rate = inputs.composed_plan_cost_rate(composition.placement)?; - let baseline = raw_recompute_cost_rate(&inputs)?; - (rate < baseline).then_some((rate, baseline, inputs)) - }; - match composition.placement { - OperationPlacement::Read => { - let child_candidates: Vec<&'a ReplacementSubDAG> = match already_committed { - // SAFETY-free: the pointer was taken from `groups`'s own - // candidate storage, which outlives this borrow. - Some(ptr) => child_group - .candidates - .iter() - .filter(|c| std::ptr::eq(*c, ptr)) - .collect(), - None => child_group.candidates.iter().collect(), - }; - for child_candidate in child_candidates { - if !is_automatically_selectable(child_candidate, cost_model) { - continue; - } - let Replacement::Summary(summary) = &child_candidate.replacement else { - continue; - }; - if !composition.accepts_child(summary) { - continue; - } - let Some(prepared) = plans.iter().find(|p| { - p.target == Rc::as_ptr(&group.target) - && p.operation.same_as(composition) - && Rc::ptr_eq(&p.child, summary) - }) else { - continue; - }; - let Some((rate, baseline, inputs)) = cost(summary, already_committed.is_some()) - else { - continue; - }; - options.push(CompositionOption { - candidate, - decision: CompositionDecision { - plan: Rc::clone(&prepared.plan), - child_target: &composition.child_target, - child_candidate: Some(child_candidate), - cost_rate: rate, - baseline_rate: baseline, - inputs, - }, - }); - } - } - OperationPlacement::Maintenance => { - let Some(prepared) = plans.iter().find(|p| { - p.target == Rc::as_ptr(&group.target) && p.operation.same_as(composition) - }) else { - continue; - }; - // An maintenance-time operation only pays off beneath a - // maintained summary; with nothing above it, its output is - // never read and the raw fallback is the same computation. - let Some(parent) = context.maintaining_parent.get(&Rc::as_ptr(&group.target)) - else { - continue; - }; - let Some((rate, baseline, inputs)) = cost(parent, false) else { - continue; - }; - options.push(CompositionOption { - candidate, - decision: CompositionDecision { - plan: Rc::clone(&prepared.plan), - child_target: &composition.child_target, - child_candidate: None, - cost_rate: rate, - baseline_rate: baseline, - inputs, - }, - }); - } - } - } - options +/// Default context-free strategies with typed planning-time accuracy +/// evidence. This is the production counterpart of +/// constructing [`ASAPStrategies::new_with_planning_inputs_and_evidence`] and +/// [`HydraGroupingStrategy::new_with_planning_inputs_and_evidence`] separately. +pub fn default_strategies_with_evidence<'a>( + evidence: &'a dyn AccuracyEvidenceProvider, +) -> Vec> { + vec![ + Box::new(ASAPStrategies::new_with_planning_inputs_and_evidence( + &DEFAULT_ACCURACY_MODEL, + &DEFAULT_ALLOCATOR, + evidence, + )), + Box::new( + HydraGroupingStrategy::new_with_planning_inputs_and_evidence( + &DEFAULT_ACCURACY_MODEL, + &DEFAULT_ALLOCATOR, + evidence, + ), + ), + Box::new(SharedSubDAGStrategy), + Box::new(crate::pass1::rewrite::AvgToSumOverCountStrategy), + Box::new(ExactCompositionStrategy), + ] } -impl CandidateLogicalASAPDAGs { - /// The whole-plan (cross-group) selection step the module docs' - /// "Whole-plan (cross-group) selection" section describes: one - /// [`TargetSubDAGSelection`] per discovered site, each ranked against an - /// `effective_consumer_count` that accounts for every ancestor - /// [`SharedSubDAGStrategy`] decision on the path to it — unlike - /// [`Self::cost_sorted`], whose per-group ranking only ever sees a - /// group's own raw [`TargetSubDAGCandidates::consumer_count`]. - /// Uncertified DDSketch ratios remain in [`CandidateLogicalASAPDAGs`] for downstream - /// inspection but are not chosen automatically by this selector. - pub fn global_selection(&self, cost_model: &dyn CostModel) -> GlobalSelection<'_> { - self.global_selection_impl(cost_model, None, None, None) - .expect("structural global selection cannot produce a recurrence error") - } - - /// Recurrence-aware counterpart to [`Self::global_selection`]. The same - /// whole-plan traversal and effective structural consumer counts are - /// retained, while every CSE share/recompute choice is made from the - /// corresponding recurrence profile. - pub fn global_selection_with_recurrence( - &self, - cost_model: &dyn CostModel, - profiles: &RecurrenceProfileMap, - horizon: Option, - ) -> Result, RecurrenceError> { - self.global_selection_impl(cost_model, Some(profiles), horizon, None) - } +// ── search_workload ────────────────────────────────────────────────────── - pub(crate) fn global_selection_with_candidate_costs( - &self, - cost_model: &dyn CostModel, - profiles: &RecurrenceProfileMap, - horizon: Option, - costs: &CandidateCostOverrides, - ) -> Result, RecurrenceError> { - self.global_selection_impl(cost_model, Some(profiles), horizon, Some(costs)) - } +/// Search a whole workload's pre-ASAP roots for every candidate replacement +/// [`default_strategies`] can find, deduped into a [`CandidateLogicalASAPDAGs`]. +/// Candidate generation is cost-free; ranking happens in plan selection. Use +/// [`search_workload_with`] to plug in a custom strategy set. +pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalASAPDAGs { + search_workload_with(roots, &default_strategies()) +} - fn global_selection_impl( - &self, - cost_model: &dyn CostModel, - profiles: Option<&RecurrenceProfileMap>, - horizon: Option, - candidate_costs: Option<&CandidateCostOverrides>, - ) -> Result, RecurrenceError> { - let dag = reference_dag(self); - let topo = topological_order(&self.order, &dag); - - let mut effective_uses = dag.external_root_uses.clone(); - let mut chosen_share: HashMap<*const QueryExpr, ShareDecision> = HashMap::new(); - let mut groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'_>> = HashMap::new(); - let mut context = CompositionContext::default(); - - for ptr in &topo { - let group = &self.groups[ptr]; - - let effective = effective_uses.get(ptr).copied().unwrap_or(0); - effective_uses.insert(*ptr, effective); - - // ── Exact compositions (issue #171) ───────────────────────── - // A child an earlier parent's composition committed to is - // forced to exactly that candidate — the parent/child pair is - // one decision. Otherwise, a composition here wins only when - // its cost-units-per-second rate is *known* and beats the raw - // recompute baseline; missing statistics keep the conservative - // path below. - let mut composition_decision = None; - let forced = context - .committed_child - .get(ptr) - .and_then(|&cptr| group.candidates.iter().find(|c| std::ptr::eq(*c, cptr))); - let composed = if forced.is_some() { - None - } else { - composition_options( - group, - &self.groups, - effective, - cost_model, - &context, - &self.composition_plans, - ) - .into_iter() - .min_by(|a, b| a.decision.cost_rate.0.total_cmp(&b.decision.cost_rate.0)) - }; - if let Some(option) = &composed { - if let Some(child_candidate) = option.decision.child_candidate { - context.committed_child.insert( - Rc::as_ptr(option.decision.child_target), - child_candidate as *const ReplacementSubDAG, - ); - } - if let Replacement::ExactComposition(composition) = &option.candidate.replacement { - if composition.placement == OperationPlacement::Maintenance { - // A chain of functions feeds the same summary. - if let Some(parent) = context.maintaining_parent.get(ptr).cloned() { - context - .maintaining_parent - .insert(Rc::as_ptr(&composition.child_target), parent); - } - } - } - } +/// Like [`search_workload`], but with an explicit set of context-free +/// `strategies`. The workload-dependent +/// [`RollupStrategy`] is derived and added automatically after CSE for both +/// entry points, because only this function owns the post-CSE sibling set. +/// +/// Runs [`share_common_sub_dags`] once over `roots` first — so every +/// strategy (and, transitively, every +/// [`crate::pass1::explanation::ReplacementExplanation`] a caller reads off the +/// result) sees the same already-deduplicated DAG — then discovers every +/// `TargetSubDAG` (see [`discover_targets`]) and runs the +/// fixpoint loop the module docs describe, capped at +/// [`MAX_SEARCH_ITERATIONS`] passes (see the module docs' "Termination" +/// section). Deduping candidate plans this way needs no cost model. +pub fn search_workload_with<'s, Id>( + roots: Vec<(Id, Rc)>, + strategies: &[Box], +) -> CandidateLogicalASAPDAGs { + let mut space = search_cse_workload_with(cse_workload(roots), strategies); + space.prepare_compositions(&DefaultAccuracyModel, &HashMap::new()); + space +} - let lifecycle_choice = candidate_costs - .filter(|costs| costs.finalizes(&group.target)) - .map(|costs| { - let summary = group - .candidates - .iter() - .filter(|candidate| !is_composition_candidate(candidate)) - .filter(|candidate| is_automatically_selectable(candidate, cost_model)) - .filter_map(|candidate| { - costs - .get(&group.target, candidate) - .map(|cost| (candidate, cost)) - }) - .min_by(|(_, left), (_, right)| left.0.total_cmp(&right.0)); - match (summary, costs.raw(&group.target)) { - (Some((_, summary_cost)), Some(raw)) if raw.0 <= summary_cost.0 => None, - (Some((candidate, _)), _) => Some(candidate), - (None, _) => None, - } +/// [`search_workload_with`] plus a per-root end-to-end `AccuracyTarget` +/// (issue #172) — the workload's `QueryRequirements.accuracy`, threaded +/// alongside each root. After the search, every root that carries a target +/// has its group's bound-summary [`Replacement::SubDAG`] candidates checked with +/// `accuracy_model`'s [`AccuracyModel::satisfies`]: a candidate whose +/// guarantee is fully known and misses the target is moved from +/// [`TargetSubDAGCandidates::candidates`] to [`TargetSubDAGCandidates::rejected`] *before* +/// `candidate_selection::cost_sorted`/`candidate_selection::global_selection` ever rank the +/// group. A constructible candidate with unknown accuracy remains visible for +/// downstream review under an approximate target, but default whole-plan +/// selection does not commit it. An exact target cannot accept an unknown +/// approximate summary. A kept pre-ASAP candidate is +/// exact and always survives — the raw/pre-ASAP alternative is what an +/// unsatisfiable root keeps. Logical-rewrite [`Replacement::SubDAG`] +/// candidates are not bound values and are left alone; the targets *inside* +/// a rewrite are their own groups. +/// +/// Precedence against per-node `AggIntent.accuracy` is documented in +/// [`crate::accuracy`]'s module docs. +pub fn search_workload_with_targets<'s, Id>( + roots: Vec<(Id, Rc, Option)>, + strategies: &[Box], + accuracy_model: &dyn AccuracyModel, +) -> CandidateLogicalASAPDAGs { + let mut targets = Vec::with_capacity(roots.len()); + let roots = roots + .into_iter() + .map(|(id, root, target)| { + targets.push(target); + (id, root) + }) + .collect(); + let mut space = search_cse_workload_with(cse_workload(roots), strategies); + // `cse_workload` preserves root order, so targets zip by position. + let root_ptrs: Vec<(*const OperatorNode, AccuracyTarget)> = space + .roots + .iter() + .zip(targets) + .filter_map(|((_, root), target)| target.map(|t| (Rc::as_ptr(root), t))) + .collect(); + // Whole-root proposals join the root group before its target check. + for (index, (ptr, target)) in root_ptrs.iter().enumerate() { + if root_ptrs[..index].contains(&(*ptr, target.clone())) { + continue; + } + let group = space.groups.get_mut(ptr).expect("every root has a group"); + let root = Rc::clone(&group.target); + for strategy in strategies { + let name = strategy.name(); + let proposals = strategy.propose_for_root(&root, target); + for mut candidate in proposals.candidates { + candidate.strategy = name; + group.add_candidate(candidate); + } + group + .rejected + .extend(proposals.rejected.into_iter().map(|mut rejection| { + rejection.strategy = name; + rejection + })); + } + } + let mut composition_targets: HashMap<_, Vec<_>> = HashMap::new(); + for (ptr, target) in root_ptrs { + composition_targets + .entry(ptr) + .or_default() + .push(target.clone()); + let Some(group) = space.groups.get_mut(&ptr) else { + continue; + }; + let (legal, illegal): (Vec<_>, Vec<_>) = + group + .candidates + .drain(..) + .partition(|candidate| match &candidate.replacement { + Replacement::SubDAG(node) if is_logical_rewrite(node) => true, + Replacement::SubDAG(node) => node.guarantee.as_ref().map_or_else( + || !matches!(target, AccuracyTarget::Exact), + |g| accuracy_model.satisfies(&g.optimistic_floor(), &target), + ), + // A composition's guarantee depends on the concrete child; + // prepare_compositions checks those pairs after all roots. + Replacement::ExactComposition(_) => true, }); - - let complete_plan_choice = (!forced.is_some() - && composed.is_none() - && lifecycle_choice.is_none() - && cost_model.candidate_cost_covers_complete_plan()) - .then(|| { - let effective_target = TargetSubDAG::with_consumer_count(&group.target, effective); - let bound = group - .candidates - .iter() - .filter(|candidate| { - !is_cse_candidate(candidate) - && !is_composition_candidate(candidate) - && is_automatically_selectable(candidate, cost_model) - }) - .filter_map(|candidate| { - cost_model - .candidate_cost(candidate, &effective_target) - .map(|cost| (candidate, cost)) - }) - .min_by(|(_, left), (_, right)| left.0.total_cmp(&right.0)) - .map(|(candidate, _)| candidate); - bound.or_else(|| { - (cost_model.allow_uncosted_legacy_selection() && effective >= 2) - .then(|| { - decide_with_effective_count(group, effective, cost_model).and_then( - |decision| { - let candidate = pick_shared_sub_dag_candidate(group, decision)?; - chosen_share.insert(*ptr, decision); - Some(candidate) - }, - ) - }) - .flatten() - }) - }) - .flatten(); - - let chosen = if let Some(forced) = forced { - Some(forced) - } else if let Some(option) = composed { - composition_decision = Some(option.decision); - Some(option.candidate) - } else if let Some(choice) = lifecycle_choice { - choice - } else if cost_model.candidate_cost_covers_complete_plan() { - complete_plan_choice - } else if effective >= 2 && cse_candidate_pair(group).is_some() { - let decision = if let Some(profiles) = profiles { - decide_group_with_recurrence( - group, - effective, - profiles.for_target(&group.target), - horizon, - cost_model, - )? - } else { - decide_with_effective_count(group, effective, cost_model) - }; - match decision { - Some(decision) => { - let cse = pick_shared_sub_dag_candidate(group, decision); - let effective_target = - TargetSubDAG::with_consumer_count(&group.target, effective); - let logical = group - .candidates - .iter() - .filter(|candidate| { - !is_cse_candidate(candidate) - && !is_composition_candidate(candidate) - && is_automatically_selectable(candidate, cost_model) - }) - .filter_map(|candidate| { - cost_model - .candidate_cost(candidate, &effective_target) - .map(|cost| (candidate, cost)) - }) - .min_by(|(_, a), (_, b)| a.0.total_cmp(&b.0)) - .map(|(candidate, _)| candidate); - let cse = cse.filter(|candidate| { - cost_model - .candidate_cost(candidate, &effective_target) - .is_some() - || cost_model.allow_uncosted_legacy_selection() - }); - match (cse, logical) { - (Some(cse), Some(logical)) - if cost_model - .candidate_cost(cse, &effective_target) - .is_none_or(|cse_cost| { - cost_model - .candidate_cost(logical, &effective_target) - .is_some_and(|logical_cost| logical_cost.0 < cse_cost.0) - }) => - { - Some(logical) - } - (cse, _) => { - if cse.is_some() { - chosen_share.insert(*ptr, decision); - } - cse - } - } - } - // `realize_child` couldn't produce even a logical fallback — - // not expected in practice for a target that's already - // part of a legitimate workload DAG (mirrors - // `cse_preference`'s own doc on this same degrade). - // Falling back to ordinary local ranking is still a - // valid answer, just not a cross-group-aware one; this - // group also contributes no Share collapse to its own - // children (see `multiplier`'s `_ => effective` arm). - None => rank_group(group, cost_model).into_iter().find(|candidate| { - !is_composition_candidate(candidate) - && is_automatically_selectable(candidate, cost_model) - && (cost_model - .candidate_cost( - candidate, - &TargetSubDAG::with_consumer_count(&group.target, effective), - ) - .is_some() - || cost_model.allow_uncosted_legacy_selection()) - }), - } - } else { - let effective_target = TargetSubDAG::with_consumer_count(&group.target, effective); - rank_group(group, cost_model) - .into_iter() - .find(|candidate| { - !is_cse_candidate(candidate) - && !is_composition_candidate(candidate) - && is_automatically_selectable(candidate, cost_model) - && (cost_model - .candidate_cost(candidate, &effective_target) - .is_some() - || cost_model.allow_uncosted_legacy_selection()) - }) - .or_else(|| { - cse_candidate_pair(group) - .map(|(share, _)| share) - .filter(|candidate| { - cost_model - .candidate_cost(candidate, &effective_target) - .is_some() - || cost_model.allow_uncosted_legacy_selection() - }) + group.candidates = legal; + group.rejected.extend(illegal.into_iter().map(|candidate| { + let (metric, bound, failure_probability) = match &candidate.replacement { + Replacement::SubDAG(node) => node + .guarantee + .as_ref() + .map(|g| { + ( + g.metric, + g.bound.evaluate(), + g.failure_probability.evaluate(), + ) }) + .unwrap_or(( + asap_types::ir::properties::ErrorMetric::AbsoluteValue, + None, + None, + )), + Replacement::ExactComposition(_) => ( + asap_types::ir::properties::ErrorMetric::AbsoluteValue, + None, + None, + ), }; - - // Record the maintained summary this site's bound candidate - // builds, for a child that may compose an `ValueOperationAtIngestionTime` - // beneath it. - if let (Some(Replacement::Summary(node)), QueryExpr::Aggregate { child, .. }) = - (chosen.map(|c| &c.replacement), group.target.as_ref()) - { - if let Some(summary) = maintained_summary(node) { - context - .maintaining_parent - .insert(Rc::as_ptr(child), Rc::clone(summary)); - } - } - - let outgoing_multiplier = multiplier(*ptr, &effective_uses, &chosen_share); - match chosen { - Some(ReplacementSubDAG { - replacement: Replacement::Rewrite(source), - provenance: ReplacementProvenance::AccuracyReconciliation, - .. - }) => { - // Accuracy reconciliation reads another discovered memo - // group, rather than inlining that group's children. Let - // the source group receive the uses and propagate them - // through its own selected realization when its turn - // arrives in topological order. - *effective_uses.entry(Rc::as_ptr(source)).or_insert(0) += outgoing_multiplier; - } - _ => { - let selected_rewrite = match chosen.map(|candidate| &candidate.replacement) { - Some(Replacement::Rewrite(rewrite)) => rewrite, - Some(Replacement::Summary(_) | Replacement::ExactComposition(_)) | None => { - &group.target - } - }; - for (child, edge_count) in direct_child_counts(selected_rewrite) { - *effective_uses.entry(child).or_insert(0) += - edge_count * outgoing_multiplier; - } - } - } - - groups.insert( - *ptr, - TargetSubDAGSelection { - target: &group.target, - consumer_count: group.consumer_count, - effective_consumer_count: effective, - chosen, - composition: composition_decision, + RejectedCandidate { + strategy: candidate.strategy, + description: format!("{} (root end-to-end target check)", candidate.rationale), + error: AccuracyError::TargetNotSatisfied { + metric, + bound, + failure_probability, + target: target.clone(), }, - ); - } - - Ok(GlobalSelection { - order: self.order.clone(), - groups, - assembled_nodes: RefCell::new(HashMap::new()), - }) + } + })); } + space.prepare_compositions(accuracy_model, &composition_targets); + space } -fn is_cse_candidate(candidate: &ReplacementSubDAG) -> bool { - matches!( - candidate.provenance, - ReplacementProvenance::CseShare | ReplacementProvenance::CseRecompute - ) -} - -fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn CostModel) -> bool { - candidate.provenance != ReplacementProvenance::RootPhysicalRealization - && !candidate.has_missing_accuracy_evidence() - && candidate.runtime_support_evidence(cost_model) != Some(false) -} - -/// How much one direct reference to `parent_ptr` actually costs, once -/// `parent_ptr`'s own chosen candidate (if it has a Share/Recompute pair at -/// all) is taken into account: -/// -/// - `1`, if `parent_ptr` chose [`ShareDecision::Share`] — one shared -/// execution backs every reference to it, so referencing it costs no more -/// than referencing it once. -/// - `parent_ptr`'s own `effective_consumer_count` otherwise — either it -/// chose [`ShareDecision::RecomputeIndependently`] (each of its own uses -/// gets its own independent execution, so referencing it costs as much as -/// its *own* full multiplicity), or it has no Share/Recompute decision at -/// all (not a [`SharedSubDAGStrategy`] shape — nothing here collapses -/// its multiplicity to one, so whatever multiplicity *its* ancestors -/// established simply passes through). -/// -/// Composing this recurrence transitively up the whole ancestor chain (not -/// just the immediate parent) is exactly what makes -/// [`CandidateLogicalASAPDAGs::global_selection`]'s `effective_consumer_count` differ from -/// [`TargetSubDAGCandidates::consumer_count`] whenever a `RecomputeIndependently` -/// ancestor sits anywhere on the path from a root to a site — see the -/// module docs' "Whole-plan (cross-group) selection" section. -fn multiplier( - parent_ptr: *const QueryExpr, - effective_uses: &HashMap<*const QueryExpr, usize>, - chosen_share: &HashMap<*const QueryExpr, ShareDecision>, -) -> usize { - let effective = *effective_uses.get(&parent_ptr).expect( - "topological_order guarantees a parent is processed (and its effective_consumer_count \ - recorded) before any of its children", - ); - match chosen_share.get(&parent_ptr) { - Some(ShareDecision::Share) => 1, - _ => effective, +/// The strictest accuracy among `siblings` that read the same summary input +/// as `root` — same child, grouping and filters, and the same intent apart +/// from its accuracy (and a quantile's rank, a evaluation parameter) — when +/// stricter than `root`'s own. One summary sized for the strictest consumer +/// serves every sibling: #509's summary-capability rule. +fn strictest_sibling_accuracy( + root: &OperatorNode, + siblings: &[Rc], +) -> Option { + fn approximate(intent: &AggIntent) -> Option<&AccuracyTarget> { + accuracy_target(intent).filter(|accuracy| !matches!(accuracy, AccuracyTarget::Exact)) } -} - -/// Find the explicitly-tagged CSE share/recompute pair inside `group`, even -/// when other strategies contributed additional alternatives to the same -/// memo group. Provenance makes these two orthogonal choices identifiable -/// without inferring semantics from pointer or expression shape. -fn cse_candidate_pair( - group: &TargetSubDAGCandidates, -) -> Option<(&ReplacementSubDAG, &ReplacementSubDAG)> { - let mut share = None; - let mut recompute = None; - for candidate in &group.candidates { - match candidate.provenance { - ReplacementProvenance::CseShare => { - let Replacement::Rewrite(rc) = &candidate.replacement else { - return None; - }; - if !Rc::ptr_eq(rc, &group.target) || share.replace(candidate).is_some() { - return None; - } - } - ReplacementProvenance::CseRecompute => { - let Replacement::Rewrite(rc) = &candidate.replacement else { - return None; - }; - if Rc::ptr_eq(rc, &group.target) - || rc.as_ref() != group.target.as_ref() - || recompute.replace(candidate).is_some() - { - return None; - } - } - _ => {} - } - } - Some((share?, recompute?)) -} - -/// [`CostModel::cse_share_decision`] for `group`, against an explicit -/// `effective_consumer_count` instead of `group.consumer_count` — the -/// cross-group-aware counterpart to [`cse_preference`], which uses the raw -/// structural count. `None` only when [`realize_child`] can't produce even a -/// logical fallback for `group.target` (see that function's own doc). -fn decide_with_effective_count( - group: &TargetSubDAGCandidates, - effective_consumer_count: usize, - cost_model: &dyn CostModel, -) -> Option { - let bound = realize_child(&group.target, cost_model).ok()?; - let candidate = CseCandidate { - sub_dag: &group.target, - bound_summary: &bound, - consumer_count: effective_consumer_count, - }; - Some(cost_model.cse_share_decision(&candidate)) -} - -fn decide_group_with_recurrence( - group: &TargetSubDAGCandidates, - effective_consumer_count: usize, - recurrence: RecurrenceProfile, - horizon: Option, - cost_model: &dyn CostModel, -) -> Result, RecurrenceError> { - let Some(bound) = realize_child(&group.target, cost_model).ok() else { - return Ok(None); - }; - let candidate = CseCandidate { - sub_dag: &group.target, - bound_summary: &bound, - consumer_count: effective_consumer_count, - }; - Ok(Some( - cost_model - .cse_share_decision_with_recurrence(&candidate, &recurrence, horizon)? - .decision, - )) -} - -/// The [`SharedSubDAGStrategy`] candidate matching `decision`: the one -/// that shares `group.target`'s own `Rc` for [`ShareDecision::Share`], the -/// freshly-allocated one for [`ShareDecision::RecomputeIndependently`] — -/// the same `Rc`-identity distinction [`is_duplicate_rewrite`]'s own doc -/// explains is the *only* signal this IR carries for that choice. -fn pick_shared_sub_dag_candidate( - group: &TargetSubDAGCandidates, - decision: ShareDecision, -) -> Option<&ReplacementSubDAG> { - let (share, recompute) = cse_candidate_pair(group)?; - Some(match decision { - ShareDecision::Share => share, - ShareDecision::RecomputeIndependently => recompute, - }) -} - -// ── reference DAG + topological order ───────────────────────────────── - -/// The parent/child structure [`CandidateLogicalASAPDAGs::global_selection`]'s DP walks — -/// built separately from [`discover_targets`]'s own `order`/`nodes`/`counts` -/// maps (which only track *aggregate* reference counts, not per-parent -/// breakdown or direction) rather than extending that already-reviewed, -/// already-tested pass. Same "small duplicated traversal over reshaping -/// proven code" call as [`is_shared_subtree_group`]. -struct ReferenceDAG { - /// child ptr -> `(parent ptr, edge count from that one parent)`, for - /// every direct operator-child edge in the relational-skeleton scope - /// [`walk_children`] itself uses (an edge count above 1 happens when - /// one parent references the same child from two different fields, - /// e.g. a `Join`'s `left`/`right` both being the same `Rc`). - parents_of: HashMap<*const QueryExpr, Vec<(*const QueryExpr, usize)>>, - /// parent ptr -> every distinct child ptr it directly references — the - /// reverse of `parents_of`, for [`topological_order`]'s Kahn's-algorithm - /// traversal. - children_of: HashMap<*const QueryExpr, Vec<*const QueryExpr>>, - /// How many of the workload's own `roots` point directly at each node — - /// a node's "external" use. Nothing inside the DAG decides this (it - /// isn't a reference from another discovered site), so it's never - /// subject to any ancestor's Share/Recompute choice — it's the base - /// case [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence starts from. - external_root_uses: HashMap<*const QueryExpr, usize>, -} - -/// Build an ordering DAG containing every edge that could be selected: -/// the original target's edges plus every rewrite candidate's edges. An -/// accuracy-reconciliation rewrite points at another discovered memo group, -/// so it contributes an edge to that group itself; other rewrites contribute -/// their relational children as before. The -/// DAG is deliberately only used for topological ordering; effective-use -/// counts are propagated through the one candidate actually selected. -fn reference_dag(space: &CandidateLogicalASAPDAGs) -> ReferenceDAG { - let mut dag = ReferenceDAG { - parents_of: HashMap::new(), - children_of: HashMap::new(), - external_root_uses: HashMap::new(), - }; - for (_, root) in &space.roots { - *dag.external_root_uses.entry(Rc::as_ptr(root)).or_insert(0) += 1; - } - for ptr in &space.order { - let group = &space.groups[ptr]; - record_possible_edges(*ptr, &group.target, &mut dag); - for candidate in &group.candidates { - if let Replacement::Rewrite(rewrite) = &candidate.replacement { - if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { - add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut dag); - } else { - record_possible_edges(*ptr, rewrite, &mut dag); - } - } - } - } - dag -} - -/// Record one `parent_ptr -> child` edge (both directions — see -/// [`ReferenceDAG`]'s fields), retaining the greatest multiplicity seen -/// when the target and alternative rewrites expose the same edge. -fn add_edge( - parent_ptr: *const QueryExpr, - child_ptr: *const QueryExpr, - edge_count: usize, - dag: &mut ReferenceDAG, -) { - let siblings = dag.parents_of.entry(child_ptr).or_default(); - match siblings.iter_mut().find(|(p, _)| *p == parent_ptr) { - Some((_, count)) => *count = (*count).max(edge_count), - None => siblings.push((parent_ptr, edge_count)), - } - let kids = dag.children_of.entry(parent_ptr).or_default(); - if !kids.contains(&child_ptr) { - kids.push(child_ptr); - } -} - -fn record_possible_edges(parent_ptr: *const QueryExpr, node: &QueryExpr, dag: &mut ReferenceDAG) { - for (child_ptr, edge_count) in direct_child_counts(node) { - add_edge(parent_ptr, child_ptr, edge_count, dag); - } -} - -/// Direct relational-skeleton children and their edge multiplicities. -/// `Concat` is transparent, matching [`walk_children`]'s site scope. -fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { - fn push(children: &mut Vec<(*const QueryExpr, usize)>, child: &Rc) { - let ptr = Rc::as_ptr(child); - match children.iter_mut().find(|(existing, _)| *existing == ptr) { - Some((_, count)) => *count += 1, - None => children.push((ptr, 1)), - } - } - - fn collect(node: &QueryExpr, children: &mut Vec<(*const QueryExpr, usize)>) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => { - push(children, c); - } - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => { - push(children, child); - } - Concat { - children: concat_children, - .. - } => { - for c in concat_children { - collect(c, children); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - push(children, left); - push(children, right); - } - BinaryOp { lhs, rhs, .. } => { - push(children, lhs); - push(children, rhs); - } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} - } - } - - let mut children = Vec::new(); - collect(node, &mut children); - children -} - -/// A topological order over `order` (parent before every child) via Kahn's -/// algorithm on `dag`'s reverse adjacency — needed because -/// [`discover_targets`]'s own `order` is only a valid *discovery* order -/// (first-seen-first), not a valid topological one: a node reached via two -/// different root paths can have a parent that's discovered *after* it (see -/// this function's own test for a worked diamond example), which is exactly -/// backwards for [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence. -fn topological_order(order: &[*const QueryExpr], dag: &ReferenceDAG) -> Vec<*const QueryExpr> { - let mut in_degree: HashMap<*const QueryExpr, usize> = HashMap::new(); - for ptr in order { - let degree = dag.parents_of.get(ptr).map(Vec::len).unwrap_or(0); - in_degree.insert(*ptr, degree); - } - - let mut queue: VecDeque<*const QueryExpr> = order - .iter() - .copied() - .filter(|ptr| in_degree[ptr] == 0) - .collect(); - - let mut topo = Vec::with_capacity(order.len()); - while let Some(ptr) = queue.pop_front() { - topo.push(ptr); - if let Some(children) = dag.children_of.get(&ptr) { - for child in children { - if let Some(degree) = in_degree.get_mut(child) { - *degree -= 1; - if *degree == 0 { - queue.push_back(*child); - } - } - } - } - } - - assert_eq!( - topo.len(), - order.len(), - "topological_order: the discovered-site reference DAG has a cycle — every QueryExpr \ - node is built from Rc children, which can't form one, so this indicates a bug in \ - reference_dag rather than a real cyclic workload", - ); - topo -} - -// ── default_strategies ────────────────────────────────────────────────── - -/// The context-free strategies [`search_workload`] runs in the built-in -/// [`DefaultCostModel`] configuration. Workload-dependent strategies such as -/// [`RollupStrategy`] and [`AccuracyReconciliationStrategy`] (issue #273, -/// cross-consumer accuracy reconciliation for CSE sharing — see that -/// module's own docs) are added by [`search_workload`] after CSE and target -/// discovery, when their sibling context exists. -/// [`crate::explanation::explain_replacements`] (issue #257) uses -/// this same set (via [`search_workload`]) rather than keeping a second, -/// explanation-specific list to stay in sync with. Use -/// [`default_strategies_with`] to plug in a deployment-specific -/// [`CostModel`] instead. -/// -/// [`AvgToSumOverCountStrategy`](crate::rewrite::AvgToSumOverCountStrategy) is -/// included here (issue #253) even though it's a -/// [`Replacement::Rewrite`]-only strategy with no [`CostModel`] of its own to -/// plug in — it's context-free (`matches`/`replacements` need nothing beyond -/// the target itself) exactly like [`SharedSubDAGStrategy`], so it belongs -/// in this list rather than being derived per-workload the way -/// [`RollupStrategy`] is. Rewriting `avg` into `sum`/`count` upfront is what -/// lets [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] see a -/// mergeable accumulator to sketch or share at all — see that module's own -/// doc comment for why a bare `avg` node otherwise never becomes a -/// [`ReplacementStrategy`] target for anything. -pub fn default_strategies() -> Vec> { - vec![ - Box::new(SketchAlgorithmStrategy::default_cost_model()), - Box::new(HydraGroupingStrategy::default_cost_model()), - Box::new(SharedSubDAGStrategy), - Box::new(crate::rewrite::AvgToSumOverCountStrategy), - Box::new(ExactCompositionStrategy::default_cost_model()), - ] -} - -/// Like [`default_strategies`], but [`SketchAlgorithmStrategy`] ranks/binds via -/// `cost_model` instead of the built-in [`DefaultCostModel`] — the same -/// customization point [`SketchAlgorithmStrategy::new`] itself offers. -pub fn default_strategies_with<'a>( - cost_model: &'a dyn CostModel, -) -> Vec> { - vec![ - Box::new(SketchAlgorithmStrategy::new(cost_model)), - Box::new(HydraGroupingStrategy::new(cost_model)), - Box::new(SharedSubDAGStrategy), - Box::new(crate::rewrite::SemanticEquivalentRewriteStrategy), - Box::new(ExactCompositionStrategy::new(cost_model)), - ] -} - -/// Default context-free strategies with both deployment costing and typed -/// planning-time accuracy evidence. This is the production counterpart of -/// constructing [`SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence`] and -/// [`HydraGroupingStrategy::new_with_planning_inputs_and_evidence`] separately. -pub fn default_strategies_with_evidence<'a>( - cost_model: &'a dyn CostModel, - evidence: &'a dyn AccuracyEvidenceProvider, -) -> Vec> { - vec![ - Box::new( - SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - cost_model, - &DEFAULT_ACCURACY_MODEL, - &DEFAULT_ALLOCATOR, - evidence, - ), - ), - Box::new( - HydraGroupingStrategy::new_with_planning_inputs_and_evidence( - cost_model, - &DEFAULT_ACCURACY_MODEL, - &DEFAULT_ALLOCATOR, - evidence, - ), - ), - Box::new(SharedSubDAGStrategy), - Box::new(crate::rewrite::AvgToSumOverCountStrategy), - Box::new(ExactCompositionStrategy::new(cost_model)), - ] -} - -// ── search_workload ────────────────────────────────────────────────────── - -/// Search a whole workload's pre-ASAP roots for every candidate replacement -/// [`default_strategies`] can find, deduped into a [`CandidateLogicalASAPDAGs`]. Candidate -/// *generation* uses the built-in [`DefaultCostModel`] (via -/// [`default_strategies`], the same way [`SketchAlgorithmStrategy::default_cost_model`] -/// does); call [`CandidateLogicalASAPDAGs::cost_sorted`] on the result for the final -/// `sorted_by(cost_model)` step. Use [`search_workload_with`] to plug in a -/// custom strategy set (e.g. built via [`default_strategies_with`] for a -/// deployment-specific [`CostModel`]). -pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalASAPDAGs { - search_workload_with(roots, &default_strategies()) -} - -/// Like [`search_workload`], but with an explicit set of context-free -/// `strategies` (see [`default_strategies_with`] to plug in a -/// deployment-specific [`CostModel`]). The workload-dependent -/// [`RollupStrategy`] is derived and added automatically after CSE for both -/// entry points, because only this function owns the post-CSE sibling set. -/// -/// Runs [`share_common_sub_dags`] once over `roots` first — so every -/// strategy (and, transitively, every -/// [`crate::explanation::ReplacementExplanation`] a caller reads off the -/// result) sees the same already-deduplicated DAG — then discovers every -/// `TargetSubDAG` (see [`discover_targets`]) and runs the -/// fixpoint loop the module docs describe, capped at -/// [`MAX_SEARCH_ITERATIONS`] passes (see the module docs' "Termination" -/// section). Deduping candidate plans this way needs no -/// [`CostModel`] at all — that only enters at two well-defined points: each -/// [`ReplacementStrategy`] in `strategies` may already carry its own (e.g. -/// [`SketchAlgorithmStrategy::new`]'s), and [`CandidateLogicalASAPDAGs::cost_sorted`]'s final -/// ranking step takes one explicitly. -pub fn search_workload_with<'s, Id>( - roots: Vec<(Id, Rc)>, - strategies: &[Box], -) -> CandidateLogicalASAPDAGs { - let mut space = search_cse_workload_with(cse_workload(roots), strategies); - space.prepare_compositions(&DefaultAccuracyModel, &HashMap::new()); - space -} - -/// [`search_workload_with`] plus a per-root end-to-end `AccuracyTarget` -/// (issue #172) — the workload's `QueryRequirements.accuracy`, threaded -/// alongside each root. After the search, every root that carries a target -/// has its group's bound [`Replacement::Summary`] candidates checked with -/// `accuracy_model`'s [`AccuracyModel::satisfies`]: a candidate whose -/// guarantee is fully known and misses the target is moved from -/// [`TargetSubDAGCandidates::candidates`] to [`TargetSubDAGCandidates::rejected`] *before* -/// [`CandidateLogicalASAPDAGs::cost_sorted`]/[`CandidateLogicalASAPDAGs::global_selection`] ever rank the -/// group. A constructible candidate with unknown accuracy remains visible for -/// downstream review under an approximate target, but default whole-plan -/// selection does not commit it. An exact target cannot accept an unknown -/// approximate summary. A `KeepPreAsap` candidate is -/// exact and always survives — the raw/pre-ASAP alternative is what an -/// unsatisfiable root keeps. Logical [`Replacement::Rewrite`] candidates -/// are not bound values and are left alone; the targets *inside* a rewrite -/// are their own groups. -/// -/// Precedence against per-node `AggIntent.accuracy` is documented in -/// [`crate::accuracy`]'s module docs. -pub fn search_workload_with_targets<'s, Id>( - roots: Vec<(Id, Rc, Option)>, - strategies: &[Box], - accuracy_model: &dyn AccuracyModel, -) -> CandidateLogicalASAPDAGs { - let mut targets = Vec::with_capacity(roots.len()); - let roots = roots - .into_iter() - .map(|(id, root, target)| { - targets.push(target); - (id, root) - }) - .collect(); - let mut space = search_cse_workload_with(cse_workload(roots), strategies); - // `cse_workload` preserves root order, so targets zip by position. - let root_ptrs: Vec<(*const QueryExpr, AccuracyTarget)> = space - .roots - .iter() - .zip(targets) - .filter_map(|((_, root), target)| target.map(|t| (Rc::as_ptr(root), t))) - .collect(); - // Whole-root proposals join the root group before its target check. - for (index, (ptr, target)) in root_ptrs.iter().enumerate() { - if root_ptrs[..index].contains(&(*ptr, target.clone())) { - continue; - } - let group = space.groups.get_mut(ptr).expect("every root has a group"); - let root = Rc::clone(&group.target); - for strategy in strategies { - let name = strategy.name(); - let proposals = strategy.propose_for_root(&root, target); - for mut candidate in proposals.candidates { - candidate.strategy = name; - group.add_candidate(candidate); - } - group - .rejected - .extend(proposals.rejected.into_iter().map(|mut rejection| { - rejection.strategy = name; - rejection - })); - } - } - let mut composition_targets: HashMap<_, Vec<_>> = HashMap::new(); - for (ptr, target) in root_ptrs { - composition_targets - .entry(ptr) - .or_default() - .push(target.clone()); - let Some(group) = space.groups.get_mut(&ptr) else { - continue; - }; - let (legal, illegal): (Vec<_>, Vec<_>) = - group - .candidates - .drain(..) - .partition(|candidate| match &candidate.replacement { - Replacement::Summary(node) => node.guarantee.as_ref().map_or_else( - || !matches!(target, AccuracyTarget::Exact), - |g| accuracy_model.satisfies(&g.optimistic_floor(), &target), - ), - Replacement::Rewrite(_) => true, - // A composition's guarantee depends on the concrete child; - // prepare_compositions checks those pairs after all roots. - Replacement::ExactComposition(_) => true, - }); - group.candidates = legal; - group.rejected.extend(illegal.into_iter().map(|candidate| { - let (metric, bound, failure_probability) = match &candidate.replacement { - Replacement::Summary(node) => node - .guarantee - .as_ref() - .map(|g| { - ( - g.metric, - g.bound.evaluate(), - g.failure_probability.evaluate(), - ) - }) - .unwrap_or(( - asap_types::post_asap::ErrorMetric::AbsoluteValue, - None, - None, - )), - Replacement::Rewrite(_) => unreachable!("rewrites are never rejected here"), - Replacement::ExactComposition(_) => ( - asap_types::post_asap::ErrorMetric::AbsoluteValue, - None, - None, - ), - }; - RejectedCandidate { - strategy: candidate.strategy, - description: format!("{} (root end-to-end target check)", candidate.rationale), - error: AccuracyError::TargetNotSatisfied { - metric, - bound, - failure_probability, - target: target.clone(), - }, - } - })); - } - space.prepare_compositions(accuracy_model, &composition_targets); - space -} - -/// The strictest accuracy among `siblings` that read the same summary input -/// as `root` — same child, grouping and filters, and the same intent apart -/// from its accuracy (and a quantile's rank, a readout parameter) — when -/// stricter than `root`'s own. One summary sized for the strictest consumer -/// serves every sibling: #509's summary-capability rule. -fn strictest_sibling_accuracy( - root: &QueryExpr, - siblings: &[Rc], -) -> Option { - fn approximate(intent: &AggIntent) -> Option<&AccuracyTarget> { - accuracy_target(intent).filter(|accuracy| !matches!(accuracy, AccuracyTarget::Exact)) - } - let QueryExpr::Aggregate { - reduction, - filters, - child, - .. - } = root - else { - return None; + let Some(NonASAPOp::Aggregate { + reduction, + filters, + child, + .. + }) = root.non_asap() + else { + return None; }; let intent = bindable_intent(root)?; let own = accuracy_budget(approximate(intent)?); - let (mut eps, mut delta) = own; - for sibling in siblings { - let QueryExpr::Aggregate { - reduction: sibling_reduction, - filters: sibling_filters, - child: sibling_child, - .. - } = sibling.as_ref() - else { - continue; - }; - let Some(other) = bindable_intent(sibling) else { - continue; - }; - let Some(accuracy) = approximate(other) else { - continue; - }; - let same_intent = match (intent, other) { - (AggIntent::Quantile { col, .. }, AggIntent::Quantile { col: other_col, .. }) => { - col == other_col - } - _ => override_accuracy(intent, accuracy) == *other, - }; - if same_intent - && sibling_reduction == reduction - && sibling_filters == filters - && (Rc::ptr_eq(sibling_child, child) || sibling_child == child) - { - let (sibling_eps, sibling_delta) = accuracy_budget(accuracy); - eps = eps.min(sibling_eps); - delta = delta.min(sibling_delta); - } - } - if (eps, delta) == own { - None - } else if delta == DEFAULT_DELTA { - Some(AccuracyTarget::Epsilon(eps)) - } else { - Some(AccuracyTarget::EpsilonDelta { - epsilon: eps, - delta, - }) - } -} - -fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { - // `share_common_sub_dags` wants owned `QueryExpr`s, not already-`Rc` - // roots — the same `Rc::try_unwrap`-with-clone-fallback pattern - // `asap_types::pre_asap::cse::intern_child` itself uses to recover an - // owned node without cloning in the common (uniquely-owned) case. - let owned_roots: Vec<(Id, QueryExpr)> = roots - .into_iter() - .map(|(id, rc)| { - let expr = Rc::try_unwrap(rc).unwrap_or_else(|shared| (*shared).clone()); - (id, expr) - }) - .collect(); - share_common_sub_dags(owned_roots) -} - -fn search_cse_workload_with<'s, Id>( - cse_roots: Vec<(Id, Rc)>, - strategies: &[Box], -) -> CandidateLogicalASAPDAGs { - let mut order = Vec::new(); - let mut nodes = HashMap::new(); - let mut counts: HashMap<*const QueryExpr, usize> = HashMap::new(); - discover_targets(&cse_roots, &mut order, &mut nodes, &mut counts); - let siblings: Vec> = order - .iter() - .filter_map(|ptr| { - let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Aggregate { .. }).then(|| Rc::clone(node)) - }) - .collect(); - let rollup_strategy = RollupStrategy::new(&siblings); - let accuracy_reconciliation_strategy = AccuracyReconciliationStrategy::new(&siblings); - let limits: Vec> = order - .iter() - .filter_map(|ptr| { - let node = &nodes[ptr]; - matches!(node.as_ref(), QueryExpr::Limit { .. }).then(|| Rc::clone(node)) - }) - .collect(); - let topk_reuse_strategy = TopKLimitReuseStrategy::new(&limits); - - let mut groups: HashMap<*const QueryExpr, TargetSubDAGCandidates> = HashMap::new(); - for ptr in &order { - groups.insert( - *ptr, - TargetSubDAGCandidates::new(Rc::clone(&nodes[ptr]), counts[ptr]), - ); - } - - // Round-based frontier: every target is asked exactly once per strategy - // (never re-asked — see the module docs' "Termination" section on why - // that matters for `Replacement::Summary` dedup specifically). A round - // can grow the *next* round's frontier only by a candidate's own - // reachable children exposing a genuinely new, not-yet-known `Rc` — see - // `discover_new_descendant_targets`. - let mut frontier = order.clone(); - let mut rounds = 0usize; - while !frontier.is_empty() { - rounds += 1; - assert!( - rounds <= MAX_SEARCH_ITERATIONS, - "search_workload: fixpoint search did not converge within {MAX_SEARCH_ITERATIONS} \ - rounds — a registered ReplacementStrategy's Replacement::Rewrite candidates keep \ - exposing new, never-before-seen descendant structure every round. \ - SketchAlgorithmStrategy/SharedSubDAGStrategy never do this (see replacement.rs's \ - module docs' \"Termination\" section); check any custom strategies passed to \ - search_workload_with.", - ); - - let targets_before = order.len(); - for ptr in &frontier { - let (root, consumer_count) = { - let group = &groups[ptr]; - (Rc::clone(&group.target), group.consumer_count) - }; - let strictest = strictest_sibling_accuracy(&root, &siblings); - let mut target = TargetSubDAG::with_consumer_count(&root, consumer_count); - target.strictest_sibling_accuracy = strictest.as_ref(); - - let mut proposed = Vec::new(); - let mut rejected = Vec::new(); - for strategy in strategies { - if strategy.matches(&target) { - let name = strategy.name(); - let proposals = strategy.propose(&target); - proposed.extend(proposals.candidates.into_iter().map(|mut candidate| { - candidate.strategy = name; - candidate - })); - rejected.extend(proposals.rejected.into_iter().map(|mut rejection| { - rejection.strategy = name; - rejection - })); - } - } - if rollup_strategy.matches(&target) { - let name = rollup_strategy.name(); - proposed.extend(rollup_strategy.replacements(&target).into_iter().map( - |mut candidate| { - candidate.strategy = name; - candidate - }, - )); - } - if accuracy_reconciliation_strategy.matches(&target) { - let name = accuracy_reconciliation_strategy.name(); - proposed.extend( - accuracy_reconciliation_strategy - .replacements(&target) - .into_iter() - .map(|mut candidate| { - candidate.strategy = name; - candidate - }), - ); - } - if topk_reuse_strategy.matches(&target) { - let name = topk_reuse_strategy.name(); - proposed.extend(topk_reuse_strategy.replacements(&target).into_iter().map( - |mut candidate| { - candidate.strategy = name; - candidate - }, - )); - } - - for candidate in &proposed { - if let Replacement::Rewrite(rc) = &candidate.replacement { - discover_new_descendant_targets(rc, &mut order, &mut nodes, &mut counts); - } - } - - let group = groups - .get_mut(ptr) - .expect("every discovered target has a group"); - for candidate in proposed { - group.add_candidate(candidate); - } - group.rejected.extend(rejected); - } - - // Any pointer `discover_new_descendant_targets` appended to `order` - // this round is a genuinely new target — give it a group and process - // it next round. Targets already in `groups` are never revisited. - let new_targets = &order[targets_before..]; - for ptr in new_targets { - groups.entry(*ptr).or_insert_with(|| { - TargetSubDAGCandidates::new(Rc::clone(&nodes[ptr]), counts[ptr]) - }); - } - frontier = new_targets.to_vec(); - } - - add_effective_count_cse_candidates(&order, &mut groups); - - CandidateLogicalASAPDAGs { - roots: cse_roots, - groups, - order, - composition_plans: Vec::new(), - } -} - -/// Materialize share/recompute alternatives for descendants whose raw edge -/// count is one but whose effective count can exceed one when a repeated -/// ancestor is recomputed. We only do this when an ordinary repeated group -/// proves that `SharedSubDAGStrategy` is part of this search's strategy set. -fn add_effective_count_cse_candidates( - order: &[*const QueryExpr], - groups: &mut HashMap<*const QueryExpr, TargetSubDAGCandidates>, -) { - let mut possible_children: HashMap<*const QueryExpr, Vec<*const QueryExpr>> = HashMap::new(); - for ptr in order { - let group = &groups[ptr]; - let children = possible_children.entry(*ptr).or_default(); - for (child, _) in direct_child_counts(&group.target) { - if !children.contains(&child) { - children.push(child); - } - } - for candidate in &group.candidates { - if let Replacement::Rewrite(rewrite) = &candidate.replacement { - for (child, _) in direct_child_counts(rewrite) { - if !children.contains(&child) { - children.push(child); - } - } - } - } - } - - let mut potentially_repeated = HashSet::new(); - let mut queue = VecDeque::new(); - for ptr in order { - let group = &groups[ptr]; - if group.consumer_count >= 2 && cse_candidate_pair(group).is_some() { - potentially_repeated.insert(*ptr); - queue.push_back(*ptr); - } - } - while let Some(parent) = queue.pop_front() { - if let Some(children) = possible_children.get(&parent) { - for child in children { - if groups.contains_key(child) && potentially_repeated.insert(*child) { - queue.push_back(*child); - } - } - } - } - - for ptr in order { - let group = groups - .get_mut(ptr) - .expect("every discovered site has a group"); - if potentially_repeated.contains(ptr) && cse_candidate_pair(group).is_none() { - let target = Rc::clone(&group.target); - let site = TargetSubDAG::with_consumer_count(&target, 2); - for mut candidate in SharedSubDAGStrategy.replacements(&site) { - candidate.rationale = format!( - "{}: this sub-DAG can become repeated when a repeated ancestor is recomputed; \ - global_selection decides using its effective consumer count", - match candidate.provenance { - ReplacementProvenance::CseShare => "build once and share", - ReplacementProvenance::CseRecompute => "recompute independently", - _ => unreachable!("SharedSubDAGStrategy only emits CSE candidates"), - } - ); - group.add_candidate(candidate); - } - } - } -} - -// ── target discovery ───────────────────────────────────────────────────── - -/// Walk every root's whole DAG, discovering one `TargetSubDAG` per distinct -/// `Rc` and its real `consumer_count` — see the module docs' "Where -/// `TargetSubDAG` discovery comes from" section for the full rationale. -fn discover_targets( - roots: &[(Id, Rc)], - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, -) { - for (_, root) in roots { - walk(root, order, nodes, counts); - } -} - -/// Scan `candidate`'s **children** (deliberately never `candidate`'s own -/// top-level pointer — see the module docs' "Termination" section: a -/// [`Replacement::Rewrite`]'s value is an alternative *for* the target that -/// proposed it, never a new target of its own) for any `Rc` not already -/// known, appending each to `order`/`nodes`/`counts` so -/// [`search_workload_with`]'s next round processes it. A no-op when every -/// child is already known — the case both shipped strategies always produce -/// (see that section). -fn discover_new_descendant_targets( - candidate: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, -) { - walk_children(candidate, order, nodes, counts); -} - -/// Visit `node`: count this occurrence, and — the first time this exact -/// `Rc` is seen — record it as a target and recurse into its children. -fn walk( - node: &Rc, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, -) { - let ptr = Rc::as_ptr(node); - let already_visited = counts.contains_key(&ptr); - *counts.entry(ptr).or_insert(0) += 1; - if !already_visited { - order.push(ptr); - nodes.insert(ptr, Rc::clone(node)); - walk_children(node, order, nodes, counts); - } -} - -/// `node`'s own **relational-skeleton** operator children — the same scope -/// `asap_types::pre_asap::cse::share_common_sub_dags`/`rebuild_children` -/// itself uses (see that module's "Algorithm" section) and -/// `tests::count_consumers` mirrors for its own fixtures. Exhaustive over -/// every `QueryExpr` variant: a new variant fails to compile here until this -/// match is extended too. -fn walk_children( - node: &QueryExpr, - order: &mut Vec<*const QueryExpr>, - nodes: &mut HashMap<*const QueryExpr, Rc>, - counts: &mut HashMap<*const QueryExpr, usize>, -) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, order, nodes, counts), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => walk(child, order, nodes, counts), - Concat { children, .. } => { - for c in children { - walk_children(c, order, nodes, counts); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - walk(left, order, nodes, counts); - walk(right, order, nodes, counts); - } - BinaryOp { lhs, rhs, .. } => { - walk(lhs, order, nodes, counts); - walk(rhs, order, nodes, counts); - } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::accuracy::PropagationStats; - use crate::cost_model::Cost; - use crate::test_support::lower_promql; - use asap_types::pre_asap::agg_intent::{ - agg_is_exact, default_cardinality, default_quantile, MathFunc, TimeFunc, - }; - use asap_types::pre_asap::query_expr::{Reduction as ReductionTy, Source}; - use asap_types::pre_asap::schema::{DataType, Field, Schema as SchemaTy}; - use asap_types::types::AccuracyTarget; - use std::collections::HashMap; - - // Candidate shape without execution timing: what is computed, not where. - fn timing_free_shape(node: &Rc) -> serde_json::Value { - fn strip(value: &mut serde_json::Value) { - match value { - serde_json::Value::Object(fields) => { - fields.remove("timing"); - fields.values_mut().for_each(strip); - } - serde_json::Value::Array(values) => values.iter_mut().for_each(strip), - _ => {} - } - } - let mut shape = - serde_json::to_value(asap_types::post_asap::compile_post_asap_dag(node).unwrap()) - .unwrap(); - strip(&mut shape); - shape - } - - // Rate inventories never offer two candidates that differ only in timing. - #[test] - fn rate_candidate_inventories_have_no_timing_only_duplicates() { - for (query, accuracy) in [ - ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), - ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), - ] { - let root = Rc::new(lower_promql(query, accuracy)); - let inventory = search_workload(vec![(0usize, root)]) - .enumerate_candidate_dags(4096) - .unwrap(); - let shapes = inventory - .candidates - .iter() - .map(|forest| timing_free_shape(&forest[0].1)) - .collect::>(); - for (i, shape) in shapes.iter().enumerate() { - assert!(!shapes[..i].contains(shape), "{query}: duplicate {i}"); - } - } - } - - // Grouped Sum over Rate readouts stays a summary state in the inventory, - // so lifecycle assignment can place it in precompute or at query time. - #[test] - fn grouped_rate_sum_inventory_keeps_sum_state_for_lifecycle_placement() { - let root = Rc::new(lower_promql( - "sum by(job)(rate(m[1m]))", - AccuracyTarget::Exact, - )); - let inventory = search_workload(vec![(0usize, root)]) - .enumerate_candidate_dags(4096) - .unwrap(); - let is_exact = |node: &SummaryNode, kind: ExactKind| { - matches!(&node.expr, SummaryExpr::SummaryAgg { - family: FieldDataType::ExactAggregate(k, _), .. - } if *k == kind) - }; - assert!(inventory.candidates.iter().any(|forest| { - let SummaryExpr::ValueOperation { child: sum, .. } = &forest[0].1.expr else { - return false; - }; - let SummaryExpr::SummaryAgg { child: rate, .. } = &sum.expr else { - return false; - }; - is_exact(sum, ExactKind::Sum) - && matches!(&rate.expr, SummaryExpr::ValueOperation { - child, operation: ValueOperation::FinalizeExactAccumulator, .. - } if is_exact(child, ExactKind::Rate)) - })); - } - - // Every exposed query result has a readout; internal accumulator frontiers stay states. - #[test] - fn query_candidate_roots_do_not_leak_exact_accumulator_state() { - for query in [ - "sum by(job)(rate(m[1m]))", - "sum by(job)(m)", - "sum_over_time(m[1m])", - ] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact)); - let space = search_workload(vec![(0usize, root.clone())]); - let inventory = space.enumerate_candidate_dags(4096).unwrap(); - assert!(!inventory.candidates.is_empty()); - let strategy = SketchAlgorithmStrategy::new(&DefaultCostModel); - for candidate in strategy.propose(&TargetSubDAG::new(&root)).candidates { - if let Replacement::Summary(node) = candidate.replacement { - let output = finalize_query_candidate(node, &root).unwrap(); - assert!( - output - .schema - .fields - .iter() - .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), - "direct candidate {query} leaks state" - ); - } - } - let selected = space - .global_selection(&DefaultCostModel) - .assemble_selected_query(&space.roots[0].1) - .unwrap() - .unwrap(); - for node in inventory - .candidates - .iter() - .map(|forest| &forest[0].1) - .chain(std::iter::once(&selected)) - { - assert!( - node.schema - .fields - .iter() - .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), - "{query}: query root leaks state: {:?}", - node.schema - ); - } - } - } - - #[test] - fn unpriced_inventory_retains_quantile_families_and_raw_execution() { - let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); - let space = search_workload(vec![(0usize, query)]); - let inventory = space.enumerate_candidate_dags(4096).unwrap(); - let roots = inventory - .candidates - .iter() - .map(|forest| format!("{:?}", forest[0].1)) - .collect::>(); - assert!(roots.iter().any(|root| root.contains("Kll"))); - assert!(roots.iter().any(|root| root.contains("DDSketch"))); - assert!(inventory - .candidates - .iter() - .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); - } - - // Independent roots must not require materializing their Cartesian product. - #[test] - fn root_inventory_preserves_choices_without_workload_cartesian_expansion() { - let roots = (0..24usize) - .map(|id| { - ( - id, - Rc::new(agg( - vec![2], - default_quantile((id + 1) as f64 / 25.0), - metric_scan(&["job"]), - )), - ) - }) - .collect(); - let space = search_workload(roots); - assert!(space.enumerate_candidate_dags(4096).is_err()); - for id in 0..24 { - let inventory = space.enumerate_candidate_dags_for_root(&id, 4096).unwrap(); - assert!(inventory - .candidates - .iter() - .all(|forest| forest.len() == 1 && forest[0].0 == id)); - let descriptions = inventory - .candidates - .iter() - .map(|forest| format!("{:?}", forest[0].1)) - .collect::>(); - assert!(descriptions.iter().any(|node| node.contains("Kll"))); - assert!(descriptions.iter().any(|node| node.contains("DDSketch"))); - assert!(inventory - .candidates - .iter() - .any(|forest| matches!(forest[0].1.expr, SummaryExpr::KeepPreAsap(_)))); - } - assert!(space.enumerate_candidate_dags_for_root(&24, 4096).is_err()); - assert!(space.enumerate_candidate_dags_for_root(&0, 0).is_err()); - } - - // Factoring changes enumeration, not the set of root computations. - #[test] - fn root_inventory_matches_projection_of_exhaustive_workload_inventory() { - let roots = (0..2usize) - .map(|id| { - ( - id, - Rc::new(agg( - vec![2], - default_quantile(0.5 + id as f64 * 0.4), - metric_scan(&["job"]), - )), - ) - }) - .collect(); - let space = search_workload(roots); - let full = space.enumerate_candidate_dags(4096).unwrap(); - for id in 0..2 { - let inventory = space.enumerate_candidate_dags_for_root(&id, 4096).unwrap(); - for forest in &full.candidates { - let node = &forest.iter().find(|(root, _)| *root == id).unwrap().1; - assert!(inventory.candidates.iter().any(|one| &one[0].1 == node)); - } - for one in &inventory.candidates { - assert!(full.candidates.iter().any(|forest| forest - .iter() - .any(|(root, node)| *root == id && node == &one[0].1))); - } - } - } - - #[test] - fn inventory_budget_never_returns_a_silent_partial_search() { - let query = Rc::new(agg(vec![2], default_quantile(0.9), metric_scan(&["job"]))); - let space = search_workload(vec![(0usize, query)]); - assert!(space.enumerate_candidate_dags(0).is_err()); - } - - fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { - Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(left)), - op: asap_types::pre_asap::CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(right)), - })) - } - - // Finite samples can overflow a sum although their native average is finite. - #[test] - fn temporal_average_requires_finite_division_guard() { - let root = Rc::new(lower_promql("avg_over_time(a[5m])", AccuracyTarget::Exact)); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); - let operator = candidates - .iter() - .find_map(|c| match &c.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::BinaryOp { operator, .. } => Some(operator), - _ => None, - }, - _ => None, - }) - .expect("maintained average candidate"); - assert!(operator.checked_finite_division); - assert!( - crate::rewrite::SemanticEquivalentRewriteStrategy - .replacements(&TargetSubDAG::new(&root)) - .is_empty(), - "an unconditional pre-ASAP rewrite would bypass the runtime guard" - ); - } - - // Approximate requests also admit exact temporal ranking candidates. - #[test] - fn approximate_temporal_topk_admits_exact_maintained_values() { - let root = Rc::new(lower_promql( - "topk by(job)(1,count_over_time(a[5m]))", - AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - )); - let planning_inputs = - CandidatePlanningInputs::with_default_accuracy(&crate::cost_model::DefaultCostModel); - let node = exact_topk_over_temporal_values(&root, planning_inputs) - .unwrap() - .expect("exact ranking is legal for an approximate request"); - assert!(node.guarantee.as_ref().unwrap().is_exact()); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); - } - - // Exact Top-K consumes the Planner's maintained temporal values. - #[test] - fn exact_temporal_topk_has_a_maintained_value_candidate() { - for query in [ - "topk(5, sum_over_time(a[5m]))", - "topk by(job)(5, count_over_time(a[5m]))", - ] { - let root = Rc::new(lower_promql(query, AccuracyTarget::Exact)); - let planning_inputs = CandidatePlanningInputs::with_default_accuracy( - &crate::cost_model::DefaultCostModel, - ); - let node = exact_topk_over_temporal_values(&root, planning_inputs) - .unwrap() - .expect("exact Top-K candidate"); - assert!(node.guarantee.as_ref().unwrap().is_exact()); - let SummaryExpr::ValueOperation { - child: sorted, - operation: - ValueOperation::Limit { - n, - offset, - partition_by, - }, - .. - } = &node.expr - else { - panic!("temporal TopK must compose Sort and Limit"); - }; - assert_eq!((*n, *offset), (5, 0)); - let SummaryExpr::ValueOperation { - operation: - ValueOperation::Sort { - keys, - partition_by: sort_groups, - }, - child: values, - .. - } = &sorted.expr - else { - panic!("Limit must consume sorted temporal values"); - }; - assert_eq!(sort_groups, partition_by); - assert_eq!( - partition_by.keys().len(), - usize::from(query.contains("by(job)")) - ); - assert_eq!(keys.len(), 1); - assert!(!keys[0].ascending); - assert_eq!(node.schema, values.schema); - asap_types::post_asap::compile_post_asap_dag(&node).unwrap(); - } - } - - // A bounded exact mean can share the relative division proof with a quantile. - #[test] - fn bounded_mean_quantile_ratio_is_certified() { - struct Domain; - impl AccuracyEvidenceProvider for Domain { - fn quantile_input_domain( - &self, - _: &QueryExpr, - ) -> Option { - Some(crate::accuracy::QuantileInputDomain { - lower: 1.0, - upper: 1000.0, - max_samples: 10000, - contract: "finite test population".into(), - }) - } - } - let target = AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, + let (mut eps, mut delta) = own; + for sibling in siblings { + let Some(NonASAPOp::Aggregate { + reduction: sibling_reduction, + filters: sibling_filters, + child: sibling_child, + .. + }) = sibling.non_asap() + else { + continue; }; - let inputs = CandidatePlanningInputs { - evidence: &Domain, - ..CandidatePlanningInputs::with_default_accuracy(&DefaultCostModel) + let Some(other) = bindable_intent(sibling) else { + continue; }; - for query in [ - "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", - "quantile_over_time(0.5,a[5m]) / avg_over_time(a[5m])", - ] { - let root = Rc::new(lower_promql(query, target.clone())); - let node = realize_binary(&root, inputs, Some(&target)) - .unwrap() - .expect("bounded ratio candidate"); - assert!(DefaultAccuracyModel.satisfies(node.guarantee.as_ref().unwrap(), &target)); - } - } - - // Missing domain proof permits an uncertified direct quantile ratio only. - #[test] - fn quantile_ratio_without_input_proof_has_no_root_guarantee() { - let target = AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, + let Some(accuracy) = approximate(other) else { + continue; }; - let root = Rc::new(lower_promql( - "quantile_over_time(0.5,a[5m]) / quantile_over_time(0.9,a[5m])", - target.clone(), - )); - let planning_inputs = - CandidatePlanningInputs::with_default_accuracy(&crate::cost_model::DefaultCostModel); - let candidate = realize_binary(&root, planning_inputs, Some(&target)) - .unwrap() - .expect("direct quantile ratio candidate"); - assert!(candidate.guarantee.is_none()); - - let other = Rc::new(lower_promql( - "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", - target.clone(), - )); - assert!(realize_binary(&other, planning_inputs, Some(&target)) - .unwrap() - .is_none()); - } - - #[test] - fn relational_join_predicate_requires_and_normalizes_cross_input_columns() { - let forward = normalize_cross_input_equi_predicate(&equi_pred(1, 3), 2, 4) - .expect("left-to-right equality"); - let reverse = normalize_cross_input_equi_predicate(&equi_pred(3, 1), 2, 4) - .expect("right-to-left equality"); - assert_eq!(forward, reverse, "reverse equality must be canonicalized"); - assert!(normalize_cross_input_equi_predicate(&equi_pred(0, 1), 2, 4).is_none()); - assert!(normalize_cross_input_equi_predicate(&equi_pred(0, 4), 2, 4).is_none()); - } - - #[test] - fn relational_join_is_exact_only_when_both_inputs_are_exact() { - let exact = ResultGuarantee::exact("test exact input"); - assert!(relational_join_guarantee(Some(&exact), Some(&exact)) - .is_some_and(|guarantee| guarantee.is_exact())); - assert!(relational_join_guarantee(Some(&exact), None).is_none()); - assert!(relational_join_guarantee(None, Some(&exact)).is_none()); - } - - fn eps(e: f64) -> AccuracyTarget { - AccuracyTarget::Epsilon(e) - } - - // ── realizations_for_intent / sizing ─────────────────────────────── - - /// The most-preferred `Realization` — `realizations_for_intent(intent, - /// &DefaultCostModel)`'s head — for tests that only care about the - /// default pick, not the full candidate list. - fn preferred(intent: &AggIntent) -> Realization { - realizations_for_intent(intent, &DefaultCostModel) - .into_iter() - .next() - .expect("every intent has at least one Realization") - } - - /// Shorthand for asserting the realization *category*. - #[derive(Debug, PartialEq)] - enum Cat { - Sketch(SketchAlgorithm), - Acc(ExactKind), - Pass, - } - - fn cat(intent: &AggIntent) -> Cat { - match preferred(intent) { - Realization::ExactAggregate { kind, .. } => Cat::Acc(kind), - Realization::Sketch(kind) => Cat::Sketch(kind.algorithm().clone()), - Realization::PassThrough => Cat::Pass, - other => { - panic!("this coverage matrix expects only Exact/Sketch/PassThrough, got {other:?}") - } - } - } - - /// The `AggIntent → SummaryKind` coverage matrix (issue #98): every intent - /// variant maps to a sketch, an exact accumulator, or an explicit - /// pass-through. `realizations_for_intent`'s match is exhaustive, so a - /// new variant cannot compile without a decision; this matrix pins what - /// each decision *is* (its preferred/first candidate). - #[test] - fn agg_intent_to_summary_kind_coverage_matrix() { - use AggIntent as A; - use Cat::*; - use ExactKind as E; - use SketchAlgorithm as K; - let matrix: Vec<(A, Cat)> = vec![ - // approximate-capable, at an ε target → sketch - (default_quantile(0.99), Sketch(K::Kll)), - (default_cardinality(), Sketch(K::Hll)), - ( - A::Cardinality { - cols: vec![0, 1], - accuracy: eps(0.01), - }, - Sketch(K::Hll), - ), - ( - A::Count { - accuracy: eps(0.01), - }, - Sketch(K::Cms), - ), - ( - A::TopK { - k: 10, - accuracy: eps(0.01), - }, - Sketch(K::CmsWithHeap), - ), - // the same intents at Exact → exact realization - ( - A::Quantile { - col: None, - q: 0.5, - accuracy: AccuracyTarget::Exact, - }, - Pass, - ), - ( - A::Cardinality { - cols: vec![], - accuracy: AccuracyTarget::Exact, - }, - Pass, - ), - ( - A::Cardinality { - cols: vec![0, 1], - accuracy: AccuracyTarget::Exact, - }, - Pass, - ), - ( - A::Count { - accuracy: AccuracyTarget::Exact, - }, - Acc(E::Count), - ), - ( - A::TopK { - k: 10, - accuracy: AccuracyTarget::Exact, - }, - Pass, - ), - // exact mergeable accumulators - (A::Sum { col: None }, Acc(E::Sum)), - (A::Min { col: None }, Acc(E::Min)), - (A::Max { col: None }, Acc(E::Max)), - (A::Rate, Acc(E::Rate)), - (A::IRate, Acc(E::IRate)), - (A::Increase, Acc(E::Increase)), - // exact but non-mergeable → pass-through - (A::Avg { col: None }, Pass), - ( - A::StdDev { - col: None, - population: false, - }, - Pass, - ), - ( - A::Variance { - col: None, - population: true, - }, - Pass, - ), - // classic-bucket histogram_quantile is not re-sketchable (#79) - (A::HistogramQuantile { q: 0.99, le: 0 }, Pass), - // counter-derivative / range-vector functions (#44) - (A::Changes, Pass), - (A::Delta, Pass), - (A::IDelta, Pass), - (A::Deriv, Pass), - (A::Resets, Pass), - (A::PredictLinear { seconds: 60.0 }, Pass), - ( - A::DoubleExpSmoothing { - smoothing: 0.5, - trend: 0.5, - }, - Pass, - ), - // native-histogram accessors (#43) - (A::HistogramCount, Pass), - (A::HistogramSum, Pass), - (A::HistogramAvg, Pass), - (A::HistogramStdDev, Pass), - (A::HistogramStdVar, Pass), - ( - A::HistogramFraction { - lower: 0.0, - upper: 1.0, - }, - Pass, - ), - // per-sample transforms (#45, #46) + presence (#47) - (A::Math(MathFunc::Abs), Pass), - (A::TimeFn(TimeFunc::Hour), Pass), - (A::Absent, Pass), - (A::AbsentOverTime, Pass), - (A::PresentOverTime, Pass), - // extended aggregations (#49) - (A::Group, Pass), - (A::CountValues { label: "v".into() }, Pass), - // additional range reducers (#51) - (A::LastOverTime, Pass), - (A::FirstOverTime, Pass), - (A::MadOverTime, Pass), - (A::TsOfMinOverTime, Pass), - (A::TsOfMaxOverTime, Pass), - (A::TsOfFirstOverTime, Pass), - (A::TsOfLastOverTime, Pass), - ]; - for (intent, expected) in &matrix { - assert_eq!(&cat(intent), expected, "realization for {intent:?}"); - } - // Every accumulator pick is mergeable; every sketch pick is on a - // genuinely approximate target (the `agg_is_*` helpers stay truthful). - for (intent, expected) in &matrix { - if let Cat::Acc(_) = expected { - assert!(agg_is_mergeable(intent), "{intent:?}"); - } - if let Cat::Sketch(_) = expected { - assert!( - !agg_is_exact(intent) || matches!(intent, AggIntent::Count { .. }), - "{intent:?} sketches only under an approximate target" - ); + let same_intent = match (intent, other) { + (AggIntent::Quantile { col, .. }, AggIntent::Quantile { col: other_col, .. }) => { + col == other_col } + _ => override_accuracy(intent, accuracy) == *other, + }; + if same_intent + && sibling_reduction == reduction + && sibling_filters == filters + && (Rc::ptr_eq(sibling_child, child) || sibling_child == child) + { + let (sibling_eps, sibling_delta) = accuracy_budget(accuracy); + eps = eps.min(sibling_eps); + delta = delta.min(sibling_delta); } } - - // Correlation must never acquire a single-input sketch or scalar accumulator. - #[test] - fn pearson_corr_keeps_exact_paired_input() { - let intent = AggIntent::PearsonCorr { left: 0, right: 1 }; - assert!(matches!( - realizations_for_intent(&intent, &crate::cost_model::DefaultCostModel).as_slice(), - [Realization::PassThrough] - )); - assert!(summary_candidates(&intent).is_empty()); + if (eps, delta) == own { + None + } else if delta == DEFAULT_DELTA { + Some(AccuracyTarget::Epsilon(eps)) + } else { + Some(AccuracyTarget::EpsilonDelta { + epsilon: eps, + delta, + }) } +} - #[test] - fn accuracy_target_drives_the_boundary() { - // Same intent, three targets → three different decisions. - let exact = AggIntent::Quantile { - col: None, - q: 0.99, - accuracy: AccuracyTarget::Exact, - }; - assert_eq!(preferred(&exact), Realization::PassThrough); - - let approx = default_quantile(0.99); // ε = 0.01 - assert_eq!( - preferred(&approx), - Realization::Sketch(SketchKind::new( - SketchAlgorithm::Kll, - SketchParams::Kll { k: 269 }, - )) - ); +fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { + share_common_sub_dags(roots) +} - let looser = AggIntent::Quantile { - col: None, - q: 0.99, - accuracy: eps(0.05), - }; - assert_eq!( - preferred(&looser), - Realization::Sketch(SketchKind::new( - SketchAlgorithm::Kll, - SketchParams::Kll { k: 52 }, - )) +fn search_cse_workload_with<'s, Id>( + cse_roots: Vec<(Id, Rc)>, + strategies: &[Box], +) -> CandidateLogicalASAPDAGs { + for (_, root) in &cse_roots { + assert!( + !root.contains_asap(), + "search_workload: a workload root already contains an ASAP operator \ + ({}); replacement search takes the front end's pre-ASAP DAG only", + root.operator.kind_name() ); } + let mut order = Vec::new(); + let mut nodes = HashMap::new(); + let mut counts: HashMap<*const OperatorNode, usize> = HashMap::new(); + discover_targets(&cse_roots, &mut order, &mut nodes, &mut counts); + let siblings: Vec> = order + .iter() + .filter_map(|ptr| { + let node = &nodes[ptr]; + matches!(node.non_asap(), Some(NonASAPOp::Aggregate { .. })).then(|| Rc::clone(node)) + }) + .collect(); + let rollup_strategy = RollupStrategy::new(&siblings); + let accuracy_reconciliation_strategy = AccuracyReconciliationStrategy::new(&siblings); + let limits: Vec> = order + .iter() + .filter_map(|ptr| { + let node = &nodes[ptr]; + matches!(node.non_asap(), Some(NonASAPOp::Limit { .. })).then(|| Rc::clone(node)) + }) + .collect(); + let topk_reuse_strategy = TopKLimitReuseStrategy::new(&limits); - #[test] - fn default_cardinality_sizes_hll_to_its_rse_magnitude() { - assert_eq!( - preferred(&default_cardinality()), - Realization::Sketch(SketchKind::new( - SketchAlgorithm::Hll, - SketchParams::Hll { precision: 14 }, - )) + let mut groups: HashMap<*const OperatorNode, TargetSubDAGCandidates> = HashMap::new(); + for ptr in &order { + groups.insert( + *ptr, + TargetSubDAGCandidates::new(Rc::clone(&nodes[ptr]), counts[ptr]), ); } - // Exact counting remains a legal candidate under an approximate target. - #[test] - fn approximate_count_includes_exact_accumulator_candidate() { - let intent = AggIntent::Count { - accuracy: eps(0.01), - }; - assert!(realizations_for_intent(&intent, &DefaultCostModel) - .iter() - .any(|candidate| matches!( - candidate, - Realization::ExactAggregate { - kind: ExactKind::Count, - .. - } - ))); - } - - #[test] - fn epsilon_delta_sizes_cms_depth() { - let intent = AggIntent::Count { - accuracy: AccuracyTarget::EpsilonDelta { - epsilon: 0.001, - delta: 0.001, - }, - }; - assert_eq!( - preferred(&intent), - Realization::Sketch(SketchKind::new( - SketchAlgorithm::Cms, - SketchParams::Cms { - width: 2719, - depth: 7 - }, // ⌈e/0.001⌉, ⌈ln 1000⌉ - )) - ); - // Epsilon-only falls back to DEFAULT_DELTA → depth 5. - let intent = AggIntent::Count { - accuracy: eps(0.001), - }; - assert_eq!( - preferred(&intent), - Realization::Sketch(SketchKind::new( - SketchAlgorithm::Cms, - SketchParams::Cms { - width: 2719, - depth: 5 - }, - )) + // Round-based frontier: every target is asked exactly once per strategy + // (never re-asked — see the module docs' "Termination" section on why + // that matters for `Replacement::Summary` dedup specifically). A round + // can grow the *next* round's frontier only by a candidate's own + // reachable children exposing a genuinely new, not-yet-known `Rc` — see + // `discover_new_descendant_targets`. + let mut frontier = order.clone(); + let mut rounds = 0usize; + while !frontier.is_empty() { + rounds += 1; + assert!( + rounds <= MAX_SEARCH_ITERATIONS, + "search_workload: fixpoint search did not converge within {MAX_SEARCH_ITERATIONS} \ + rounds — a registered ReplacementStrategy's Replacement::Rewrite candidates keep \ + exposing new, never-before-seen descendant structure every round. \ + ASAPStrategies/SharedSubDAGStrategy never do this (see replacement.rs's \ + module docs' \"Termination\" section); check any custom strategies passed to \ + search_workload_with.", ); - } - #[test] - fn topk_heap_capacity_respects_accuracy_and_output_count() { - let intent = AggIntent::TopK { - k: 25, - accuracy: eps(0.01), - }; - match preferred(&intent) { - Realization::Sketch(kind) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap => { - let SketchParams::CmsWithHeap { - width, - depth, - heap_size, - } = kind.params() - else { - unreachable!("SketchKind validates CmsWithHeap params") - }; - assert_eq!(*heap_size, 100); - assert_eq!(*width, 272); // ⌈e/0.01⌉ - assert_eq!(*depth, 5); - } - other => panic!("expected CmsWithHeap, got {other:?}"), - } - } + let targets_before = order.len(); + for ptr in &frontier { + let (root, consumer_count) = { + let group = &groups[ptr]; + (Rc::clone(&group.target), group.consumer_count) + }; + let strictest = strictest_sibling_accuracy(&root, &siblings); + let mut target = TargetSubDAG::with_consumer_count(&root, consumer_count); + target.strictest_sibling_accuracy = strictest.as_ref(); - #[test] - fn candidate_lists_match_the_issue_map() { - assert_eq!( - summary_candidates(&default_quantile(0.5)), - &[SketchAlgorithm::Kll, SketchAlgorithm::DDSketch] - ); - assert_eq!( - summary_candidates(&default_cardinality()), - &[ - SketchAlgorithm::Hll, - SketchAlgorithm::Theta, - SketchAlgorithm::Kmv, - SketchAlgorithm::UnivMon - ] - ); - assert_eq!( - summary_candidates(&AggIntent::TopK { - k: 5, - accuracy: eps(0.01) - }), - &[ - SketchAlgorithm::CmsWithHeap, - SketchAlgorithm::CountSketchWithHeap - ] - ); - assert_eq!( - summary_candidates(&AggIntent::Count { - accuracy: eps(0.01) - }), - &[ - SketchAlgorithm::Cms, - SketchAlgorithm::CountSketch, - SketchAlgorithm::UnivMon - ] - ); - assert!(summary_candidates(&AggIntent::Rate).is_empty()); - } + let mut proposed = Vec::new(); + let mut rejected = Vec::new(); + for strategy in strategies { + if strategy.matches(&target) { + let name = strategy.name(); + let proposals = strategy.propose(&target); + proposed.extend(proposals.candidates.into_iter().map(|mut candidate| { + candidate.strategy = name; + candidate + })); + rejected.extend(proposals.rejected.into_iter().map(|mut rejection| { + rejection.strategy = name; + rejection + })); + } + } + if rollup_strategy.matches(&target) { + let name = rollup_strategy.name(); + proposed.extend(rollup_strategy.replacements(&target).into_iter().map( + |mut candidate| { + candidate.strategy = name; + candidate + }, + )); + } + if accuracy_reconciliation_strategy.matches(&target) { + let name = accuracy_reconciliation_strategy.name(); + proposed.extend( + accuracy_reconciliation_strategy + .replacements(&target) + .into_iter() + .map(|mut candidate| { + candidate.strategy = name; + candidate + }), + ); + } + if topk_reuse_strategy.matches(&target) { + let name = topk_reuse_strategy.name(); + proposed.extend(topk_reuse_strategy.replacements(&target).into_iter().map( + |mut candidate| { + candidate.strategy = name; + candidate + }, + )); + } - #[test] - fn realizations_for_intent_enumerates_every_candidate_ranked() { - // Quantile's candidate list is [Kll, DDSketch] — realizations_for_intent - // must return both, ranked with the DefaultCostModel's preferred - // (Kll) first. - let kinds: Vec = - realizations_for_intent(&default_quantile(0.99), &DefaultCostModel) - .into_iter() - .map(|realization| match realization { - Realization::Sketch(kind) => kind.algorithm().clone(), - other => panic!("expected Sketch, got {other:?}"), - }) - .collect(); - assert_eq!(kinds, vec![SketchAlgorithm::Kll, SketchAlgorithm::DDSketch]); - } + for candidate in &proposed { + if let Replacement::SubDAG(rc) = &candidate.replacement { + if is_logical_rewrite(rc) { + discover_new_descendant_targets(rc, &mut order, &mut nodes, &mut counts); + } + } + } - #[test] - fn degenerate_epsilon_saturates_to_tightest_params() { - let intent = AggIntent::Quantile { - col: None, - q: 0.99, - accuracy: eps(0.0), - }; - assert_eq!( - preferred(&intent), - Realization::Sketch(SketchKind::new( - SketchAlgorithm::Kll, - SketchParams::Kll { k: 65_535 }, - )) - ); + let group = groups + .get_mut(ptr) + .expect("every discovered target has a group"); + for candidate in proposed { + group.add_candidate(candidate); + } + group.rejected.extend(rejected); + } + + // Any pointer `discover_new_descendant_targets` appended to `order` + // this round is a genuinely new target — give it a group and process + // it next round. Targets already in `groups` are never revisited. + let new_targets = &order[targets_before..]; + for ptr in new_targets { + groups.entry(*ptr).or_insert_with(|| { + TargetSubDAGCandidates::new(Rc::clone(&nodes[ptr]), counts[ptr]) + }); + } + frontier = new_targets.to_vec(); } - // ── posterior_aware_size_params (issue #239, integration point 2) ────── + add_effective_count_cse_candidates(&order, &mut groups); - fn count_intent(e: f64) -> AggIntent { - AggIntent::Count { accuracy: eps(e) } + CandidateLogicalASAPDAGs { + roots: cse_roots, + groups, + order, + composition_plans: Vec::new(), } +} - #[test] - fn posterior_aware_sizing_shrinks_width_under_stated_assumption() { - let intent = count_intent(0.01); - let worst_case = default_size_params(SketchAlgorithm::Cms, &intent, 0.01, 0.01); - let relaxed = posterior_aware_size_params( - SketchAlgorithm::Cms, - &intent, - 0.01, - 0.01, - ExpectedCaseSizing { - width_relaxation: 0.5, - }, - ); - match (worst_case, relaxed) { - ( - SketchParams::Cms { - width: w0, - depth: d0, - }, - SketchParams::Cms { - width: w1, - depth: d1, - }, - ) => { - assert!( - w1 < w0, - "expected relaxed width {w1} to be strictly smaller than worst-case {w0}" - ); - assert_eq!(d0, d1, "depth must be unaffected by width_relaxation"); +/// Materialize share/recompute alternatives for descendants whose raw edge +/// count is one but whose effective count can exceed one when a repeated +/// ancestor is recomputed. We only do this when an ordinary repeated group +/// proves that `SharedSubDAGStrategy` is part of this search's strategy set. +fn add_effective_count_cse_candidates( + order: &[*const OperatorNode], + groups: &mut HashMap<*const OperatorNode, TargetSubDAGCandidates>, +) { + let mut possible_children: HashMap<*const OperatorNode, Vec<*const OperatorNode>> = + HashMap::new(); + for ptr in order { + let group = &groups[ptr]; + let children = possible_children.entry(*ptr).or_default(); + for (child, _) in direct_child_counts(&group.target) { + if !children.contains(&child) { + children.push(child); + } + } + for candidate in &group.candidates { + if let Replacement::SubDAG(rewrite) = &candidate.replacement { + if !is_logical_rewrite(rewrite) { + continue; + } + for (child, _) in direct_child_counts(rewrite) { + if !children.contains(&child) { + children.push(child); + } + } } - other => panic!("expected Cms/Cms pair, got {other:?}"), } } - #[test] - fn posterior_aware_sizing_at_full_relaxation_matches_worst_case() { - // width_relaxation = 1.0 must reproduce default_size_params exactly - // — the "no risk taken" boundary. - let intent = count_intent(0.01); - let worst_case = default_size_params(SketchAlgorithm::Cms, &intent, 0.01, 0.01); - let relaxed = posterior_aware_size_params( - SketchAlgorithm::Cms, - &intent, - 0.01, - 0.01, - ExpectedCaseSizing { - width_relaxation: 1.0, - }, - ); - assert_eq!(worst_case, relaxed); - } - - #[test] - fn posterior_aware_sizing_invalid_relaxation_falls_back_to_worst_case() { - let intent = count_intent(0.01); - let worst_case = default_size_params(SketchAlgorithm::Cms, &intent, 0.01, 0.01); - for bad in [0.0, -0.5, 1.5, f64::NAN, f64::INFINITY] { - let relaxed = posterior_aware_size_params( - SketchAlgorithm::Cms, - &intent, - 0.01, - 0.01, - ExpectedCaseSizing { - width_relaxation: bad, - }, - ); - assert_eq!( - worst_case, relaxed, - "width_relaxation={bad} should fall back to the worst-case width" - ); + let mut potentially_repeated = HashSet::new(); + let mut queue = VecDeque::new(); + for ptr in order { + let group = &groups[ptr]; + if group.consumer_count >= 2 && cse_candidate_pair(group).is_some() { + potentially_repeated.insert(*ptr); + queue.push_back(*ptr); } } - - #[test] - fn posterior_aware_sizing_does_not_apply_cms_l1_relaxation_to_count_sketch() { - let cms_heap_intent = AggIntent::TopK { - k: 7, - accuracy: eps(0.01), - }; - let assumption = ExpectedCaseSizing { - width_relaxation: 0.25, - }; - // CountSketch - assert_eq!( - posterior_aware_size_params( - SketchAlgorithm::CountSketch, - &count_intent(0.01), - 0.01, - 0.01, - assumption - ), - default_size_params( - SketchAlgorithm::CountSketch, - &count_intent(0.01), - 0.01, - 0.01 - ), - ); - // CmsWithHeap / CountSketchWithHeap carry k through untouched. - match posterior_aware_size_params( - SketchAlgorithm::CmsWithHeap, - &cms_heap_intent, - 0.01, - 0.01, - assumption, - ) { - SketchParams::CmsWithHeap { - width, - depth, - heap_size, - } => { - assert_eq!(width, 68); - assert_eq!(depth, 5); - assert_eq!(heap_size, 100); + while let Some(parent) = queue.pop_front() { + if let Some(children) = possible_children.get(&parent) { + for child in children { + if groups.contains_key(child) && potentially_repeated.insert(*child) { + queue.push_back(*child); + } } - other => panic!("expected CmsWithHeap, got {other:?}"), } } - #[test] - fn posterior_aware_sizing_leaves_non_cms_kinds_unchanged() { - // Kll/Hll/etc. have no width_relaxation concept — must be byte-for- - // byte identical to default_size_params. - let intent = default_quantile(0.99); - let assumption = ExpectedCaseSizing { - width_relaxation: 0.1, - }; - assert_eq!( - posterior_aware_size_params(SketchAlgorithm::Kll, &intent, 0.01, 0.01, assumption), - default_size_params(SketchAlgorithm::Kll, &intent, 0.01, 0.01), - ); - } - - #[test] - fn default_size_params_unchanged_by_new_function_existing() { - // Regression pin: default_size_params's own worst-case behavior for - // existing callers must be untouched by adding - // posterior_aware_size_params alongside it. - assert_eq!( - default_size_params(SketchAlgorithm::Cms, &count_intent(0.001), 0.001, 0.001), - SketchParams::Cms { - width: 2719, - depth: 7 - }, - ); + for ptr in order { + let group = groups + .get_mut(ptr) + .expect("every discovered site has a group"); + if potentially_repeated.contains(ptr) && cse_candidate_pair(group).is_none() { + let target = Rc::clone(&group.target); + let site = TargetSubDAG::with_consumer_count(&target, 2); + for mut candidate in SharedSubDAGStrategy.replacements(&site) { + candidate.rationale = format!( + "{}: this sub-DAG can become repeated when a repeated ancestor is recomputed; \ + global_selection decides using its effective consumer count", + match candidate.provenance { + ReplacementProvenance::CseShare => "build once and share", + ReplacementProvenance::CseRecompute => "recompute independently", + _ => unreachable!("SharedSubDAGStrategy only emits CSE candidates"), + } + ); + group.add_candidate(candidate); + } + } } +} - // ── SketchAlgorithmStrategy / SharedSubDAGStrategy fixtures ─────────── +// ── target discovery ───────────────────────────────────────────────────── - fn metric_scan(labels: &[&str]) -> QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: SchemaTy::with_time_index(columns, 0, vec![]), - } +/// Walk every root's whole DAG, discovering one `TargetSubDAG` per distinct +/// `Rc` and its real `consumer_count` — see the module docs' "Where +/// `TargetSubDAG` discovery comes from" section for the full rationale. +pub(crate) fn discover_targets( + roots: &[(Id, Rc)], + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, +) { + for (_, root) in roots { + walk(root, order, nodes, counts); + } +} + +/// Scan `candidate`'s **children** (deliberately never `candidate`'s own +/// top-level pointer — see the module docs' "Termination" section: a +/// logical rewrite's value is an alternative *for* the target that +/// proposed it, never a new target of its own) for any `Rc` not already +/// known, appending each to `order`/`nodes`/`counts` so +/// [`search_workload_with`]'s next round processes it. A no-op when every +/// child is already known — the case both shipped strategies always produce +/// (see that section). +fn discover_new_descendant_targets( + candidate: &Rc, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, +) { + walk_children(candidate, order, nodes, counts); +} + +/// Visit `node`: count this occurrence, and — the first time this exact +/// `Rc` is seen — record it as a target and recurse into its children. +fn walk( + node: &Rc, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, +) { + let ptr = Rc::as_ptr(node); + let already_visited = counts.contains_key(&ptr); + *counts.entry(ptr).or_insert(0) += 1; + if !already_visited { + order.push(ptr); + nodes.insert(ptr, Rc::clone(node)); + walk_children(node, order, nodes, counts); } +} - fn agg(by: Vec, intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: ReductionTy::by(by), - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), +/// `node`'s own operator children ([`OperatorNode::children`]: operator +/// inputs plus the operator nodes its scalar expressions read), the same +/// scope `asap_types::ir::cse::share_common_sub_dags` itself uses and +/// `tests::count_consumers` mirrors for its own fixtures. `Concat` is +/// transparent: its branches are walked in place of it. +fn walk_children( + node: &OperatorNode, + order: &mut Vec<*const OperatorNode>, + nodes: &mut HashMap<*const OperatorNode, Rc>, + counts: &mut HashMap<*const OperatorNode, usize>, +) { + if let Some(NonASAPOp::Concat { children, .. }) = node.non_asap() { + for c in children { + walk_children(c, order, nodes, counts); } + return; } + for child in node.children() { + walk(child, order, nodes, counts); + } +} - // ── SketchAlgorithmStrategy ───────────────────────────────────────────── +#[cfg(test)] +mod tests { + use super::*; + use crate::accuracy::PropagationStats; + use crate::test_support::{agg, agg_per_entity, lower_promql, maintained, metric_scan, timed}; + use asap_types::ir::operator::agg_intent::{ + agg_is_exact, default_cardinality, default_quantile, MathFunc, TimeFunc, + }; + use asap_types::ir::operator::operator_properties::{Reduction as ReductionTy, Source}; + use asap_types::ir::schema::ColumnId; + use asap_types::ir::schema::{DataType, Field, Schema as SchemaTy}; + use asap_types::ir::Predicate; + use asap_types::ir::TimeRangeKind; - #[test] - fn matches_a_bindable_aggregate() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let target = TargetSubDAG::new(&q); - assert!(SketchAlgorithmStrategy::default_cost_model().matches(&target)); + use asap_types::types::AccuracyTarget; + use std::collections::HashMap; + + // Candidate shape without execution timing: what is computed, not where. + fn timing_free_shape(node: &Rc) -> serde_json::Value { + fn strip(value: &mut serde_json::Value) { + match value { + serde_json::Value::Object(fields) => { + fields.remove("timing"); + fields.values_mut().for_each(strip); + } + serde_json::Value::Array(values) => values.iter_mut().for_each(strip), + _ => {} + } + } + let mut shape = serde_json::to_value( + asap_types::ir::export::compile_physical_asap_dag(&timed(node)).unwrap(), + ) + .unwrap(); + strip(&mut shape); + shape } + // Rate inventories never offer two candidates that differ only in timing. #[test] - fn does_not_match_a_multi_intent_or_having_aggregate() { - let strategy = SketchAlgorithmStrategy::default_cost_model(); - - let multi = Rc::new(QueryExpr::Aggregate { - reduction: ReductionTy::by(vec![2]), - measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }); - let target = TargetSubDAG::new(&multi); - assert!(!strategy.matches(&target)); - assert!(strategy.replacements(&target).is_empty()); - - let mut having_q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut having_q { - *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), - ))); + fn rate_candidate_inventories_have_no_timing_only_duplicates() { + for (query, accuracy) in [ + ("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact), + ("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)), + ] { + let root = lower_promql(query, accuracy); + let inventory = search_workload(vec![(0usize, root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let shapes = inventory + .candidates + .iter() + .map(|forest| timing_free_shape(&forest[0].1)) + .collect::>(); + for (i, shape) in shapes.iter().enumerate() { + assert!(!shapes[..i].contains(shape), "{query}: duplicate {i}"); + } } - let having_q = Rc::new(having_q); - let target = TargetSubDAG::new(&having_q); - assert!(!strategy.matches(&target)); - assert!(strategy.replacements(&target).is_empty()); } + // Grouped Sum over Rate evaluations stays a summary state in the inventory, + // so materialization assignment can place it in precompute or at query time. #[test] - fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); - let target = TargetSubDAG::new(&scan); - assert!(!SketchAlgorithmStrategy::default_cost_model().matches(&target)); - assert!(SketchAlgorithmStrategy::default_cost_model() - .replacements(&target) - .is_empty()); + fn grouped_rate_sum_inventory_keeps_sum_state_for_materialization_placement() { + let root = lower_promql("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact); + let inventory = search_workload(vec![(0usize, root)]) + .enumerate_candidate_dags(4096) + .unwrap(); + let is_exact = |node: &OperatorNode, kind: ExactKind| { + matches!(&node.operator, Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::ExactAggregate(k, _), .. + }) if *k == kind) + }; + assert!(inventory.candidates.iter().any(|forest| { + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: sum }) = &forest[0].1.operator else { + return false; + }; + let Operator::ASAP(ASAPOp::SummaryAgg { child: rate, .. }) = &sum.operator else { + return false; + }; + is_exact(sum, ExactKind::Sum) + && matches!(&rate.operator, Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) if is_exact(child, ExactKind::Rate)) + })); } + // Every exposed query result has a evaluation; internal accumulator frontiers stay states. #[test] - fn approximate_quantile_enumerates_every_summary_candidate() { - // Quantile's candidate list is [Kll, DDSketch] (summary_candidates) — - // every entry must come back as its own bound SummaryNode candidate, - // not just Kll (the CostModel-ranked head realizations_for_intent commits to). - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); - assert_eq!( - replacements.len(), - 2, - "expected 2 candidates, got {replacements:?}" - ); - - let kinds: Vec = replacements - .iter() - .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { - panic!("expected a Summary replacement") + fn query_candidate_roots_do_not_leak_exact_accumulator_state() { + for query in [ + "sum by(job)(rate(m[1m]))", + "sum by(job)(m)", + "sum_over_time(m[1m])", + ] { + let root = lower_promql(query, AccuracyTarget::Exact); + let space = search_workload(vec![(0usize, root.clone())]); + let inventory = space.enumerate_candidate_dags(4096).unwrap(); + assert!(!inventory.candidates.is_empty()); + let strategy = ASAPStrategies::default(); + for candidate in strategy.propose(&TargetSubDAG::new(&root)).candidates { + if let Replacement::SubDAG(node) = candidate.replacement { + let output = finalize_query_candidate(node, &root).unwrap(); + assert!( + output + .schema + .fields + .iter() + .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), + "direct candidate {query} leaks state" + ); } - }) - .collect(); - assert!(kinds.contains(&SketchAlgorithm::Kll), "{kinds:?}"); - assert!(kinds.contains(&SketchAlgorithm::DDSketch), "{kinds:?}"); - assert!( - replacements.iter().all(|r| !r.rationale.is_empty()), - "every candidate must carry a rationale" - ); + } + for node in inventory.candidates.iter().map(|forest| &forest[0].1) { + assert!( + node.schema + .fields + .iter() + .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), + "{query}: query root leaks state: {:?}", + node.schema + ); + } + } } #[test] - fn cardinality_epsilon_delta_keeps_unknown_accuracy_candidates() { - let q = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); - let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); - let kinds: Vec = replacements + fn unpriced_inventory_retains_quantile_families_and_raw_execution() { + let query = agg(vec![2], default_quantile(0.9), metric_scan(&["job"])); + let space = search_workload(vec![(0usize, query)]); + let inventory = space.enumerate_candidate_dags(4096).unwrap(); + let roots = inventory + .candidates .iter() - .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { - panic!("expected a Summary replacement") - } - }) - .collect(); - assert_eq!( - kinds, - vec![ - SketchAlgorithm::Hll, - SketchAlgorithm::Theta, - SketchAlgorithm::Kmv, - SketchAlgorithm::UnivMon, - ] - ); - - let q = Rc::new(agg( - vec![2], - AggIntent::Cardinality { - cols: vec![], - accuracy: AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, - }, - metric_scan(&["job"]), - )); - let kinds: Vec<_> = SketchAlgorithmStrategy::default_cost_model() - .replacements(&TargetSubDAG::new(&q)) + .map(|forest| format!("{:?}", forest[0].1)) + .collect::>(); + assert!(roots.iter().any(|root| root.contains("Kll"))); + assert!(roots.iter().any(|root| root.contains("DDSketch"))); + assert!(inventory + .candidates .iter() - .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { - panic!("expected a Summary replacement") - } - }) - .collect(); - assert_eq!( - kinds, - vec![ - SketchAlgorithm::Hll, - SketchAlgorithm::Theta, - SketchAlgorithm::Kmv, - SketchAlgorithm::UnivMon, - ] - ); + .any(|forest| !forest[0].1.contains_asap())); } + // Independent roots must not require materializing their Cartesian product. #[test] - fn exact_accuracy_target_yields_exactly_one_pass_through_candidate() { - // Exact quantile has no sketch candidate at all — realizations_for_intent - // produces PassThrough, the only option, so exactly one candidate. - let intent = AggIntent::Quantile { - col: None, - q: 0.99, - accuracy: AccuracyTarget::Exact, - }; - let q = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); - let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); - assert_eq!(replacements.len(), 1, "{replacements:?}"); - assert!(matches!( - &replacements[0].replacement, - Replacement::Summary(node) if matches!( - node.expr, - asap_types::post_asap::SummaryExpr::KeepPreAsap(_) - ) - )); - assert!(replacements[0].rationale.contains("only realization")); + fn root_inventory_preserves_choices_without_workload_cartesian_expansion() { + let roots = (0..24usize) + .map(|id| { + ( + id, + agg( + vec![2], + default_quantile((id + 1) as f64 / 25.0), + metric_scan(&["job"]), + ), + ) + }) + .collect(); + let space = search_workload(roots); + assert!(space.enumerate_candidate_dags(4096).is_err()); + for id in 0..24 { + let inventory = space.enumerate_candidate_dags_for_root(&id, 4096).unwrap(); + assert!(inventory + .candidates + .iter() + .all(|forest| forest.len() == 1 && forest[0].0 == id)); + let descriptions = inventory + .candidates + .iter() + .map(|forest| format!("{:?}", forest[0].1)) + .collect::>(); + assert!(descriptions.iter().any(|node| node.contains("Kll"))); + assert!(descriptions.iter().any(|node| node.contains("DDSketch"))); + assert!(inventory + .candidates + .iter() + .any(|forest| !forest[0].1.contains_asap())); + } + assert!(space.enumerate_candidate_dags_for_root(&24, 4096).is_err()); + assert!(space.enumerate_candidate_dags_for_root(&0, 0).is_err()); } + // Factoring changes enumeration, not the set of root computations. #[test] - fn exact_mergeable_intent_yields_exactly_one_accumulator_candidate() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); - let target = TargetSubDAG::new(&q); - let replacements = SketchAlgorithmStrategy::default_cost_model().replacements(&target); - assert_eq!(replacements.len(), 1, "{replacements:?}"); - assert!(matches!( - &replacements[0].replacement, - Replacement::Summary(node) if matches!( - node.expr, - asap_types::post_asap::SummaryExpr::SummaryAgg { .. } - ) - )); - } - - /// A custom `CostModel` doesn't change *which* candidates are enumerated - /// (still every `summary_candidates` entry) — only which one - /// `realizations_for_intent` itself would prefer first, and how each - /// candidate's own params are sized. - struct PreferDDSketch; - impl CostModel for PreferDDSketch { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - let mut v = candidates.to_vec(); - if let Some(pos) = v.iter().position(|k| *k == SketchAlgorithm::DDSketch) { - let dd = v.remove(pos); - v.insert(0, dd); + fn root_inventory_matches_projection_of_exhaustive_workload_inventory() { + let roots = (0..2usize) + .map(|id| { + ( + id, + agg( + vec![2], + default_quantile(0.5 + id as f64 * 0.4), + metric_scan(&["job"]), + ), + ) + }) + .collect(); + let space = search_workload(roots); + let full = space.enumerate_candidate_dags(4096).unwrap(); + for id in 0..2 { + let inventory = space.enumerate_candidate_dags_for_root(&id, 4096).unwrap(); + for forest in &full.candidates { + let node = &forest.iter().find(|(root, _)| *root == id).unwrap().1; + assert!(inventory.candidates.iter().any(|one| &one[0].1 == node)); + } + for one in &inventory.candidates { + assert!(full.candidates.iter().any(|forest| forest + .iter() + .any(|(root, node)| *root == id && node == &one[0].1))); } - v } } #[test] - fn custom_cost_model_still_enumerates_every_candidate_not_just_its_own_pick() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let target = TargetSubDAG::new(&q); - let custom = PreferDDSketch; - let replacements = SketchAlgorithmStrategy::new(&custom).replacements(&target); - let kinds: Vec = replacements - .iter() - .map(|r| match &r.replacement { - Replacement::Summary(node) => summary_family_algorithm(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => { - panic!("expected a Summary replacement") - } - }) - .collect(); - assert!(kinds.contains(&SketchAlgorithm::Kll)); - assert!(kinds.contains(&SketchAlgorithm::DDSketch)); - assert_eq!(kinds.len(), 2); + fn inventory_budget_never_returns_a_silent_partial_search() { + let query = agg(vec![2], default_quantile(0.9), metric_scan(&["job"])); + let space = search_workload(vec![(0usize, query)]); + assert!(space.enumerate_candidate_dags(0).is_err()); } - /// Constructing the outer target's candidates never leaks its algorithm - /// choice into the nested aggregate. Existing approximate composition - /// remains governed by the accuracy model, independently of #171's exact - /// value-operation candidates. - #[test] - fn enumerating_the_targets_candidates_does_not_leak_into_a_nested_aggregate() { - // outer: quantile(0.99, ...) over inner: quantile(0.5, m) — both - // Quantile, so both share the [Kll, DDSketch] candidate list. - // - // Rank-over-rank has no registered rule in `DefaultAccuracyModel` - // (issue #172 — see `approximate_over_approximate_is_rejected_by_default`), - // so this test injects `RankAdditiveModel` to admit the composition - // and keep exercising the per-node enumeration property it is about. - let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); - let target = TargetSubDAG::new(&outer); - let replacements = SketchAlgorithmStrategy::new_with_planning_inputs( - &DefaultCostModel, - &RankAdditiveModel, - &EqualSplitAllocator, - ) - .replacements(&target); + fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(left)), + op: asap_types::ir::scalar::CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(right)), + semantics: asap_types::ir::ExprSemantics::Sql, + }) + } - assert_eq!(replacements.len(), 2, "{replacements:?}"); - assert!(replacements - .iter() - .all(|candidate| { matches!(candidate.replacement, Replacement::Summary(_)) })); - // The inner target is still independently enumerated and ranked — - // a custom cost model that prefers DDSketch for it is honored, and - // nothing about the outer target's choice reaches it. - let space = search_workload_with( - vec![("q", Rc::clone(&outer))], - &default_strategies_with(&PreferDDSketchViaCostModel), - ); - let QueryExpr::Aggregate { child, .. } = space.roots[0].1.as_ref() else { - unreachable!() - }; - let inner_group = space - .candidates_for_target(child) - .expect("inner quantile is a target"); - let inner_kinds: Vec = inner_group - .candidates + // Finite samples can overflow a sum although their native average is finite. + #[test] + fn temporal_average_requires_finite_division_guard() { + let root = lower_promql("avg_over_time(a[5m])", AccuracyTarget::Exact); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); + let operator = candidates .iter() - .filter_map(|c| match &c.replacement { - Replacement::Summary(node) => sketch_kind_of(node), + .find_map(|c| match &c.replacement { + Replacement::SubDAG(node) => match &node.operator { + Operator::NonASAP(NonASAPOp::BinaryOp { operator, .. }) => Some(operator), + _ => None, + }, _ => None, }) - .collect(); - assert_eq!( - inner_kinds, - vec![SketchAlgorithm::DDSketch, SketchAlgorithm::Kll], - "the nested inner aggregate keeps its own cost-model-ranked candidates" + .expect("maintained average candidate"); + assert!(operator.checked_finite_division); + assert!( + crate::pass1::rewrite::SemanticEquivalentRewriteStrategy + .replacements(&TargetSubDAG::new(&root)) + .is_empty(), + "an unconditional pre-ASAP rewrite would bypass the runtime guard" ); } - /// The `FieldDataType`'s committed `SketchAlgorithm`, from the top - /// `SummaryAgg` reachable under a (possibly `SummaryEstimate`-wrapped) - /// bound root. - fn summary_family_algorithm(node: &SummaryNode) -> SketchAlgorithm { - match &node.expr { - asap_types::post_asap::SummaryExpr::SummaryEstimate { summary_input, .. } => { - summary_family_algorithm(summary_input) - } - asap_types::post_asap::SummaryExpr::SummaryAgg { family, .. } => match family { - asap_types::post_asap::FieldDataType::Sketch(kind, _) => kind.algorithm().clone(), - other => panic!("expected a Sketch family, got {other:?}"), + // Approximate requests also admit exact temporal ranking candidates. + #[test] + fn approximate_temporal_topk_admits_exact_maintained_values() { + let root = lower_promql( + "topk by(job)(1,count_over_time(a[5m]))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, }, - other => panic!("expected SummaryAgg/SummaryEstimate, got {other:?}"), - } + ); + let planning_inputs = CandidatePlanningInputs::with_default_accuracy(); + let node = exact_topk_over_temporal_values(&root, planning_inputs) + .unwrap() + .expect("exact ranking is legal for an approximate request"); + assert!(node.guarantee.as_ref().unwrap().is_exact()); + crate::test_support::time_and_export(&node).unwrap(); } - // ── SharedSubDAGStrategy ──────────────────────────────────────────── - + // Exact Top-K consumes the Planner's maintained temporal values. #[test] - fn does_not_match_a_single_consumer_target() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); - let target = TargetSubDAG::new(&q); - assert_eq!(target.consumer_count, 1); - assert!(!SharedSubDAGStrategy.matches(&target)); - assert!(SharedSubDAGStrategy.replacements(&target).is_empty()); + fn exact_temporal_topk_has_a_maintained_value_candidate() { + for query in [ + "topk(5, sum_over_time(a[5m]))", + "topk by(job)(5, count_over_time(a[5m]))", + ] { + let root = lower_promql(query, AccuracyTarget::Exact); + let planning_inputs = CandidatePlanningInputs::with_default_accuracy(); + let node = exact_topk_over_temporal_values(&root, planning_inputs) + .unwrap() + .expect("exact Top-K candidate"); + assert!(node.guarantee.as_ref().unwrap().is_exact()); + let Operator::NonASAP(NonASAPOp::Limit { + child: sorted, + n, + offset, + partition_by, + }) = &node.operator + else { + panic!("temporal TopK must compose Sort and Limit"); + }; + assert_eq!((*n, *offset), (Some(5), 0)); + let Operator::NonASAP(NonASAPOp::Sort { + keys, + partition_by: sort_groups, + child: values, + }) = &sorted.operator + else { + panic!("Limit must consume sorted temporal values"); + }; + assert_eq!(sort_groups, partition_by); + assert_eq!( + partition_by.keys().len(), + usize::from(query.contains("by(job)")) + ); + assert_eq!(keys.len(), 1); + assert!(!keys[0].ascending); + assert_eq!(node.schema, values.schema); + crate::test_support::time_and_export(&node).unwrap(); + } } + // A bounded exact mean can share the relative division proof with a quantile. #[test] - fn two_or_more_consumers_yields_the_share_vs_independent_pair() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); - let target = TargetSubDAG::with_consumer_count(&q, 2); - assert!(SharedSubDAGStrategy.matches(&target)); - - let replacements = SharedSubDAGStrategy.replacements(&target); - assert_eq!(replacements.len(), 2, "{replacements:?}"); - - let shared = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, - other => panic!("expected a Rewrite replacement, got {other:?}"), + fn bounded_mean_quantile_ratio_is_certified() { + struct Domain; + impl AccuracyEvidenceProvider for Domain { + fn quantile_input_domain( + &self, + _: &OperatorNode, + ) -> Option { + Some(crate::accuracy::QuantileInputDomain { + lower: 1.0, + upper: 1000.0, + max_samples: 10000, + contract: "finite test population".into(), + }) + } + } + let target = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, }; - assert!( - Rc::ptr_eq(shared, &q), - "the 'build once and share' candidate must be the same Rc as the target" - ); - assert!(replacements[0].rationale.contains("build once and share")); + let inputs = CandidatePlanningInputs { + evidence: &Domain, + ..CandidatePlanningInputs::with_default_accuracy() + }; + for query in [ + "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", + "quantile_over_time(0.5,a[5m]) / avg_over_time(a[5m])", + ] { + let root = lower_promql(query, target.clone()); + let node = realize_binary(&root, inputs, Some(&target)) + .unwrap() + .expect("bounded ratio candidate"); + assert!(DefaultAccuracyModel.satisfies(node.guarantee.as_ref().unwrap(), &target)); + } + } - let independent = match &replacements[1].replacement { - Replacement::Rewrite(rc) => rc, - other => panic!("expected a Rewrite replacement, got {other:?}"), + // Missing domain proof permits an uncertified direct quantile ratio only. + #[test] + fn quantile_ratio_without_input_proof_has_no_root_guarantee() { + let target = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, }; - assert!( - !Rc::ptr_eq(independent, &q), - "the 'build independently' candidate must be a distinct Rc from the target" + let root = lower_promql( + "quantile_over_time(0.5,a[5m]) / quantile_over_time(0.9,a[5m])", + target.clone(), ); - assert_eq!( - **independent, *q, - "the 'build independently' candidate must still be structurally identical" + let planning_inputs = CandidatePlanningInputs::with_default_accuracy(); + let candidate = realize_binary(&root, planning_inputs, Some(&target)) + .unwrap() + .expect("direct quantile ratio candidate"); + assert!(candidate.guarantee.is_none()); + + let other = lower_promql( + "avg_over_time(a[5m]) / quantile_over_time(0.5,a[5m])", + target.clone(), ); - assert!(replacements[1].rationale.contains("build independently")); + assert!(realize_binary(&other, planning_inputs, Some(&target)) + .unwrap() + .is_none()); + } + + fn eps(e: f64) -> AccuracyTarget { + AccuracyTarget::Epsilon(e) + } + + // ── realizations_for_intent / sizing ─────────────────────────────── + + /// The most-preferred `Realization` — `realizations_for_intent(intent, + /// &DefaultCostModel)`'s head — for tests that only care about the + /// default pick, not the full candidate list. + fn preferred(intent: &AggIntent) -> Realization { + realizations_for_intent(intent) + .into_iter() + .next() + .expect("every intent has at least one Realization") } - #[test] - fn three_consumers_are_reported_verbatim_in_both_rationales() { - let q = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); - let target = TargetSubDAG::with_consumer_count(&q, 3); - let replacements = SharedSubDAGStrategy.replacements(&target); - assert!(replacements[0].rationale.contains('3')); - assert!(replacements[1].rationale.contains('3')); + /// Shorthand for asserting the realization *category*. + #[derive(Debug, PartialEq)] + enum Cat { + Sketch(SketchAlgorithm), + Acc(ExactKind), + Pass, } - /// Builds realistic multi-consumer `TargetSubDAG`s the same way this - /// module's own [`discover_targets`]/`walk` does: dedup by `Rc::as_ptr`, - /// walking only the relational-skeleton operator children - /// `asap_types::pre_asap::cse::share_common_sub_dags` itself scopes to, - /// so a shared node nested below another shared node is only ever - /// counted at the highest (maximal) point sharing starts. Test-only: - /// this module deliberately does not ship a workload-wide discovery - /// pass of its own (see the module docs' "Non-goals"). - fn count_consumers(roots: &[Rc]) -> HashMap<*const QueryExpr, usize> { - fn walk(node: &Rc, counts: &mut HashMap<*const QueryExpr, usize>) { - let ptr = Rc::as_ptr(node); - let already_visited = counts.contains_key(&ptr); - *counts.entry(ptr).or_insert(0) += 1; - if !already_visited { - walk_children(node, counts); - } - } - fn walk_children(node: &QueryExpr, counts: &mut HashMap<*const QueryExpr, usize>) { - use QueryExpr::*; - match node { - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => {} - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => walk(c, counts), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | Sort { child, .. } - | Limit { child, .. } => walk(child, counts), - Concat { children, .. } => { - for c in children { - walk_children(c, counts); - } - } - Join { left, right, .. } | SetOp { left, right, .. } => { - walk(left, counts); - walk(right, counts); - } - BinaryOp { lhs, rhs, .. } => { - walk(lhs, counts); - walk(rhs, counts); - } - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => {} + fn cat(intent: &AggIntent) -> Cat { + match preferred(intent) { + Realization::ExactAggregate { kind, .. } => Cat::Acc(kind), + Realization::Sketch(kind) => Cat::Sketch(kind.algorithm().clone()), + Realization::PassThrough => Cat::Pass, + other => { + panic!("this coverage matrix expects only Exact/Sketch/PassThrough, got {other:?}") } } + } - let mut counts = HashMap::new(); - for root in roots { - walk(root, &mut counts); + /// The `AggIntent → SummaryKind` coverage matrix (issue #98): every intent + /// variant maps to a sketch, an exact accumulator, or an explicit + /// pass-through. `realizations_for_intent`'s match is exhaustive, so a + /// new variant cannot compile without a decision; this matrix pins what + /// each decision *is* (its preferred/first candidate). + #[test] + fn agg_intent_to_summary_kind_coverage_matrix() { + use AggIntent as A; + use Cat::*; + use ExactKind as E; + use SketchAlgorithm as K; + let matrix: Vec<(A, Cat)> = vec![ + // approximate-capable, at an ε target → sketch + (default_quantile(0.99), Sketch(K::Kll)), + (default_cardinality(), Sketch(K::Hll)), + ( + A::Cardinality { + cols: vec![0, 1], + accuracy: eps(0.01), + }, + Sketch(K::Hll), + ), + ( + A::Count { + accuracy: eps(0.01), + }, + Sketch(K::Cms), + ), + ( + A::TopK { + k: 10, + accuracy: eps(0.01), + }, + Sketch(K::CmsWithHeap), + ), + // the same intents at Exact → exact realization + ( + A::Quantile { + col: None, + q: 0.5, + accuracy: AccuracyTarget::Exact, + }, + Pass, + ), + ( + A::Cardinality { + cols: vec![], + accuracy: AccuracyTarget::Exact, + }, + Pass, + ), + ( + A::Cardinality { + cols: vec![0, 1], + accuracy: AccuracyTarget::Exact, + }, + Pass, + ), + ( + A::Count { + accuracy: AccuracyTarget::Exact, + }, + Acc(E::Count), + ), + ( + A::TopK { + k: 10, + accuracy: AccuracyTarget::Exact, + }, + Pass, + ), + // exact mergeable accumulators + (A::Sum { col: None }, Acc(E::Sum)), + (A::Min { col: None }, Acc(E::Min)), + (A::Max { col: None }, Acc(E::Max)), + (A::Rate, Acc(E::Rate)), + (A::IRate, Acc(E::IRate)), + (A::Increase, Acc(E::Increase)), + // exact but non-mergeable → pass-through + (A::Avg { col: None }, Pass), + ( + A::StdDev { + col: None, + population: false, + }, + Pass, + ), + ( + A::Variance { + col: None, + population: true, + }, + Pass, + ), + // classic-bucket histogram_quantile is not re-sketchable (#79) + (A::HistogramQuantile { q: 0.99, le: 0 }, Pass), + // counter-derivative / range-vector functions (#44) + (A::Changes, Pass), + (A::Delta, Pass), + (A::IDelta, Pass), + (A::Deriv, Pass), + (A::Resets, Pass), + (A::PredictLinear { seconds: 60.0 }, Pass), + ( + A::DoubleExpSmoothing { + smoothing: 0.5, + trend: 0.5, + }, + Pass, + ), + // native-histogram accessors (#43) + (A::HistogramCount, Pass), + (A::HistogramSum, Pass), + (A::HistogramAvg, Pass), + (A::HistogramStdDev, Pass), + (A::HistogramStdVar, Pass), + ( + A::HistogramFraction { + lower: 0.0, + upper: 1.0, + }, + Pass, + ), + // per-sample transforms (#45, #46) + presence (#47) + (A::Math(MathFunc::Abs), Pass), + (A::TimeFn(TimeFunc::Hour), Pass), + (A::Absent, Pass), + (A::AbsentOverTime, Pass), + (A::PresentOverTime, Pass), + // extended aggregations (#49) + (A::Group, Pass), + (A::CountValues { label: "v".into() }, Pass), + // additional range reducers (#51) + (A::LastOverTime, Pass), + (A::FirstOverTime, Pass), + (A::MadOverTime, Pass), + (A::TsOfMinOverTime, Pass), + (A::TsOfMaxOverTime, Pass), + (A::TsOfFirstOverTime, Pass), + (A::TsOfLastOverTime, Pass), + ]; + for (intent, expected) in &matrix { + assert_eq!(&cat(intent), expected, "realization for {intent:?}"); + } + // Every accumulator pick is mergeable; every sketch pick is on a + // genuinely approximate target (the `agg_is_*` helpers stay truthful). + for (intent, expected) in &matrix { + if let Cat::Acc(_) = expected { + assert!(agg_is_mergeable(intent), "{intent:?}"); + } + if let Cat::Sketch(_) = expected { + assert!( + !agg_is_exact(intent) || matches!(intent, AggIntent::Count { .. }), + "{intent:?} sketches only under an approximate target" + ); + } } - counts } + // Correlation must never acquire a single-input sketch or scalar accumulator. #[test] - fn realistic_cse_output_produces_a_two_consumer_target() { - // Two workload roots that `share_common_sub_dags` collapses onto one - // Rc (mirrors `explanation`'s and `cse`'s own fixtures): a grouped - // Sum aggregate over the same scan, built independently at each root. - let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let shared = asap_types::pre_asap::cse::share_common_sub_dags(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!(Rc::ptr_eq(ra, rb), "fixture sanity: the two roots merged"); - - let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); - let counts = count_consumers(&roots); - let count = counts[&Rc::as_ptr(&roots[0])]; - assert_eq!(count, 2); - - let target = TargetSubDAG::with_consumer_count(&roots[0], count); - assert!(SharedSubDAGStrategy.matches(&target)); - assert_eq!(SharedSubDAGStrategy.replacements(&target).len(), 2); + fn pearson_corr_keeps_exact_paired_input() { + let intent = AggIntent::PearsonCorr { left: 0, right: 1 }; + assert!(matches!( + realizations_for_intent(&intent).as_slice(), + [Realization::PassThrough] + )); + assert!(summary_candidates(&intent).is_empty()); } - // ── search_workload / CandidateLogicalASAPDAGs / TargetSubDAGCandidates (merged from search.rs) ── - // - // Reuses this test module's own `metric_scan`/`agg` fixture helpers - // above (identical to `search.rs`'s own copies, which are dropped here - // to avoid a duplicate-definition collision now that both test modules - // share one file) and `count_consumers` above (which mirrors - // `discover_targets`' own real, non-test traversal for these fixtures). - - // ── discovery + MEMO shape ─────────────────────────────────────────── - #[test] - fn single_bindable_aggregate_keeps_unprovable_hydra_candidates() { - let intent = AggIntent::Count { - accuracy: AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }, + fn accuracy_target_drives_the_boundary() { + // Same intent, three targets → three different decisions. + let exact = AggIntent::Quantile { + col: None, + q: 0.99, + accuracy: AccuracyTarget::Exact, }; - let root = Rc::new(agg(vec![2], intent, metric_scan(&["job"]))); - let space = search_workload(vec![("q", root)]); - - // One group for the Aggregate, one for its Scan child. - assert_eq!(space.len(), 2); + assert_eq!(preferred(&exact), Realization::PassThrough); - let agg_group = space - .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) - .expect("an Aggregate group must be discovered"); - assert_eq!(agg_group.consumer_count, 1); - assert_eq!( - agg_group.candidates.len(), - 6, - "Hydra candidates with unknown evidence remain available: {:?}", - agg_group.candidates - ); - assert!(agg_group - .candidates - .iter() - .all(|c| matches!(c.replacement, Replacement::Summary(_)))); + let approx = default_quantile(0.99); // ε = 0.01 assert_eq!( - agg_group - .candidates - .iter() - .filter(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { - return false; - }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { - return false; - }; - matches!( - &summary_input.expr, - SummaryExpr::SummaryAgg { - grouping: GroupingStrategy::SharedMultiSubpopulation { .. }, - .. - } - ) - }) - .count(), - 2, - "Hydra candidates remain visible with symbolic shared-grid error" + preferred(&approx), + Realization::Sketch(SketchKind::new( + SketchAlgorithm::Kll, + SketchParams::Kll { k: 269 }, + )) ); + + let looser = AggIntent::Quantile { + col: None, + q: 0.99, + accuracy: eps(0.05), + }; assert_eq!( - agg_group - .candidates - .iter() - .filter(|candidate| candidate.has_missing_accuracy_evidence()) - .count(), - 2 + preferred(&looser), + Realization::Sketch(SketchKind::new( + SketchAlgorithm::Kll, + SketchParams::Kll { k: 52 }, + )) ); - let selected = space.global_selection(&DefaultCostModel); - assert!(!selected - .for_target(&space.roots[0].1) - .unwrap() - .chosen - .is_some_and(ReplacementSubDAG::has_missing_accuracy_evidence)); + } - let scan_group = space - .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) - .expect("a Scan group must be discovered"); - assert_eq!(scan_group.consumer_count, 1); - assert!( - scan_group.candidates.is_empty(), - "no strategy matches a bare Scan" + #[test] + fn default_cardinality_sizes_hll_to_its_rse_magnitude() { + assert_eq!( + preferred(&default_cardinality()), + Realization::Sketch(SketchKind::new( + SketchAlgorithm::Hll, + SketchParams::Hll { precision: 14 }, + )) ); } + // Exact counting remains a legal candidate under an approximate target. #[test] - fn cardinality_group_keeps_all_four_candidates() { - let root = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); - let space = search_workload(vec![("q", root)]); - let agg_group = space - .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) - .unwrap(); - assert_eq!(agg_group.candidates.len(), 4); - assert!(agg_group.candidates.iter().any(|candidate| matches!( - &candidate.replacement, - Replacement::Summary(node) if node.guarantee.is_none() - && candidate.has_missing_accuracy_evidence() - ))); - - let root = Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))); - let targeted = search_workload_with_targets( - vec![( - "q", - root, - Some(AccuracyTarget::EpsilonDelta { - epsilon: 0.01, - delta: 0.01, - }), - )], - &default_strategies(), - &DefaultAccuracyModel, - ); - let target = &targeted.roots[0].1; - assert!(targeted - .candidates_for_target(target) - .unwrap() - .candidates + fn approximate_count_includes_exact_accumulator_candidate() { + let intent = AggIntent::Count { + accuracy: eps(0.01), + }; + assert!(realizations_for_intent(&intent) .iter() .any(|candidate| matches!( - &candidate.replacement, - Replacement::Summary(node) if node.guarantee.is_none() - && candidate.has_missing_accuracy_evidence() + candidate, + Realization::ExactAggregate { + kind: ExactKind::Count, + .. + } ))); - assert!(!targeted - .global_selection(&DefaultCostModel) - .for_target(target) - .unwrap() - .chosen - .is_some_and(ReplacementSubDAG::has_missing_accuracy_evidence)); + } - let exact_target = search_workload_with_targets( - vec![( - "q", - Rc::new(agg(vec![2], default_cardinality(), metric_scan(&["job"]))), - Some(AccuracyTarget::Exact), - )], - &default_strategies(), - &DefaultAccuracyModel, + #[test] + fn epsilon_delta_sizes_cms_depth() { + let intent = AggIntent::Count { + accuracy: AccuracyTarget::EpsilonDelta { + epsilon: 0.001, + delta: 0.001, + }, + }; + assert_eq!( + preferred(&intent), + Realization::Sketch(SketchKind::new( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 2719, + depth: 7 + }, // ⌈e/0.001⌉, ⌈ln 1000⌉ + )) + ); + // Epsilon-only falls back to DEFAULT_DELTA → depth 5. + let intent = AggIntent::Count { + accuracy: eps(0.001), + }; + assert_eq!( + preferred(&intent), + Realization::Sketch(SketchKind::new( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 2719, + depth: 5 + }, + )) ); - assert!(exact_target - .candidates_for_target(&exact_target.roots[0].1) - .unwrap() - .candidates - .iter() - .all(|candidate| !candidate.has_missing_accuracy_evidence())); } #[test] - fn shared_aggregate_across_two_roots_gets_both_strategies_candidates() { - // Two independently-built, structurally identical Sum aggregates: - // share_common_sub_dags (run inside search_workload) collapses them - // onto one Rc with consumer_count 2, so this single group should - // carry SketchAlgorithmStrategy's one ExactAggregate candidate *and* - // SharedSubDAGStrategy's share-vs-recompute pair. - let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); - - // roots[0] and roots[1] must have merged onto the same Rc. - assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); + fn topk_heap_capacity_respects_accuracy_and_output_count() { + let intent = AggIntent::TopK { + k: 25, + accuracy: eps(0.01), + }; + match preferred(&intent) { + Realization::Sketch(kind) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap => { + let SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } = kind.params() + else { + unreachable!("SketchKind validates CmsWithHeap params") + }; + assert_eq!(*heap_size, 100); + assert_eq!(*width, 272); // ⌈e/0.01⌉ + assert_eq!(*depth, 5); + } + other => panic!("expected CmsWithHeap, got {other:?}"), + } + } - let group = space.candidates_for_target(&space.roots[0].1).unwrap(); - assert_eq!(group.consumer_count, 2); + #[test] + fn candidate_lists_match_the_issue_map() { assert_eq!( - group.candidates.len(), - 3, - "1 ExactAggregate Summary + 2 Rewrite (share/recompute): {:?}", - group.candidates + summary_candidates(&default_quantile(0.5)), + &[SketchAlgorithm::Kll, SketchAlgorithm::DDSketch] ); - - let summary_count = group - .candidates - .iter() - .filter(|c| matches!(c.replacement, Replacement::Summary(_))) - .count(); - let rewrite_count = group - .candidates - .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) - .count(); - assert_eq!(summary_count, 1); - assert_eq!(rewrite_count, 2); - - // The two Rewrite candidates must NOT have collapsed into one - // (the "false-positive dedup" failure mode `is_duplicate_rewrite` - // exists to prevent). - let one_is_the_target = group.candidates.iter().any( - |c| matches!(&c.replacement, Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target)), + assert_eq!( + summary_candidates(&default_cardinality()), + &[ + SketchAlgorithm::Hll, + SketchAlgorithm::Theta, + SketchAlgorithm::Kmv, + SketchAlgorithm::UnivMon + ] ); - let one_is_not = group.candidates.iter().any(|c| { - matches!(&c.replacement, Replacement::Rewrite(rc) if !Rc::ptr_eq(rc, &group.target)) - }); - assert!(one_is_the_target && one_is_not); + assert_eq!( + summary_candidates(&AggIntent::TopK { + k: 5, + accuracy: eps(0.01) + }), + &[ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap + ] + ); + assert_eq!( + summary_candidates(&AggIntent::Count { + accuracy: eps(0.01) + }), + &[ + SketchAlgorithm::Cms, + SketchAlgorithm::CountSketch, + SketchAlgorithm::UnivMon + ] + ); + assert!(summary_candidates(&AggIntent::Rate).is_empty()); + } + + #[test] + fn realizations_for_intent_enumerates_every_candidate_ranked() { + // Quantile's candidate list is [Kll, DDSketch] — realizations_for_intent + // must return both, ranked with the DefaultCostModel's preferred + // (Kll) first. + let kinds: Vec = realizations_for_intent(&default_quantile(0.99)) + .into_iter() + .map(|realization| match realization { + Realization::Sketch(kind) => kind.algorithm().clone(), + other => panic!("expected Sketch, got {other:?}"), + }) + .collect(); + assert_eq!(kinds, vec![SketchAlgorithm::Kll, SketchAlgorithm::DDSketch]); } #[test] - fn nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered() { - // A shared grouped Aggregate nested under two *different*, - // unshared Filter parents — real consumer_count must come from - // walking the whole DAG, not just root-level pointer identity - // (a naive whole-root-only consumer-count pass would miss this; - // this module's discover_targets must not). - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - - let shared = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); - // Different predicates so the two Filter *parents* stay distinct - // (don't themselves merge under CSE) — only their shared `child` - // should collapse onto one `Rc`. - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::clone(&shared), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::clone(&shared), + fn degenerate_epsilon_saturates_to_tightest_params() { + let intent = AggIntent::Quantile { + col: None, + q: 0.99, + accuracy: eps(0.0), }; - - let space = search_workload(vec![("a", Rc::new(root_a)), ("b", Rc::new(root_b))]); assert_eq!( - space.len(), - 4, - "2 distinct Filters + 1 shared Aggregate + 1 shared Scan" - ); - - // `share_common_sub_dags` re-clones+re-interns anything that already - // had more than one owner going in (see `cse.rs`'s own doc on - // `intern_child`'s clone-fallback path) — so the post-CSE shared - // node is a *fresh* Rc, structurally equal to (but not the same - // pointer as) the pre-search `shared` variable. Recover it from the - // post-CSE root's own `child` field instead of the stale `shared` - // handle. - let QueryExpr::Filter { - child: post_cse_shared_a, - .. - } = space.roots[0].1.as_ref() - else { - panic!("expected a Filter root"); - }; - let QueryExpr::Filter { - child: post_cse_shared_b, - .. - } = space.roots[1].1.as_ref() - else { - panic!("expected a Filter root"); - }; - assert!( - Rc::ptr_eq(post_cse_shared_a, post_cse_shared_b), - "fixture sanity: the two Filters' children must still merge" - ); - let post_cse_shared = post_cse_shared_a; - let group = space - .candidates_for_target(post_cse_shared) - .expect("shared node must be a discovered target"); - assert_eq!(group.consumer_count, 2); - assert!( - SharedSubDAGStrategy.matches(&TargetSubDAG::with_consumer_count( - post_cse_shared, - group.consumer_count + preferred(&intent), + Realization::Sketch(SketchKind::new( + SketchAlgorithm::Kll, + SketchParams::Kll { k: 65_535 }, )) ); } - // ── dedup ──────────────────────────────────────────────────────────── + // ── posterior_aware_size_params (issue #239, integration point 2) ────── + + fn count_intent(e: f64) -> AggIntent { + AggIntent::Count { accuracy: eps(e) } + } #[test] - fn add_candidate_rejects_a_true_rewrite_duplicate() { - // SharedSubDAGStrategy's `Replacement::Rewrite` candidates are - // real `QueryExpr` values with `PartialEq`, so `add_candidate` can - // (and must) actually reject a genuine repeat — unlike the - // `Replacement::Summary` case (see the test below). - let root = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); - let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 2); - let target = TargetSubDAG::with_consumer_count(&root, 2); - let mut inserted = 0; - for candidate in SharedSubDAGStrategy.replacements(&target) { - if group.add_candidate(candidate) { - inserted += 1; + fn posterior_aware_sizing_shrinks_width_under_stated_assumption() { + let intent = count_intent(0.01); + let worst_case = default_size_params(SketchAlgorithm::Cms, &intent, 0.01, 0.01); + let relaxed = posterior_aware_size_params( + SketchAlgorithm::Cms, + &intent, + 0.01, + 0.01, + ExpectedCaseSizing { + width_relaxation: 0.5, + }, + ); + match (worst_case, relaxed) { + ( + SketchParams::Cms { + width: w0, + depth: d0, + }, + SketchParams::Cms { + width: w1, + depth: d1, + }, + ) => { + assert!( + w1 < w0, + "expected relaxed width {w1} to be strictly smaller than worst-case {w0}" + ); + assert_eq!(d0, d1, "depth must be unaffected by width_relaxation"); } + other => panic!("expected Cms/Cms pair, got {other:?}"), } - assert_eq!(inserted, 2, "share + recompute-independently candidates"); + } - // Re-adding the identical candidate list must add nothing new: the - // "share" candidate is literally the same Rc as before, and the - // "recompute independently" candidate is a fresh Rc but - // structurally identical value, both already covered by - // `is_duplicate_rewrite`. - let mut re_inserted = 0; - for candidate in SharedSubDAGStrategy.replacements(&target) { - if group.add_candidate(candidate) { - re_inserted += 1; - } - } - assert_eq!( - re_inserted, 0, - "re-proposing the same Rewrite candidates must not grow the group" + #[test] + fn posterior_aware_sizing_at_full_relaxation_matches_worst_case() { + // width_relaxation = 1.0 must reproduce default_size_params exactly + // — the "no risk taken" boundary. + let intent = count_intent(0.01); + let worst_case = default_size_params(SketchAlgorithm::Cms, &intent, 0.01, 0.01); + let relaxed = posterior_aware_size_params( + SketchAlgorithm::Cms, + &intent, + 0.01, + 0.01, + ExpectedCaseSizing { + width_relaxation: 1.0, + }, ); - assert_eq!(group.candidates.len(), 2); + assert_eq!(worst_case, relaxed); } #[test] - fn add_candidate_never_dedups_summary_candidates() { - // Documented, deliberate consequence of `SummaryNode` deriving no - // `PartialEq` (see `is_duplicate_summary`'s own doc): re-proposing - // the same `Replacement::Summary` candidates DOES grow the group — - // this module refuses to guess at an equality check it can't back - // with a real `PartialEq`. `search_workload_with` never actually - // does this in practice (every target is asked exactly once — see the - // module docs' "Termination" section), so this test exists to pin - // the documented behavior, not to endorse calling `replacements` - // twice for the same target. - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 1); - let strategy = SketchAlgorithmStrategy::default_cost_model(); - let target = TargetSubDAG::new(&root); - for candidate in strategy.replacements(&target) { - group.add_candidate(candidate); + fn posterior_aware_sizing_invalid_relaxation_falls_back_to_worst_case() { + let intent = count_intent(0.01); + let worst_case = default_size_params(SketchAlgorithm::Cms, &intent, 0.01, 0.01); + for bad in [0.0, -0.5, 1.5, f64::NAN, f64::INFINITY] { + let relaxed = posterior_aware_size_params( + SketchAlgorithm::Cms, + &intent, + 0.01, + 0.01, + ExpectedCaseSizing { + width_relaxation: bad, + }, + ); + assert_eq!( + worst_case, relaxed, + "width_relaxation={bad} should fall back to the worst-case width" + ); } - assert_eq!(group.candidates.len(), 2); + } - for candidate in strategy.replacements(&target) { - group.add_candidate(candidate); + #[test] + fn posterior_aware_sizing_does_not_apply_cms_l1_relaxation_to_count_sketch() { + let cms_heap_intent = AggIntent::TopK { + k: 7, + accuracy: eps(0.01), + }; + let assumption = ExpectedCaseSizing { + width_relaxation: 0.25, + }; + // CountSketch + assert_eq!( + posterior_aware_size_params( + SketchAlgorithm::CountSketch, + &count_intent(0.01), + 0.01, + 0.01, + assumption + ), + default_size_params( + SketchAlgorithm::CountSketch, + &count_intent(0.01), + 0.01, + 0.01 + ), + ); + // CmsWithHeap / CountSketchWithHeap carry k through untouched. + match posterior_aware_size_params( + SketchAlgorithm::CmsWithHeap, + &cms_heap_intent, + 0.01, + 0.01, + assumption, + ) { + SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } => { + assert_eq!(width, 68); + assert_eq!(depth, 5); + assert_eq!(heap_size, 100); + } + other => panic!("expected CmsWithHeap, got {other:?}"), } + } + + #[test] + fn posterior_aware_sizing_leaves_non_cms_kinds_unchanged() { + // Kll/Hll/etc. have no width_relaxation concept — must be byte-for- + // byte identical to default_size_params. + let intent = default_quantile(0.99); + let assumption = ExpectedCaseSizing { + width_relaxation: 0.1, + }; assert_eq!( - group.candidates.len(), - 4, - "Summary candidates are never deduped by this module — see is_duplicate_summary" + posterior_aware_size_params(SketchAlgorithm::Kll, &intent, 0.01, 0.01, assumption), + default_size_params(SketchAlgorithm::Kll, &intent, 0.01, 0.01), ); } #[test] - fn is_duplicate_rewrite_never_merges_share_with_recompute() { - let target = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, - metric_scan(&["job"]), - )); - let share = Rc::clone(&target); - let recompute = Rc::new((*target).clone()); - assert!(!Rc::ptr_eq(&share, &recompute)); + fn default_size_params_unchanged_by_new_function_existing() { + // Regression pin: default_size_params's own worst-case behavior for + // existing callers must be untouched by adding + // posterior_aware_size_params alongside it. assert_eq!( - *share, *recompute, - "fixture sanity: same value, different Rc" + default_size_params(SketchAlgorithm::Cms, &count_intent(0.001), 0.001, 0.001), + SketchParams::Cms { + width: 2719, + depth: 7 + }, ); - assert!(!is_duplicate_rewrite(&share, &recompute, &target)); - assert!(!is_duplicate_rewrite(&recompute, &share, &target)); } + // ── ASAPStrategies / SharedSubDAGStrategy fixtures ─────────── + + // ── ASAPStrategies ───────────────────────────────────────────── + #[test] - fn is_duplicate_rewrite_catches_a_real_repeat() { - let target = Rc::new(agg( - vec![2], - AggIntent::Sum { col: None }, + fn matches_a_bindable_aggregate() { + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let target = TargetSubDAG::new(&q); + assert!(ASAPStrategies::default().matches(&target)); + } + + #[test] + fn does_not_match_a_multi_intent_or_having_aggregate() { + let strategy = ASAPStrategies::default(); + + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: ReductionTy::by(vec![2]), + measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); + let target = TargetSubDAG::new(&multi); + assert!(!strategy.matches(&target)); + assert!(strategy.replacements(&target).is_empty()); + + let having_q = crate::test_support::aggregate( + ReductionTy::by(vec![2]), + vec![default_quantile(0.99)], + vec![], + Some(asap_types::ir::Predicate(ScalarExpr::Literal( + asap_types::ir::scalar::ScalarValue::Boolean(true), + ))), metric_scan(&["job"]), - )); - let first_recompute = Rc::new((*target).clone()); - let second_recompute = Rc::new((*target).clone()); - assert!(!Rc::ptr_eq(&first_recompute, &second_recompute)); - assert!(is_duplicate_rewrite( - &first_recompute, - &second_recompute, - &target - )); + ); + let target = TargetSubDAG::new(&having_q); + assert!(!strategy.matches(&target)); + assert!(strategy.replacements(&target).is_empty()); } - // ── cost-based ranking ─────────────────────────────────────────────── + #[test] + fn does_not_match_a_non_aggregate_node() { + let scan = metric_scan(&["job"]); + let target = TargetSubDAG::new(&scan); + assert!(!ASAPStrategies::default().matches(&target)); + assert!(ASAPStrategies::default().replacements(&target).is_empty()); + } #[test] - fn cost_sorted_orders_shared_sub_dag_candidates_by_cse_share_decision() { - // Many consumers of a cheap-to-recompute, cheap-to-maintain exact - // accumulator: cse_share_decision should prefer Share (see - // cost_model.rs's own `cse_share_decision_shares_when_recompute_dominates_maintenance`). - let mut roots = Vec::new(); - let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - for i in 0..20 { - roots.push((i, Rc::new(shared.clone()))); - } - let space = search_workload(roots); - let group = space.candidates_for_target(&space.roots[0].1).unwrap(); - assert_eq!(group.consumer_count, 20); + fn approximate_quantile_enumerates_every_summary_candidate() { + // Quantile's candidate list is [Kll, DDSketch] (summary_candidates) — + // every entry must come back as its own bound summary candidate, + // not just Kll (the CostModel-ranked head realizations_for_intent commits to). + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let target = TargetSubDAG::new(&q); + let replacements = ASAPStrategies::default().replacements(&target); + assert_eq!( + replacements.len(), + 2, + "expected 2 candidates, got {replacements:?}" + ); - let ranked = space.cost_sorted(&DefaultCostModel); - let ranked_group = ranked - .iter() - .find(|g| Rc::ptr_eq(g.target, &space.roots[0].1)) - .unwrap(); - assert!(matches!( - &ranked_group.candidates[0].replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, &group.target) - )); - let rewrites: Vec<&ReplacementSubDAG> = ranked_group - .candidates + let kinds: Vec = replacements .iter() - .filter(|c| matches!(c.replacement, Replacement::Rewrite(_))) - .copied() + .map(|r| match &r.replacement { + Replacement::SubDAG(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { + panic!("expected a Summary replacement") + } + }) .collect(); - assert_eq!(rewrites.len(), 2); - let first_shares_target = match &rewrites[0].replacement { - Replacement::Rewrite(rc) => Rc::ptr_eq(rc, &group.target), - Replacement::Summary(_) | Replacement::ExactComposition(_) => false, - }; + assert!(kinds.contains(&SketchAlgorithm::Kll), "{kinds:?}"); + assert!(kinds.contains(&SketchAlgorithm::DDSketch), "{kinds:?}"); assert!( - first_shares_target, - "with 20 cheap consumers, Share should rank first: {rewrites:?}" + replacements.iter().all(|r| !r.rationale.is_empty()), + "every candidate must carry a rationale" ); } #[test] - fn cost_sorted_orders_sketch_candidates_by_rank_candidates() { - struct PreferDDSketch; - impl CostModel for PreferDDSketch { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - let mut v = candidates.to_vec(); - if let Some(pos) = v.iter().position(|k| *k == SketchAlgorithm::DDSketch) { - let dd = v.remove(pos); - v.insert(0, dd); - } - v - } - } - - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let space = search_workload(vec![("q", root)]); - let ranked = space.cost_sorted(&PreferDDSketch); - let agg_group = ranked + fn cardinality_epsilon_delta_keeps_unknown_accuracy_candidates() { + let q = agg(vec![2], default_cardinality(), metric_scan(&["job"])); + let target = TargetSubDAG::new(&q); + let replacements = ASAPStrategies::default().replacements(&target); + let kinds: Vec = replacements .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) - .unwrap(); - assert_eq!(agg_group.candidates.len(), 2); - let first_kind = match &agg_group.candidates[0].replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, - }; - assert_eq!(first_kind, Some(SketchAlgorithm::DDSketch)); - } - - #[test] - fn grouping_cost_cannot_resurrect_unprovable_hydra_candidates() { - struct EstimatedSubpopulations(usize); - - impl CostModel for EstimatedSubpopulations { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { - Some(self.0) - } - } + .map(|r| match &r.replacement { + Replacement::SubDAG(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { + panic!("expected a Summary replacement") + } + }) + .collect(); + assert_eq!( + kinds, + vec![ + SketchAlgorithm::Hll, + SketchAlgorithm::Theta, + SketchAlgorithm::Kmv, + SketchAlgorithm::UnivMon, + ] + ); - fn first_grouping(estimated_count: usize) -> GroupingStrategy { - let model = EstimatedSubpopulations(estimated_count); - let intent = AggIntent::Count { + let q = agg( + vec![2], + AggIntent::Cardinality { + cols: vec![], accuracy: AccuracyTarget::EpsilonDelta { epsilon: 0.01, delta: 0.01, }, - }; - let root = Rc::new(agg( - vec![2, 3], - intent, - metric_scan(&["tenant_id", "endpoint"]), - )); - let strategies = default_strategies_with(&model); - let space = search_workload_with(vec![("tenant_endpoint_count", root)], &strategies); - let ranked = space.cost_sorted(&model); - let aggregate = ranked - .iter() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) - .expect("aggregate group"); - let Replacement::Summary(node) = &aggregate.candidates[0].replacement else { - panic!("grouping candidate must be a summary") - }; - summary_grouping(node) - .expect("bound summary grouping") - .clone() - } - - assert_eq!( - first_grouping(10_000), - GroupingStrategy::PerSubpopulationInstance + }, + metric_scan(&["job"]), ); + let kinds: Vec<_> = ASAPStrategies::default() + .replacements(&TargetSubDAG::new(&q)) + .iter() + .map(|r| match &r.replacement { + Replacement::SubDAG(node) => summary_family_algorithm(node), + Replacement::ExactComposition(_) => { + panic!("expected a Summary replacement") + } + }) + .collect(); assert_eq!( - first_grouping(10), - GroupingStrategy::PerSubpopulationInstance + kinds, + vec![ + SketchAlgorithm::Hll, + SketchAlgorithm::Theta, + SketchAlgorithm::Kmv, + SketchAlgorithm::UnivMon, + ] ); } - /// [`RankedTargetSubDAGCandidates::costs`] is a per-candidate annotation, aligned - /// index-for-index with `candidates` — each entry must equal what - /// calling [`CostModel::estimate_cost`] directly on that same candidate - /// and target produces, not some other (or stale) number. #[test] - fn cost_sorted_pairs_each_candidate_with_its_own_estimate_cost() { - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let space = search_workload(vec![("q", root)]); - let ranked = space.cost_sorted(&DefaultCostModel); - let agg_group = ranked - .iter() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) - .unwrap(); - assert_eq!( - agg_group.costs.len(), - agg_group.candidates.len(), - "costs must be aligned 1:1 with candidates" - ); - assert!(!agg_group.costs.is_empty()); - - let target = TargetSubDAG::with_consumer_count(agg_group.target, agg_group.consumer_count); - for (candidate, &cost) in agg_group.candidates.iter().zip(&agg_group.costs) { - assert_eq!( - cost, - DefaultCostModel.estimate_cost(candidate, &target), - "RankedTargetSubDAGCandidates::costs must match calling CostModel::estimate_cost directly \ - for the same candidate/target" - ); - } + fn exact_accuracy_target_yields_exactly_one_pass_through_candidate() { + // Exact quantile has no sketch candidate at all — realizations_for_intent + // produces PassThrough, the only option, so exactly one candidate. + let intent = AggIntent::Quantile { + col: None, + q: 0.99, + accuracy: AccuracyTarget::Exact, + }; + let q = agg(vec![2], intent, metric_scan(&["job"])); + let target = TargetSubDAG::new(&q); + let replacements = ASAPStrategies::default().replacements(&target); + assert_eq!(replacements.len(), 1, "{replacements:?}"); + assert!(matches!( + &replacements[0].replacement, + Replacement::SubDAG(node) if !node.contains_asap() + )); + assert!(replacements[0].rationale.contains("only realization")); } - // ── global_selection (issue #271) ─────────────────────────────────── - - /// A `CostModel` with a constant, `sub_dag`-independent recompute cost - /// and shared-maintenance cost, chosen (40 recompute-per-use, 100 - /// maintenance) so that a `SharedSubDAGStrategy` group's - /// `cse_share_decision` flips exactly between a consumer count of 2 - /// (recompute total 80, below maintenance: `RecomputeIndependently`) - /// and a consumer count of 3 (recompute total 120, above - /// maintenance: `Share`) — the precise threshold - /// `effective_consumer_count_corrects_a_nested_groups_share_decision` - /// needs to cross. - struct ConstantCseCost; - impl CostModel for ConstantCseCost { - fn allow_uncosted_legacy_selection(&self) -> bool { - true - } - - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - fn cse_recompute_cost(&self, _candidate: &CseCandidate) -> Cost { - Cost(40.0) - } - fn cse_shared_maintenance_cost(&self, _candidate: &CseCandidate) -> Cost { - Cost(100.0) - } + #[test] + fn exact_mergeable_intent_yields_exactly_one_accumulator_candidate() { + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let target = TargetSubDAG::new(&q); + let replacements = ASAPStrategies::default().replacements(&target); + assert_eq!(replacements.len(), 1, "{replacements:?}"); + assert!(matches!( + &replacements[0].replacement, + Replacement::SubDAG(node) if matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { .. }) + ) + )); } - /// A costed logical choice must not panic when an explicitly allowed CSE - /// choice has no numeric cost. + /// Constructing the outer target's candidates never leaks its algorithm + /// choice into the nested aggregate. Existing approximate composition + /// remains governed by the accuracy model, independently of #171's exact + /// value-operation candidates. #[test] - fn costed_logical_candidate_beats_uncosted_legacy_cse_choice() { - struct MixedCost; - impl CostModel for MixedCost { - fn allow_uncosted_legacy_selection(&self) -> bool { - true - } + fn enumerating_the_targets_candidates_does_not_leak_into_a_nested_aggregate() { + // outer: quantile(0.99, ...) over inner: quantile(0.5, m) — both + // Quantile, so both share the [Kll, DDSketch] candidate list. + // + // Rank-over-rank has no registered rule in `DefaultAccuracyModel` + // (issue #172 — see `approximate_over_approximate_is_rejected_by_default`), + // so this test injects `RankAdditiveModel` to admit the composition + // and keep exercising the per-node enumeration property it is about. + let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["job"])); + let outer = agg(vec![], default_quantile(0.99), inner); + let target = TargetSubDAG::new(&outer); + let replacements = + ASAPStrategies::new_with_planning_inputs(&RankAdditiveModel, &EqualSplitAllocator) + .replacements(&target); - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } + assert_eq!(replacements.len(), 2, "{replacements:?}"); + assert!(replacements.iter().all(|candidate| { + matches!(&candidate.replacement, Replacement::SubDAG(n) if n.contains_asap()) + })); + // The inner target is still independently enumerated, and nothing + // about the outer target's choice reaches it. + let space = search_workload(vec![("q", Rc::clone(&outer))]); + let Some(NonASAPOp::Aggregate { child, .. }) = space.roots[0].1.non_asap() else { + unreachable!() + }; + let inner_group = space + .candidates_for_target(child) + .expect("inner quantile is a target"); + let inner_kinds: Vec = inner_group + .candidates + .iter() + .filter_map(|c| match &c.replacement { + Replacement::SubDAG(node) if node.contains_asap() => { + Some(summary_family_algorithm(node)) + } + _ => None, + }) + .collect(); + assert_eq!( + inner_kinds, + vec![SketchAlgorithm::Kll, SketchAlgorithm::DDSketch], + "the nested inner aggregate keeps its own candidates" + ); + } - fn candidate_cost( - &self, - candidate: &ReplacementSubDAG, - _target: &TargetSubDAG<'_>, - ) -> Option { - (!is_cse_candidate(candidate)).then_some(Cost(1.0)) + /// The `FieldDataType`'s committed `SketchAlgorithm`, from the top + /// `SummaryAgg` reachable under a (possibly `SummaryEstimate`-wrapped) + /// bound root. + fn summary_family_algorithm(node: &OperatorNode) -> SketchAlgorithm { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + summary_family_algorithm(summary_input) } + Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) => match family { + asap_types::ir::schema::FieldDataType::Sketch(kind, _) => kind.algorithm().clone(), + other => panic!("expected a Sketch family, got {other:?}"), + }, + other => panic!("expected SummaryAgg/SummaryEstimate, got {other:?}"), } + } - let aggregate = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let space = search_workload(vec![("left", Rc::clone(&aggregate)), ("right", aggregate)]); - let root = &space.roots[0].1; - assert!(cse_candidate_pair(space.candidates_for_target(root).unwrap()).is_some()); - let selected = space.global_selection(&MixedCost); - let chosen = selected.for_target(root).unwrap().chosen.unwrap(); - assert!(!is_cse_candidate(chosen)); + // ── SharedSubDAGStrategy ──────────────────────────────────────────── + + #[test] + fn does_not_match_a_single_consumer_target() { + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let target = TargetSubDAG::new(&q); + assert_eq!(target.consumer_count, 1); + assert!(!SharedSubDAGStrategy.matches(&target)); + assert!(SharedSubDAGStrategy.replacements(&target).is_empty()); } #[test] - fn global_selection_matches_cost_sorted_for_a_non_interacting_workload() { - // No nested sharing at all — global_selection's effective_consumer_count - // must equal the group's own raw consumer_count, and its `chosen` - // candidate must be cost_sorted's top pick, for both the sketch - // group and its child Scan. - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let space = search_workload(vec![("q", root)]); + fn two_or_more_consumers_yields_the_share_vs_independent_pair() { + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let target = TargetSubDAG::with_consumer_count(&q, 2); + assert!(SharedSubDAGStrategy.matches(&target)); - let ranked = space.cost_sorted(&DefaultCostModel); - let selected = space.global_selection(&DefaultCostModel); - assert_eq!(ranked.len(), selected.target_selections().count()); + let replacements = SharedSubDAGStrategy.replacements(&target); + assert_eq!(replacements.len(), 2, "{replacements:?}"); - for ranked_group in &ranked { - let selected_group = selected.for_target(ranked_group.target).unwrap(); - assert_eq!( - selected_group.effective_consumer_count, ranked_group.consumer_count, - "no ancestor is ever RecomputeIndependently here, so effective must equal raw" - ); - assert_eq!( - selected_group.chosen.map(|c| &c.rationale), - ranked_group.candidates.first().map(|c| &c.rationale), - "with no cross-group interaction, global_selection's pick must match \ - cost_sorted's top-ranked candidate" - ); - } + let shared = match &replacements[0].replacement { + Replacement::SubDAG(rc) => rc, + other => panic!("expected a Rewrite replacement, got {other:?}"), + }; + assert!( + Rc::ptr_eq(shared, &q), + "the 'build once and share' candidate must be the same Rc as the target" + ); + assert!(replacements[0].rationale.contains("build once and share")); + + let independent = match &replacements[1].replacement { + Replacement::SubDAG(rc) => rc, + other => panic!("expected a Rewrite replacement, got {other:?}"), + }; + assert!( + !Rc::ptr_eq(independent, &q), + "the 'build independently' candidate must be a distinct Rc from the target" + ); + assert_eq!( + **independent, *q, + "the 'build independently' candidate must still be structurally identical" + ); + assert!(replacements[1].rationale.contains("build independently")); } #[test] - fn global_selection_leaves_an_unmatched_group_as_none() { - // A bare Scan: no registered strategy has an opinion on it, so it - // gets a group with an empty candidate list (see TargetSubDAGCandidates's own - // doc) — global_selection must not invent a candidate for it. - let root = Rc::new(metric_scan(&["job"])); - let space = search_workload(vec![("q", root)]); - let selected = space.global_selection(&DefaultCostModel); - let scan_group = selected - .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Scan { .. })) - .unwrap(); - assert!(scan_group.chosen.is_none()); - assert_eq!(scan_group.effective_consumer_count, 1); + fn three_consumers_are_reported_verbatim_in_both_rationales() { + let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let target = TargetSubDAG::with_consumer_count(&q, 3); + let replacements = SharedSubDAGStrategy.replacements(&target); + assert!(replacements[0].rationale.contains('3')); + assert!(replacements[1].rationale.contains('3')); } - #[test] - fn global_selection_falls_back_to_local_ranking_for_sketch_family_groups() { - // SketchAlgorithmStrategy groups have no cross-group-aware cost hook - // (rank_candidates takes no consumer_count) — global_selection must - // still return cost_sorted's own top pick for them (documented in - // the module docs' "Whole-plan (cross-group) selection" section), - // not silently drop the candidate or fall back to discovery order. - struct PreferDDSketch; - impl CostModel for PreferDDSketch { - fn allow_uncosted_legacy_selection(&self) -> bool { - true + /// Builds realistic multi-consumer `TargetSubDAG`s the same way this + /// module's own [`discover_targets`]/`walk` does: dedup by `Rc::as_ptr`, + /// walking only the relational-skeleton operator children + /// `asap_types::ir::cse::share_common_sub_dags` itself scopes to, + /// so a shared node nested below another shared node is only ever + /// counted at the highest (maximal) point sharing starts. Test-only: + /// this module deliberately does not ship a workload-wide discovery + /// pass of its own (see the module docs' "Non-goals"). + fn count_consumers(roots: &[Rc]) -> HashMap<*const OperatorNode, usize> { + fn walk(node: &Rc, counts: &mut HashMap<*const OperatorNode, usize>) { + let ptr = Rc::as_ptr(node); + let already_visited = counts.contains_key(&ptr); + *counts.entry(ptr).or_insert(0) += 1; + if !already_visited { + walk_children(node, counts); } - - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - let mut v = candidates.to_vec(); - if let Some(pos) = v.iter().position(|k| *k == SketchAlgorithm::DDSketch) { - let dd = v.remove(pos); - v.insert(0, dd); + } + fn walk_children(node: &OperatorNode, counts: &mut HashMap<*const OperatorNode, usize>) { + if let Some(NonASAPOp::Concat { children, .. }) = node.non_asap() { + for c in children { + walk_children(c, counts); } - v + return; + } + for child in node.children() { + walk(child, counts); } } - let root = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); - let space = search_workload(vec![("q", root)]); - let selected = space.global_selection(&PreferDDSketch); - let agg_group = selected - .target_selections() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) - .unwrap(); - let kind = match &agg_group.chosen.unwrap().replacement { - Replacement::Summary(node) => sketch_kind_of(node), - Replacement::Rewrite(_) | Replacement::ExactComposition(_) => None, - }; - assert_eq!(kind, Some(SketchAlgorithm::DDSketch)); - - struct Uncosted; - impl CostModel for Uncosted { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } + let mut counts = HashMap::new(); + for root in roots { + walk(root, &mut counts); } - assert!(space - .global_selection(&Uncosted) - .for_target(&space.roots[0].1) - .unwrap() - .chosen - .is_none()); + counts } #[test] - fn mixed_rewrite_group_keeps_and_selects_its_explicit_cse_pair() { - let target = Rc::new(metric_scan(&["job"])); - let mut group = TargetSubDAGCandidates::new(Rc::clone(&target), 2); - group.candidates = vec![ - ReplacementSubDAG { - strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::clone(&target)), - provenance: ReplacementProvenance::CseShare, - rationale: "share".into(), - }, - ReplacementSubDAG { - strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new(target.as_ref().clone())), - provenance: ReplacementProvenance::CseRecompute, - rationale: "recompute".into(), - }, - ReplacementSubDAG { - strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::CurrentTimestamp)), - provenance: ReplacementProvenance::LogicalRewrite, - rationale: "different rewrite strategy".into(), - }, - ]; + fn realistic_cse_output_produces_a_two_consumer_target() { + // Two workload roots that `share_common_sub_dags` collapses onto one + // Rc (mirrors `explanation`'s and `cse`'s own fixtures): a grouped + // Sum aggregate over the same scan, built independently at each root. + let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let shared = asap_types::ir::cse::share_common_sub_dags(vec![("a", a), ("b", b)]); + let [(_, ra), (_, rb)] = shared.as_slice() else { + panic!("expected 2 roots"); + }; + assert!(Rc::ptr_eq(ra, rb), "fixture sanity: the two roots merged"); - assert!(cse_candidate_pair(&group).is_some()); - let ranked = rank_group(&group, &ConstantCseCost); - assert_eq!( - ranked - .iter() - .map(|c| c.rationale.as_str()) - .collect::>(), - vec!["recompute", "different rewrite strategy", "share"], - "the preferred CSE choice must be ranked without losing the unrelated rewrite" - ); - let chosen = pick_shared_sub_dag_candidate( - &group, - decide_with_effective_count(&group, 2, &ConstantCseCost).unwrap(), - ) - .unwrap(); - assert_eq!(chosen.provenance, ReplacementProvenance::CseRecompute); + let roots: Vec> = shared.into_iter().map(|(_, rc)| rc).collect(); + let counts = count_consumers(&roots); + let count = counts[&Rc::as_ptr(&roots[0])]; + assert_eq!(count, 2); + + let target = TargetSubDAG::with_consumer_count(&roots[0], count); + assert!(SharedSubDAGStrategy.matches(&target)); + assert_eq!(SharedSubDAGStrategy.replacements(&target).len(), 2); } + // ── search_workload / CandidateLogicalASAPDAGs / TargetSubDAGCandidates (merged from search.rs) ── + // + // Reuses this test module's own `metric_scan`/`agg` fixture helpers + // above (identical to `search.rs`'s own copies, which are dropped here + // to avoid a duplicate-definition collision now that both test modules + // share one file) and `count_consumers` above (which mirrors + // `discover_targets`' own real, non-test traversal for these fixtures). + + // ── discovery + MEMO shape ─────────────────────────────────────────── + #[test] - fn effective_consumer_count_corrects_a_nested_groups_share_decision() { - // The interaction issue #271 describes: an outer shared sub-DAG `a` - // (referenced by 2 roots, so consumer_count == 2) wraps an inner - // shared sub-DAG `c` (referenced once through `a`'s own child edge, - // plus once more directly by a third, separate root — so `c`'s own - // *raw* structural consumer_count is also 2, independent of `a`). - // - // root1 ─┐ - // ├─▶ a = Filter(child = c) ─▶ c = Dedup(job) - // root2 ─┘ - // root3 ───────────────────────────▶ c (same shared Rc) - // - // `a` and `c` are both non-`Aggregate` nodes (`Filter`/`Dedup`) so - // neither is bindable — each group is a *clean* two-candidate - // SharedSubDAGStrategy share-vs-recompute pair, with no - // SketchAlgorithmStrategy `Summary` candidate mixed in to complicate - // ranking (see `shared_aggregate_across_two_roots_gets_both_strategies_candidates` - // for what a *mixed*-shape group looks like — deliberately avoided - // here to isolate the SharedSubDAGStrategy-only interaction). - // - // Under ConstantCseCost, consumer_count == 2 loses to maintenance - // (2 * 40 = 80 < 100 ⇒ RecomputeIndependently); consumer_count == 3 wins - // (3 * 40 = 120 > 100 ⇒ Share). `cost_sorted` only ever sees `c`'s raw - // count (2) and picks RecomputeIndependently for it — the WRONG - // answer once `a` itself is accounted for: `a`'s own decision is - // also RecomputeIndependently (same 80-vs-100 threshold), so `a` - // actually runs twice, and each run recomputes `c` once more — - // `c`'s *true* effective count is 2 (via `a`) + 1 (via root3) = 3, - // which flips its own decision to Share. Only global_selection, - // which folds `a`'s decision into `c`'s effective_consumer_count - // before deciding `c`, gets this right. - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - - let c = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), - }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(c()), + fn single_bindable_aggregate_keeps_unprovable_hydra_candidates() { + let intent = AggIntent::Count { + accuracy: AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, }; + let root = agg(vec![2], intent, metric_scan(&["job"])); + let space = search_workload(vec![("q", root)]); - let space = search_workload(vec![ - ("root1", Rc::new(a())), - ("root2", Rc::new(a())), - ("root3", Rc::new(c())), - ]); + // One group for the Aggregate, one for its Scan child. + assert_eq!(space.len(), 2); - // Fixture sanity: root1/root2 merged onto one shared `a`, and `c` - // (root1/root2's shared child, and root3 itself) merged onto one - // shared `c` with raw consumer_count 2, and both groups are clean - // (non-mixed) two-candidate SharedSubDAGStrategy pairs. - assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); - let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_via_a, .. } = a_rc.as_ref() else { - panic!("expected root1/root2 to still be a Filter"); - }; - assert!(Rc::ptr_eq(c_via_a, &space.roots[2].1)); - let a_group = space.candidates_for_target(a_rc).unwrap(); - let c_group = space.candidates_for_target(c_via_a).unwrap(); - assert_eq!( - a_group.consumer_count, 2, - "fixture sanity: a has 2 consumers" - ); - assert_eq!( - c_group.consumer_count, 2, - "fixture sanity: c has 2 raw consumers (via a's child edge, and via root3)" - ); - assert_eq!( - a_group.candidates.len(), - 2, - "fixture sanity: a is a clean Rewrite pair" - ); + let agg_group = space + .target_subdag_candidates() + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) + .expect("an Aggregate group must be discovered"); + assert_eq!(agg_group.consumer_count, 1); assert_eq!( - c_group.candidates.len(), - 2, - "fixture sanity: c is a clean Rewrite pair" + agg_group.candidates.len(), + 6, + "Hydra candidates with unknown evidence remain available: {:?}", + agg_group.candidates ); - - // The naive/local answer: cost_sorted ranks c using its raw count - // (2) alone and prefers RecomputeIndependently. - let ranked = space.cost_sorted(&ConstantCseCost); - let c_ranked = ranked + assert!(agg_group + .candidates .iter() - .find(|g| Rc::ptr_eq(g.target, c_via_a)) - .unwrap(); - let c_top_shares = matches!( - &c_ranked.candidates[0].replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, c_via_a) - ); - assert!( - !c_top_shares, - "cost_sorted, blind to a's own decision, must (wrongly) prefer \ - RecomputeIndependently for c using its raw consumer_count of 2" - ); - - // The corrected, cross-group-aware answer: global_selection folds - // a's own RecomputeIndependently choice into c's effective count - // (2 from a + 1 from root3 = 3) and flips to Share. - let selected = space.global_selection(&ConstantCseCost); - let a_selected = selected.for_target(a_rc).unwrap(); - let c_selected = selected.for_target(c_via_a).unwrap(); - + .all(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap()))); assert_eq!( - a_selected.effective_consumer_count, 2, - "a has no interacting ancestor" - ); - let a_shares = matches!( - &a_selected.chosen.unwrap().replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, a_rc) - ); - assert!( - !a_shares, - "fixture sanity: a itself must also choose RecomputeIndependently" + agg_group + .candidates + .iter() + .filter(|candidate| { + let Replacement::SubDAG(node) = &candidate.replacement else { + return false; + }; + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = + &node.operator + else { + return false; + }; + matches!( + &summary_input.operator, + Operator::ASAP(ASAPOp::SummaryAgg { + grouping: GroupingStrategy::SharedMultiSubpopulation { .. }, + .. + }) + ) + }) + .count(), + 2, + "Hydra candidates remain visible with symbolic shared-grid error" ); - assert_eq!( - c_selected.effective_consumer_count, 3, - "c's effective count must be 2 (a, itself recomputed twice) + 1 (root3)" - ); - let c_shares = matches!( - &c_selected.chosen.unwrap().replacement, - Replacement::Rewrite(rc) if Rc::ptr_eq(rc, c_via_a) + agg_group + .candidates + .iter() + .filter(|candidate| candidate.has_missing_accuracy_evidence()) + .count(), + 2 ); + let scan_group = space + .target_subdag_candidates() + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Scan { .. }))) + .expect("a Scan group must be discovered"); + assert_eq!(scan_group.consumer_count, 1); assert!( - c_shares, - "global_selection must flip c to Share once a's own recomputation is accounted for" + scan_group.candidates.is_empty(), + "no strategy matches a bare Scan" ); } #[test] - fn complete_plan_costs_reject_unbound_cse_arms() { - struct CompletePlanCost; - impl CostModel for CompletePlanCost { - fn candidate_cost_covers_complete_plan(&self) -> bool { - true - } - - fn candidate_cost( - &self, - candidate: &ReplacementSubDAG, - _target: &TargetSubDAG<'_>, - ) -> Option { - assert!(!is_cse_candidate(candidate)); - None - } - - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn cse_share_decision(&self, _candidate: &CseCandidate) -> ShareDecision { - ShareDecision::RecomputeIndependently - } - } - - let shared = Rc::new(QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), - }); - let space = search_workload(vec![ - ("left", Rc::clone(&shared)), - ("right", Rc::clone(&shared)), - ]); - let planned = &space.roots[0].1; - - let selected = space.global_selection(&CompletePlanCost); - assert!(selected.for_target(planned).unwrap().chosen.is_none()); - } - - #[test] - fn effective_repetition_materializes_a_cse_choice_for_a_single_edge_child() { - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - - let c = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), - }; - let a = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(c()), - }; - let space = search_workload(vec![("root1", Rc::new(a())), ("root2", Rc::new(a()))]); - let a_rc = &space.roots[0].1; - let QueryExpr::Filter { child: c_rc, .. } = a_rc.as_ref() else { - panic!("expected Filter root"); - }; + fn cardinality_group_keeps_all_four_candidates() { + let root = agg(vec![2], default_cardinality(), metric_scan(&["job"])); + let space = search_workload(vec![("q", root)]); + let agg_group = space + .target_subdag_candidates() + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) + .unwrap(); + assert_eq!(agg_group.candidates.len(), 4); + assert!(agg_group.candidates.iter().any(|candidate| matches!( + &candidate.replacement, + Replacement::SubDAG(node) if node.guarantee.is_none() + && candidate.has_missing_accuracy_evidence() + ))); - assert_eq!(space.candidates_for_target(c_rc).unwrap().consumer_count, 1); - assert!(cse_candidate_pair(space.candidates_for_target(c_rc).unwrap()).is_some()); + let root = agg(vec![2], default_cardinality(), metric_scan(&["job"])); + let targeted = search_workload_with_targets( + vec![( + "q", + root, + Some(AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }), + )], + &default_strategies(), + &DefaultAccuracyModel, + ); + let target = &targeted.roots[0].1; + assert!(targeted + .candidates_for_target(target) + .unwrap() + .candidates + .iter() + .any(|candidate| matches!( + &candidate.replacement, + Replacement::SubDAG(node) if node.guarantee.is_none() + && candidate.has_missing_accuracy_evidence() + ))); - let selected = space.global_selection(&ConstantCseCost); - let child = selected.for_target(c_rc).unwrap(); - assert_eq!(child.effective_consumer_count, 2); - assert!(child.chosen.is_some()); + let exact_target = search_workload_with_targets( + vec![( + "q", + agg(vec![2], default_cardinality(), metric_scan(&["job"])), + Some(AccuracyTarget::Exact), + )], + &default_strategies(), + &DefaultAccuracyModel, + ); + assert!(exact_target + .candidates_for_target(&exact_target.roots[0].1) + .unwrap() + .candidates + .iter() + .all(|candidate| !candidate.has_missing_accuracy_evidence())); } #[test] - fn shared_ancestor_keeps_a_single_use_cse_descendant_selected() { - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - - struct AlwaysShare; - impl CostModel for AlwaysShare { - fn allow_uncosted_legacy_selection(&self) -> bool { - true - } + fn shared_aggregate_across_two_roots_gets_both_strategies_candidates() { + // Two independently-built, structurally identical Sum aggregates: + // share_common_sub_dags (run inside search_workload) collapses them + // onto one Rc with consumer_count 2, so this single group should + // carry ASAPStrategies's one ExactAggregate candidate *and* + // SharedSubDAGStrategy's share-vs-recompute pair. + let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let space = search_workload(vec![("a", a), ("b", b)]); - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } + // roots[0] and roots[1] must have merged onto the same Rc. + assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); - fn cse_share_decision(&self, _candidate: &CseCandidate) -> ShareDecision { - ShareDecision::Share - } - } + let group = space.candidates_for_target(&space.roots[0].1).unwrap(); + assert_eq!(group.consumer_count, 2); + assert_eq!( + group.candidates.len(), + 3, + "1 ExactAggregate Summary + 2 Rewrite (share/recompute): {:?}", + group.candidates + ); - let child = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["job"])), - }; - let parent = || QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(child()), - }; - let space = search_workload(vec![ - ("root1", Rc::new(parent())), - ("root2", Rc::new(parent())), - ]); - let parent_rc = &space.roots[0].1; - let QueryExpr::Filter { - child: child_rc, .. - } = parent_rc.as_ref() - else { - panic!("expected Filter root"); - }; + // Old `Replacement::Summary` ↔ a `Subtree` containing an ASAP node; + // old `Replacement::Rewrite` ↔ a pure pre-ASAP `Subtree`. + let summary_count = group + .candidates + .iter() + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap())) + .count(); + let rewrite_count = group + .candidates + .iter() + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if !n.contains_asap())) + .count(); + assert_eq!(summary_count, 1); + assert_eq!(rewrite_count, 2); - let selected = space.global_selection(&AlwaysShare); - assert_eq!( - selected - .for_target(parent_rc) - .unwrap() - .effective_consumer_count, - 2 + // The two Rewrite candidates must NOT have collapsed into one + // (the "false-positive dedup" failure mode `is_duplicate_rewrite` + // exists to prevent). + let one_is_the_target = group.candidates.iter().any( + |c| matches!(&c.replacement, Replacement::SubDAG(rc) if Rc::ptr_eq(rc, &group.target)), ); - let child_selection = selected.for_target(child_rc).unwrap(); - assert_eq!(child_selection.effective_consumer_count, 1); - assert_eq!( - child_selection.chosen.map(|candidate| candidate.provenance), - Some(ReplacementProvenance::CseShare), - "a descendant collapsed to one execution still needs a selected plan" + let one_is_not = group.candidates.iter().any( + |c| matches!(&c.replacement, Replacement::SubDAG(rc) if !Rc::ptr_eq(rc, &group.target)), ); + assert!(one_is_the_target && one_is_not); } #[test] - fn global_selection_propagates_uses_through_the_selected_rewrite() { - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - - struct ReplaceFilterChild; - impl ReplacementStrategy for ReplaceFilterChild { - fn matches(&self, target: &TargetSubDAG<'_>) -> bool { - matches!(target.root.as_ref(), QueryExpr::Filter { .. }) - } + fn nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered() { + // A shared grouped Aggregate nested under two *different*, + // unshared Filter parents — real consumer_count must come from + // walking the whole DAG, not just root-level pointer identity + // (a naive whole-root-only consumer-count pass would miss this; + // this module's discover_targets must not). + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; - fn replacements(&self, _target: &TargetSubDAG<'_>) -> Vec { - vec![ReplacementSubDAG { - strategy: "ReplaceFilterChild", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(metric_scan(&["replacement"])), - })), - provenance: ReplacementProvenance::LogicalRewrite, - rationale: "replace the Filter and its input".into(), - }] - } - } + let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + // Different predicates so the two Filter *parents* stay distinct + // (don't themselves merge under CSE) — only their shared `child` + // should collapse onto one `Rc`. + let root_a = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: Rc::clone(&shared), + })) + .unwrap(); + let root_b = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: Rc::clone(&shared), + })) + .unwrap(); - let original_child = Rc::new(metric_scan(&["original"])); - let root = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::clone(&original_child), - }); - let strategies: Vec> = vec![Box::new(ReplaceFilterChild)]; - let space = search_workload_with(vec![("q", root)], &strategies); - let root = &space.roots[0].1; - let selected = space.global_selection(&DefaultCostModel); - let Replacement::Rewrite(rewrite) = &selected - .for_target(root) - .unwrap() - .chosen - .unwrap() - .replacement - else { - panic!("expected logical rewrite"); - }; - let QueryExpr::Dedup { - child: replacement_child, + let space = search_workload(vec![("a", root_a), ("b", root_b)]); + assert_eq!( + space.len(), + 4, + "2 distinct Filters + 1 shared Aggregate + 1 shared Scan" + ); + + // `share_common_sub_dags` re-clones+re-interns anything that already + // had more than one owner going in (see `cse.rs`'s own doc on + // `intern_child`'s clone-fallback path) — so the post-CSE shared + // node is a *fresh* Rc, structurally equal to (but not the same + // pointer as) the pre-search `shared` variable. Recover it from the + // post-CSE root's own `child` field instead of the stale `shared` + // handle. + let Some(NonASAPOp::Filter { + child: post_cse_shared_a, .. - } = rewrite.as_ref() + }) = space.roots[0].1.non_asap() else { - panic!("expected Dedup rewrite"); + panic!("expected a Filter root"); }; - let QueryExpr::Filter { - child: original_child, + let Some(NonASAPOp::Filter { + child: post_cse_shared_b, .. - } = root.as_ref() + }) = space.roots[1].1.non_asap() else { - panic!("expected Filter root"); + panic!("expected a Filter root"); }; - - assert_eq!( - selected - .for_target(original_child) - .unwrap() - .effective_consumer_count, - 0 + assert!( + Rc::ptr_eq(post_cse_shared_a, post_cse_shared_b), + "fixture sanity: the two Filters' children must still merge" ); - assert_eq!( - selected - .for_target(replacement_child) - .unwrap() - .effective_consumer_count, - 1 + let post_cse_shared = post_cse_shared_a; + let group = space + .candidates_for_target(post_cse_shared) + .expect("shared node must be a discovered target"); + assert_eq!(group.consumer_count, 2); + assert!( + SharedSubDAGStrategy.matches(&TargetSubDAG::with_consumer_count( + post_cse_shared, + group.consumer_count + )) ); } - // A cheap but physically infeasible candidate must not be selected. - #[test] - fn explicit_summary_infeasibility_prevents_selection() { - struct Unsupported; - impl CostModel for Unsupported { - fn rank_candidates( - &self, - _: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - fn candidate_cost(&self, _: &ReplacementSubDAG, _: &TargetSubDAG<'_>) -> Option { - Some(Cost(1.0)) - } - fn summary_support_evidence(&self, _: &SummaryNode) -> Option { - Some(false) - } - } - let root = Rc::new(lower_promql("sum_over_time(a[1m])", AccuracyTarget::Exact)); - let space = search_workload(vec![("q", root)]); - let selected = space.global_selection(&Unsupported); - assert!(selected - .for_target(&space.roots[0].1) - .unwrap() - .chosen - .is_none()); - } + // ── dedup ──────────────────────────────────────────────────────────── - // Composable temporal/grouped Sum must be executable as one producer. #[test] - fn grouped_temporal_sum_has_one_summary_producer_candidate() { - let root = Rc::new(lower_promql( - "sum by(job)(sum_over_time(a[1m]))", - AccuracyTarget::Exact, - )); - let candidates = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); - assert!(candidates - .iter() - .any(|candidate| matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))))); - struct PreferComposed; - impl CostModel for PreferComposed { - fn rank_candidates( - &self, - _: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - fn candidate_cost( - &self, - candidate: &ReplacementSubDAG, - _: &TargetSubDAG<'_>, - ) -> Option { - Some(Cost( - if matches!(&candidate.replacement, - Replacement::Summary(node) if matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))) - { - 1.0 - } else { - 100.0 - }, - )) + fn add_candidate_rejects_a_true_rewrite_duplicate() { + // SharedSubDAGStrategy's `Replacement::Rewrite` candidates are + // real `OperatorNode` values with `PartialEq`, so `add_candidate` can + // (and must) actually reject a genuine repeat — unlike the + // `Replacement::Summary` case (see the test below). + let root = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 2); + let target = TargetSubDAG::with_consumer_count(&root, 2); + let mut inserted = 0; + for candidate in SharedSubDAGStrategy.replacements(&target) { + if group.add_candidate(candidate) { + inserted += 1; } } - let space = search_workload(vec![("q", root.clone())]); - let selected = space.global_selection(&PreferComposed); - let node = selected.assemble_target(&space.roots[0].1).unwrap(); - assert!(matches!(&node.expr, - SummaryExpr::SummaryAgg { reduction: Reduction::Reduce(_), child, .. } - if matches!(child.expr, SummaryExpr::KeepPreAsap(_)))); - } + assert_eq!(inserted, 2, "share + recompute-independently candidates"); - // Mixed candidate ranking must honor explicit costs, not legacy estimates. - #[test] - fn mixed_candidate_ranking_uses_explicit_candidate_costs() { - struct ExplicitCosts; - impl CostModel for ExplicitCosts { - fn rank_candidates( - &self, - _: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - fn candidate_cost( - &self, - candidate: &ReplacementSubDAG, - _: &TargetSubDAG<'_>, - ) -> Option { - Some(Cost( - if candidate.provenance == ReplacementProvenance::LogicalRewrite { - 1.0 - } else { - 100.0 - }, - )) + // Re-adding the identical candidate list must add nothing new: the + // "share" candidate is literally the same Rc as before, and the + // "recompute independently" candidate is a fresh Rc but + // structurally identical value, both already covered by + // `is_duplicate_rewrite`. + let mut re_inserted = 0; + for candidate in SharedSubDAGStrategy.replacements(&target) { + if group.add_candidate(candidate) { + re_inserted += 1; } } - let root = Rc::new(lower_promql( - "sum by(job)(sum_over_time(a[1m]))", - AccuracyTarget::Exact, - )); - let space = search_workload(vec![("q", root)]); - let selection = space.global_selection(&ExplicitCosts); - let selected = selection - .for_target(&space.roots[0].1) - .unwrap() - .chosen - .unwrap(); - assert_eq!(selected.provenance, ReplacementProvenance::LogicalRewrite); + assert_eq!( + re_inserted, 0, + "re-proposing the same Rewrite candidates must not grow the group" + ); + assert_eq!(group.candidates.len(), 2); } #[test] - fn global_selection_compares_a_logical_rewrite_with_the_cse_choice() { - struct PreferLogicalRewrite; - - impl CostModel for PreferLogicalRewrite { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn estimate_cost( - &self, - candidate: &ReplacementSubDAG, - _target: &TargetSubDAG<'_>, - ) -> f64 { - match candidate.provenance { - ReplacementProvenance::LogicalRewrite => 0.0, - _ => 100.0, - } - } + fn add_candidate_never_dedups_summary_candidates() { + // Documented, deliberate consequence of `is_duplicate_summary` + // refusing value equality on `f64`-bearing summaries: re-proposing + // the same `Replacement::Summary` candidates DOES grow the group — + // this module refuses to guess at an equality check it can't back + // with a real `PartialEq`. `search_workload_with` never actually + // does this in practice (every target is asked exactly once — see the + // module docs' "Termination" section), so this test exists to pin + // the documented behavior, not to endorse calling `replacements` + // twice for the same target. + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 1); + let strategy = ASAPStrategies::default(); + let target = TargetSubDAG::new(&root); + for candidate in strategy.replacements(&target) { + group.add_candidate(candidate); } + assert_eq!(group.candidates.len(), 2); - let a = Rc::new(agg( - vec![2], - AggIntent::Avg { col: None }, - metric_scan(&["job"]), - )); - let b = Rc::new(agg( - vec![2], - AggIntent::Avg { col: None }, - metric_scan(&["job"]), - )); - let space = search_workload(vec![("a", a), ("b", b)]); - let root = &space.roots[0].1; - let selected = space.global_selection(&PreferLogicalRewrite); - + for candidate in strategy.replacements(&target) { + group.add_candidate(candidate); + } assert_eq!( - selected - .for_target(root) - .and_then(|group| group.chosen) - .map(|candidate| candidate.provenance), - Some(ReplacementProvenance::LogicalRewrite) + group.candidates.len(), + 4, + "Summary candidates are never deduped by this module — see is_duplicate_summary" ); } #[test] - fn topological_order_puts_a_later_discovered_parent_before_its_child() { - // Mirrors nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered's - // diamond fixture: discover_targets's own `order` visits root_b (a - // parent of `shared`) *after* `shared` itself, because `shared` was - // already fully walked via root_a first. A naive "process - // discover_targets's own order" DP would see root_b's child edge - // after already processing `shared` — topological_order must not - // make that mistake. - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - - let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let root_a = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(1)))), - child: Rc::new(shared.clone()), - }; - let root_b = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(2)))), - child: Rc::new(shared), - }; - let roots = vec![("a", Rc::new(root_a)), ("b", Rc::new(root_b))]; - - let mut order = Vec::new(); - let mut nodes = HashMap::new(); - let mut counts = HashMap::new(); - discover_targets(&roots, &mut order, &mut nodes, &mut counts); - let groups = order - .iter() - .map(|ptr| { - ( - *ptr, - TargetSubDAGCandidates::new(Rc::clone(&nodes[ptr]), counts[ptr]), - ) - }) - .collect(); - let space = CandidateLogicalASAPDAGs { - roots, - groups, - order: order.clone(), - composition_plans: Vec::new(), - }; - let dag = reference_dag(&space); - - // Discovery-order sanity: root_b comes after the shared child in - // discover_targets's own order (the exact non-topological case this - // test exists to cover). - let QueryExpr::Filter { - child: shared_via_a, - .. - } = space.roots[0].1.as_ref() - else { - panic!("expected a Filter root"); - }; - let shared_ptr = Rc::as_ptr(shared_via_a); - let root_b_ptr = Rc::as_ptr(&space.roots[1].1); - let shared_discovery_pos = order.iter().position(|p| *p == shared_ptr).unwrap(); - let root_b_discovery_pos = order.iter().position(|p| *p == root_b_ptr).unwrap(); - assert!( - root_b_discovery_pos > shared_discovery_pos, - "fixture sanity: discover_targets's own order must NOT already be topological here" + fn is_duplicate_rewrite_never_merges_share_with_recompute() { + let target = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let share = Rc::clone(&target); + let recompute = Rc::new((*target).clone()); + assert!(!Rc::ptr_eq(&share, &recompute)); + assert_eq!( + *share, *recompute, + "fixture sanity: same value, different Rc" ); + assert!(!is_duplicate_rewrite(&share, &recompute, &target)); + assert!(!is_duplicate_rewrite(&recompute, &share, &target)); + } - let topo = topological_order(&order, &dag); - let shared_topo_pos = topo.iter().position(|p| *p == shared_ptr).unwrap(); - let root_b_topo_pos = topo.iter().position(|p| *p == root_b_ptr).unwrap(); - assert!( - root_b_topo_pos < shared_topo_pos, - "topological_order must place root_b (a parent of the shared node) before it, \ - unlike discover_targets's own discovery order" - ); + #[test] + fn is_duplicate_rewrite_catches_a_real_repeat() { + let target = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let first_recompute = Rc::new((*target).clone()); + let second_recompute = Rc::new((*target).clone()); + assert!(!Rc::ptr_eq(&first_recompute, &second_recompute)); + assert!(is_duplicate_rewrite( + &first_recompute, + &second_recompute, + &target + )); } // ── termination ────────────────────────────────────────────────────── @@ -9738,7 +7059,7 @@ mod tests { // module docs — so this always converges in exactly 2 passes). let a = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let b = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); + let space = search_workload(vec![("a", a), ("b", b)]); assert!(!space.is_empty()); } @@ -9764,19 +7085,23 @@ mod tests { fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { let n = self.next.get(); self.next.set(n + 1); - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let fresh_inner_layer = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64(n)))), - child: Rc::clone(target.root), - }; - let outer_wrapper = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(fresh_inner_layer), - }; + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; + let fresh_inner_layer = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(n))), + child: Rc::clone(target.root), + })) + .unwrap(); + let outer_wrapper = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: fresh_inner_layer, + })) + .unwrap(); vec![ReplacementSubDAG { strategy: "AlwaysGrowingStrategy", - replacement: Replacement::Rewrite(Rc::new(outer_wrapper)), + replacement: Replacement::SubDAG(outer_wrapper), provenance: ReplacementProvenance::LogicalRewrite, rationale: format!("pathological candidate #{n}"), }] @@ -9786,18 +7111,18 @@ mod tests { #[test] #[should_panic(expected = "did not converge")] fn a_pathologically_growing_strategy_trips_the_iteration_cap() { - let root = Rc::new(metric_scan(&["job"])); + let root = metric_scan(&["job"]); let strategies: Vec> = vec![Box::new(AlwaysGrowingStrategy { next: std::cell::Cell::new(0), })]; let _ = search_workload_with(vec![("q", root)], &strategies); } - // ── realize_child / keep_pre_asap: end-to-end single-target realization ── + // ── realize_child / retain_exact: end-to-end single-target realization ── // // Moved from the former `bind.rs` (issue #251): `bind.rs`'s own // workload-wide orchestration (`implement_workload`/ // `implement_workload_with`) was deleted. Current whole-workload logical - // selection uses `CandidateLogicalASAPDAGs::global_selection`; these tests exercise + // selection uses `candidate_selection::global_selection`; these tests exercise // `construct_summary_agg`'s schema derivation end to end through // `realize_child` — production logic that still lives in this module — // so they move here rather than disappear. Unlike `bind.rs` (an @@ -9805,17 +7130,6 @@ mod tests { // pattern by hand since `realize_child` is `pub(crate)`), these tests // call `realize_child` directly. - fn agg_per_entity(intent: AggIntent, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { - reduction: ReductionTy::PerEntity, - measures: vec![intent], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(child), - } - } - fn field<'a>(schema: &'a Schema, name: &str) -> &'a Field { schema .fields @@ -9824,30 +7138,23 @@ mod tests { .unwrap_or_else(|| panic!("no field {name:?} in {schema:?}")) } - fn realize_first( - expr: &QueryExpr, - cost_model: &dyn CostModel, - ) -> Result, RealizationError> { - realize_child(&Rc::new(expr.clone()), cost_model) - } - - fn realize(expr: &QueryExpr) -> Result, RealizationError> { - realize_first(expr, &DefaultCostModel) + fn realize(expr: &OperatorNode) -> Result, RealizationError> { + realize_child(&Rc::new(expr.clone())) } #[test] fn quantile_realizes_kll_wrapped_in_estimate() { // quantile by (job) (m) at ε=0.01 → Estimate(Quantile) over - // SummaryAgg(Kll{k:269}) over KeepPreAsap(Scan). job = col 2. + // SummaryAgg(Kll{k:269}) over the kept Scan. job = col 2. let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &root.expr + }) = &root.operator else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); + panic!("expected SummaryEstimate root, got {:?}", root.operator); }; assert!(matches!(query, PostAsapSketchStatistic::Quantile { q } if *q == 0.99)); // Estimate edge: plain row shape — group key + Float64 answer. @@ -9860,203 +7167,60 @@ mod tests { FieldDataType::Plain(DataType::Utf8) ); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, reduction, - .. - } = &summary_input.expr - else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); - }; - assert_eq!( - family, - &FieldDataType::Sketch( - SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), - GroupingStrategy::default() - ) - ); - assert_eq!(input, &SummaryUpdate::column(ColumnRef::SampleValue)); - assert_eq!(reduction, &ReductionTy::by(vec![2])); - // SummaryAgg edge: the state column, named after its input, carries - // the committed family. - assert_eq!( - field(&summary_input.schema, "value").dtype, - FieldDataType::Sketch( - SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), - GroupingStrategy::default() - ) - ); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(ref e) - if matches!(**e, QueryExpr::Scan { .. }))); - } - - /// A deployment-supplied [`CostModel`] can override the default KLL - /// choice — `realize_first` (via `realize_child`) must actually consult - /// it, not just accept and ignore it (issue: cost model interface, see - /// `crate::cost_model`). - struct PreferDDSketchViaCostModel; - - impl CostModel for PreferDDSketchViaCostModel { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - let mut v = candidates.to_vec(); - if let Some(pos) = v.iter().position(|k| *k == SketchAlgorithm::DDSketch) { - let ddsketch = v.remove(pos); - v.insert(0, ddsketch); - } - v - } - } - - #[test] - fn realize_with_custom_cost_model_overrides_default_summary_choice() { - let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - - // Default: KLL (see `quantile_realizes_kll_wrapped_in_estimate` above). - let default_root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &default_root.expr else { - panic!("expected SummaryEstimate root, got {:?}", default_root.expr); - }; - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); - }; - assert!(matches!( - family, - FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll - )); - - // With `PreferDDSketchViaCostModel`: DDSketch instead, same query. - let custom_root = realize_first(&q, &PreferDDSketchViaCostModel).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &custom_root.expr else { - panic!("expected SummaryEstimate root, got {:?}", custom_root.expr); - }; - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + .. + }) = &summary_input.operator + else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!( family, &FieldDataType::Sketch( - SketchKind::new( - SketchAlgorithm::DDSketch, - SketchParams::DDSketch { alpha: 0.01 } - ), + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), GroupingStrategy::default() ) ); - } - - /// A deployment-supplied `CostModel` can realize an `AggIntent::Extension` - /// intent as a real sketch instead of the default `PassThrough` (issue - /// #150) — `realizations_for_intent` must consult `realize_extension` - /// for the `Extension` arm, and `readout` must consult - /// `readout_extension` to build its `SketchStatistic` without panicking. - struct FrequencyCostModel; - - impl CostModel for FrequencyCostModel { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn realize_extension(&self, ext_kind: &str, _payload: &serde_json::Value) -> Realization { - if ext_kind == "frequency" { - Realization::Sketch(SketchKind::new( - SketchAlgorithm::CountSketch, - SketchParams::CountSketch { - width: 256, - depth: 4, - }, - )) - } else { - Realization::PassThrough - } - } - - fn readout_extension( - &self, - ext_kind: &str, - payload: &serde_json::Value, - _col: &ColumnRef, - ) -> PostAsapSketchStatistic { - assert_eq!(ext_kind, "frequency"); - let value = payload["item"].as_str().map(str::to_string); - PostAsapSketchStatistic::PointCount { - key: ColumnRef::Named("item".into()), - value, - } - } + assert_eq!(input, &SummaryUpdate::column(ColumnRef::SampleValue)); + assert_eq!(reduction, &ReductionTy::by(vec![2])); + // SummaryAgg edge: the state column, named after its input, carries + // the committed family. + assert_eq!( + field(&summary_input.schema, "value").dtype, + FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 269 }), + GroupingStrategy::default() + ) + ); + // The kept pre-ASAP leaf is the Scan node itself (no wrapper). + assert!(matches!(child.non_asap(), Some(NonASAPOp::Scan { .. }))); + assert!(!child.contains_asap()); } #[test] fn extension_intent_stays_logical_by_default() { - // Without a CostModel overriding `realize_extension`, an - // `Extension` intent must stay `PassThrough` -- today's behavior, - // unchanged. + // Core has no realization for a deployment-specific `Extension` + // intent, so it stays `PassThrough`. let intent = AggIntent::Extension { ext_kind: "frequency".to_string(), payload: serde_json::json!({ "item": "checkout" }), }; let q = agg(vec![], intent, metric_scan(&[])); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); - } - - #[test] - fn extension_intent_realizes_via_custom_cost_model() { - let intent = AggIntent::Extension { - ext_kind: "frequency".to_string(), - payload: serde_json::json!({ "item": "checkout" }), - }; - let q = agg(vec![], intent, metric_scan(&[])); - let root = realize_first(&q, &FrequencyCostModel).unwrap(); - - let SummaryExpr::SummaryEstimate { - summary_input, - query, - } = &root.expr - else { - panic!("expected SummaryEstimate root, got {:?}", root.expr); - }; - assert!(matches!( - query, - PostAsapSketchStatistic::PointCount { key: ColumnRef::Named(k), value: Some(v) } - if k == "item" && v == "checkout" - )); - - let SummaryExpr::SummaryAgg { family, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); - }; - assert_eq!( - family, - &FieldDataType::Sketch( - SketchKind::new( - SketchAlgorithm::CountSketch, - SketchParams::CountSketch { - width: 256, - depth: 4 - } - ), - GroupingStrategy::default() - ) - ); + assert!(!root.contains_asap()); } #[test] fn exact_sum_realizes_accumulator_without_estimate() { let q = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { family, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &root.operator else { panic!( "expected bare SummaryAgg (no estimate), got {:?}", - root.expr + root.operator ); }; assert_eq!( @@ -10076,14 +7240,16 @@ mod tests { use std::time::Duration; let q = agg_per_entity( AggIntent::Rate, - QueryExpr::TimeRange { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["job"])), - }, + child: metric_scan(&["job"]), + })) + .unwrap(), ); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { family, .. } = &root.expr else { - panic!("expected SummaryAgg, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &root.operator else { + panic!("expected SummaryAgg, got {:?}", root.operator); }; assert_eq!( family, @@ -10114,17 +7280,19 @@ mod tests { use std::time::Duration; let q = agg_per_entity( default_quantile(0.99), - QueryExpr::TimeRange { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, range: Duration::from_secs(10), - child: Rc::new(metric_scan(&["job"])), - }, + child: metric_scan(&["job"]), + })) + .unwrap(), ); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { reduction, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!(reduction, &ReductionTy::PerEntity); } @@ -10142,11 +7310,11 @@ mod tests { }; let q = agg(vec![], intent, metric_scan(&["job"])); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { reduction, .. } = &summary_input.expr else { - panic!("expected SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { reduction, .. }) = &summary_input.operator else { + panic!("expected SummaryAgg, got {:?}", summary_input.operator); }; assert_eq!(reduction, &ReductionTy::by(vec![])); } @@ -10158,39 +7326,41 @@ mod tests { // accumulator. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); - let root = realize(&outer).unwrap(); + // Timing is not stored during realization: time the candidate under + // a maintained materialization assignment to read the maintenance boundary. + let root = maintained(&realize(&outer).unwrap()); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("expected estimate root, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { + panic!("expected estimate root, got {:?}", root.operator); }; - let SummaryExpr::SummaryAgg { child, family, .. } = &summary_input.expr else { - panic!("expected outer SummaryAgg, got {:?}", summary_input.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, .. }) = &summary_input.operator + else { + panic!( + "expected outer SummaryAgg, got {:?}", + summary_input.operator + ); }; assert!(matches!( family, FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::Kll )); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - } = &child.expr - else { - panic!("expected explicit maintenance readout"); + assert_eq!(child.timing, Some(ExecutionTiming::IngestionTime)); + let Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) = &child.operator else { + panic!("expected explicit maintenance evaluation"); }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: inner_family, child: leaf, .. - } = &child.expr + }) = &child.operator else { - panic!("expected inner SummaryAgg, got {:?}", child.expr); + panic!("expected inner SummaryAgg, got {:?}", child.operator); }; assert_eq!( inner_family, &FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) ); - assert!(matches!(leaf.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!leaf.contains_asap()); } /// Issue #115: the summary is built over the intent's own input columns. @@ -10225,13 +7395,15 @@ mod tests { } } - /// The update expression of the first `SummaryAgg` in the DAG. - fn find_summary_input(node: &SummaryNode) -> Option { - match &node.expr { - SummaryExpr::SummaryAgg { input, .. } if input.item.is_none() => { + /// The update expression of the first `SummaryAgg` in the tree. + fn find_summary_input(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) if input.item.is_none() => { Some(input.weight.clone()) } - SummaryExpr::SummaryEstimate { summary_input, .. } => find_summary_input(summary_input), + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + find_summary_input(summary_input) + } _ => None, } } @@ -10252,78 +7424,79 @@ mod tests { ] { let q = agg(vec![2], intent.clone(), metric_scan(&["job"])); let root = realize(&q).unwrap(); + // Kept pass-through: the pre-ASAP node itself, not a wrapper. assert!( - matches!(root.expr, SummaryExpr::KeepPreAsap(ref e) if **e == q), - "expected KeepPreAsap passthrough for {intent:?}" + !root.contains_asap() && root.operator == q.operator && root.schema == q.schema, + "expected kept pre-ASAP passthrough for {intent:?}" ); } } #[test] fn logical_parent_subsumes_bindable_child() { - // Filter over a bindable quantile: `KeepPreAsap` has no post-ASAP - // children, so the conservative fallback keeps the whole sub-DAG + // Filter over a bindable quantile: a kept non-ASAP sub-DAG has no + // summary children, so the conservative fallback keeps the whole sub-DAG // logical. - use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; - use asap_types::pre_asap::query_expr::Predicate; - let q = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; + use asap_types::ir::Predicate; + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(0.5))), - })), - child: Rc::new(agg(vec![], default_quantile(0.99), metric_scan(&[]))), - }; + right: Box::new(ScalarExpr::Literal(ScalarValue::Float64(0.5))), + semantics: asap_types::ir::ExprSemantics::Promql, + }), + child: agg(vec![], default_quantile(0.99), metric_scan(&[])), + })) + .unwrap(); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(ref e) if **e == q)); + assert!( + !root.contains_asap() && root.operator == q.operator && root.schema == q.schema, + "expected the whole Filter sub_dag kept pre-ASAP" + ); } #[test] fn having_and_multi_intent_stay_logical() { - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; - let mut q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { - *having = Some(Predicate(Rc::new(QueryExpr::Literal( - ScalarValue::Boolean(true), - )))); - } - assert!(matches!( - realize(&q).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); - - let multi = QueryExpr::Aggregate { - reduction: ReductionTy::by(vec![2]), - measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }; - assert!(matches!( - realize(&multi).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; + let q = crate::test_support::aggregate( + ReductionTy::by(vec![2]), + vec![default_quantile(0.99)], + vec![], + Some(Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true)))), + metric_scan(&["job"]), + ); + assert!(!realize(&q).unwrap().contains_asap()); + + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: ReductionTy::by(vec![2]), + measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); + assert!(!realize(&multi).unwrap().contains_asap()); } // No binding rule applies a per-measure `FILTER` (#466), so the // aggregate is retained exactly rather than bound to a summary. #[test] fn filtered_measure_stays_logical() { - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::Predicate; + use asap_types::ir::scalar::ScalarValue; let mut q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - if let QueryExpr::Aggregate { filters, .. } = &mut q { - *filters = vec![Some(Predicate(Rc::new(QueryExpr::Literal( - ScalarValue::Boolean(true), + if let Operator::NonASAP(NonASAPOp::Aggregate { filters, .. }) = + &mut Rc::make_mut(&mut q).operator + { + *filters = vec![Some(Predicate(ScalarExpr::Literal(ScalarValue::Boolean( + true, ))))]; } assert!(bindable_intent(&q).is_none()); - assert!(matches!( - realize(&q).unwrap().expr, - SummaryExpr::KeepPreAsap(_) - )); + assert!(!realize(&q).unwrap().contains_asap()); } #[test] @@ -10337,7 +7510,7 @@ mod tests { metric_scan(&["job"]), ); let root = realize(&q).unwrap(); - assert!(matches!(root.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!root.contains_asap()); } #[test] @@ -10349,20 +7522,19 @@ mod tests { }, metric_scan(&["job"]), ); - let root = Rc::new(agg( + let root = agg( vec![], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let proposals = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); + ); + let proposals = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); assert!(!proposals.is_empty()); assert!(proposals.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.as_ref().is_some_and(|g| + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|g| g.bound.evaluate().is_none() && g.failure_probability.evaluate().is_none()) ))); @@ -10402,7 +7574,7 @@ mod tests { }, metric_scan(&["job"]), ); - let q = Rc::new(agg( + let q = agg( vec![], AggIntent::TopK { k: 5, @@ -10412,9 +7584,8 @@ mod tests { }, }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopKEvidence, @@ -10423,7 +7594,7 @@ mod tests { assert!(!replacements.is_empty()); assert!(replacements.iter().all(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(|g| g.metric == ErrorMetric::TopKMembership && g.failure_probability.evaluate() == Some(0.005)) @@ -10439,16 +7610,15 @@ mod tests { }, metric_scan(&["service"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopKEvidence, @@ -10457,26 +7627,26 @@ mod tests { let node = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) if candidate.rationale.contains("CmsWithHeap") => { + Replacement::SubDAG(node) if candidate.rationale.contains("CmsWithHeap") => { Some(node) } _ => None, }) .expect("CmsWithHeap candidate"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &node.expr + }) = &node.operator else { - panic!("expected Top-K readout") + panic!("expected Top-K evaluation") }; assert!(matches!(query, PostAsapSketchStatistic::TopK { k: 10 })); - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { child, family, input, .. - } = &summary_input.expr + }) = &summary_input.operator else { panic!("expected fused summary aggregation") }; @@ -10496,7 +7666,7 @@ mod tests { FieldDataType::Sketch(kind, _) if kind.algorithm() == &SketchAlgorithm::CmsWithHeap )); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } #[test] @@ -10506,16 +7676,15 @@ mod tests { AggIntent::Sum { col: None }, metric_scan(&["service"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopKEvidence, @@ -10531,7 +7700,7 @@ mod tests { let node = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if candidate.rationale.contains("CountSketchWithHeap") => { Some(node) @@ -10539,15 +7708,16 @@ mod tests { _ => None, }) .expect("CountSketchWithHeap candidate"); - let SummaryExpr::SummaryEstimate { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, query, - } = &node.expr + }) = &node.operator else { - panic!("expected Top-K readout") + panic!("expected Top-K evaluation") }; assert!(matches!(query, PostAsapSketchStatistic::TopK { k: 5 })); - let SummaryExpr::SummaryAgg { child, input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, input, .. }) = &summary_input.operator + else { panic!("expected fused summary aggregation") }; assert!(matches!( @@ -10558,45 +7728,51 @@ mod tests { input.weight, SummaryInputExpr::Column(ColumnRef::SampleValue) ); - assert!(matches!(child.expr, SummaryExpr::KeepPreAsap(_))); + assert!(!child.contains_asap()); } #[test] fn temporal_per_entity_topk_uses_series_identity_and_sample_value() { - let inner = QueryExpr::Aggregate { - reduction: ReductionTy::PerEntity, - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: std::time::Duration::from_secs(60), - child: Rc::new(metric_scan(&["service"])), - }), - }; - let outer = Rc::new(agg( + let inner = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: ReductionTy::PerEntity, + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + range: std::time::Duration::from_secs(60), + child: metric_scan(&["service"]), + }, + )) + .unwrap(), + })) + .unwrap(); + let outer = agg( vec![2], AggIntent::TopK { k: 5, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopKEvidence, ); let candidates = strategy.replacements(&TargetSubDAG::new(&outer)); let input = candidates.iter().find_map(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return None; }; - let SummaryExpr::SummaryEstimate { summary_input, .. } = &node.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &node.operator + else { return None; }; - let SummaryExpr::SummaryAgg { input, .. } = &summary_input.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &summary_input.operator else { return None; }; input.item.is_some().then_some(input) @@ -10625,16 +7801,15 @@ mod tests { }, metric_scan(&["service", "region"]), ); - let outer = Rc::new(agg( + let outer = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopKEvidence, @@ -10644,10 +7819,10 @@ mod tests { .replacements(&TargetSubDAG::new(&outer)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { - match &summary_input.expr { - SummaryExpr::SummaryAgg { input, .. } => Some(input.clone()), + Replacement::SubDAG(node) => match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) => Some(input.clone()), _ => None, } } @@ -10678,16 +7853,15 @@ mod tests { ); // The inner aggregate outputs its grouping keys first, so column 2 is // `region`. Each region is a separate Top-K subpopulation. - let outer = Rc::new(agg( + let outer = agg( vec![2], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + ); + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &SeparatedTopKEvidence, @@ -10696,12 +7870,12 @@ mod tests { .replacements(&TargetSubDAG::new(&outer)) .into_iter() .find_map(|candidate| match candidate.replacement { - Replacement::Summary(node) => match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => { - match &summary_input.expr { - SummaryExpr::SummaryAgg { + Replacement::SubDAG(node) => match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + match &summary_input.operator { + Operator::ASAP(ASAPOp::SummaryAgg { input, reduction, .. - } => Some((input.clone(), reduction.clone())), + }) => Some((input.clone(), reduction.clone())), _ => None, } } @@ -10726,7 +7900,7 @@ mod tests { fn sql_reducer_resolves_named_input_column() { // SUM(bytes) over a tabular scan: `col` resolves positionally to the // named column, not the PromQL sample value. - let scan = QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "t".into(), }, @@ -10740,11 +7914,12 @@ mod tests { unique_keys: vec![], closed: true, }, - }; + })) + .unwrap(); let q = agg(vec![0], AggIntent::Sum { col: Some(1) }, scan); let root = realize(&q).unwrap(); - let SummaryExpr::SummaryAgg { input, .. } = &root.expr else { - panic!("expected SummaryAgg, got {:?}", root.expr); + let Operator::ASAP(ASAPOp::SummaryAgg { input, .. }) = &root.operator else { + panic!("expected SummaryAgg, got {:?}", root.operator); }; let SummaryInputExpr::Column(col) = &input.weight else { panic!("expected observation column") @@ -10754,7 +7929,7 @@ mod tests { // ── Accuracy guarantees and fail-closed composition (issue #172) ───── - use asap_types::post_asap::ErrorMetric; + use asap_types::ir::properties::ErrorMetric; /// A test-only `AccuracyModel` that *registers* a rule the default /// deliberately lacks — a sketch over rank-bounded inputs composes @@ -10809,10 +7984,12 @@ mod tests { } } - fn summary_child(node: &SummaryNode) -> &Rc { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => summary_child(summary_input), - SummaryExpr::SummaryAgg { child, .. } => child, + fn summary_child(node: &OperatorNode) -> &Rc { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + summary_child(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => child, other => panic!("expected a SummaryAgg, got {other:?}"), } } @@ -10823,9 +8000,8 @@ mod tests { // registered rule, so every outer sketch candidate is refused with a // typed reason and the raw/pre-ASAP alternative is what remains. let inner = agg(vec![2], default_quantile(0.5), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); - let proposals = - SketchAlgorithmStrategy::default_cost_model().propose(&TargetSubDAG::new(&outer)); + let outer = agg(vec![], default_quantile(0.99), inner); + let proposals = ASAPStrategies::default().propose(&TargetSubDAG::new(&outer)); assert!( proposals.candidates.is_empty(), "no outer sketch may be proposed over an approximate child without a rule: {:?}", @@ -10847,8 +8023,8 @@ mod tests { ); } // Fallback keeps the whole sub-DAG pre-ASAP — executed exactly. - let realized = realize_child(&outer, &DefaultCostModel).unwrap(); - assert!(matches!(realized.expr, SummaryExpr::KeepPreAsap(_))); + let realized = realize_child(&outer).unwrap(); + assert!(!realized.contains_asap()); assert!(realized .guarantee .as_ref() @@ -10856,9 +8032,8 @@ mod tests { // Cross-metric: a quantile over a cardinality estimate. let inner = agg(vec![2], default_cardinality(), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], default_quantile(0.99), inner)); - let proposals = - SketchAlgorithmStrategy::default_cost_model().propose(&TargetSubDAG::new(&outer)); + let outer = agg(vec![], default_quantile(0.99), inner); + let proposals = ASAPStrategies::default().propose(&TargetSubDAG::new(&outer)); assert!(proposals.candidates.is_empty()); assert!(proposals.rejected.iter().all(|r| matches!( &r.error, @@ -10870,14 +8045,14 @@ mod tests { #[test] fn exact_child_contributes_zero_error() { // quantile(0.9, sum by (job) (m)): KLL over an exact Sum accumulator - // — the readout's guarantee is exactly KLL's own local guarantee. + // — the evaluation's guarantee is exactly KLL's own local guarantee. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); let root = realize(&outer).unwrap(); let guarantee = root .guarantee .as_ref() - .expect("a readout carries a guarantee"); + .expect("a evaluation carries a guarantee"); assert_eq!(guarantee.metric, ErrorMetric::Rank); assert_eq!( guarantee.bound.evaluate(), @@ -10894,7 +8069,7 @@ mod tests { ))); // The sketch *state* node carries no guarantee; the exact // accumulator's state is its value and does. - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &root.operator else { panic!() }; assert!(summary_input.guarantee.is_none()); @@ -10905,16 +8080,19 @@ mod tests { } #[test] - fn exact_sum_can_consume_an_approximate_readout() { + fn exact_sum_can_consume_an_approximate_evaluation() { // sum(count_distinct by (job) (m)) is an outer exact summary over - // the inner HLL readout. Both summary levels remain explicit. + // the inner HLL evaluation. Both summary levels remain explicit. let inner = agg(vec![2], default_cardinality(), metric_scan(&["job"])); let outer = agg(vec![], AggIntent::Sum { col: None }, inner); let root = realize(&outer).unwrap(); - let SummaryExpr::SummaryAgg { child, .. } = &root.expr else { + let Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) = &root.operator else { panic!("outer exact sum should remain a SummaryAgg") }; - assert!(matches!(child.expr, SummaryExpr::SummaryEstimate { .. })); + assert!(matches!( + child.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); assert!(root.guarantee.is_some()); // count(...) over the same child is exact: a row count does not @@ -10935,72 +8113,32 @@ mod tests { } #[test] - fn equal_split_allocation_supports_nested_summary_readouts() { + fn equal_split_allocation_supports_nested_summary_evaluations() { // A registered rank-additive rule and valid budget split make both // summary levels explicit while preserving the composed guarantee. let inner = agg(vec![2], quantile_eps(0.5, 0.1), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], quantile_eps(0.99, 0.1), inner)); - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs( - &DefaultCostModel, - &RankAdditiveModel, - &EqualSplitAllocator, - ); + let outer = agg(vec![], quantile_eps(0.99, 0.1), inner); + let strategy = + ASAPStrategies::new_with_planning_inputs(&RankAdditiveModel, &EqualSplitAllocator); let proposals = strategy.propose(&TargetSubDAG::new(&outer)); assert!(!proposals.candidates.is_empty()); assert!(proposals.candidates.iter().all(|candidate| { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return false; }; - matches!(node.expr, SummaryExpr::SummaryEstimate { .. }) - && node.guarantee.as_ref().is_some_and(|guarantee| { - DefaultAccuracyModel.satisfies(guarantee, &AccuracyTarget::Epsilon(0.1)) - }) - })); - } - - #[test] - fn global_selection_can_choose_nested_summaries() { - // The same nested summary remains available through workload search - // and global cost ranking. - let inner = agg(vec![2], quantile_eps(0.5, 0.1), metric_scan(&["job"])); - let outer = Rc::new(agg(vec![], quantile_eps(0.99, 0.1), inner)); - let strategies: Vec> = - vec![Box::new(SketchAlgorithmStrategy::new_with_planning_inputs( - &DefaultCostModel, - &RankAdditiveModel, - &EqualSplitAllocator, - ))]; - let space = search_workload_with(vec![("q", Rc::clone(&outer))], &strategies); - let root = &space.roots[0].1; - let group = space.candidates_for_target(root).unwrap(); - assert!(!group.rejected.is_empty()); - assert!(group.candidates.iter().all(|c| match &c.replacement { - Replacement::Summary(node) => node.guarantee.as_ref().is_some_and(|g| { - DefaultAccuracyModel.satisfies(g, &AccuracyTarget::Epsilon(0.1)) - }), - Replacement::Rewrite(_) => false, - Replacement::ExactComposition(_) => false, + matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + ) && node.guarantee.as_ref().is_some_and(|guarantee| { + DefaultAccuracyModel.satisfies(guarantee, &AccuracyTarget::Epsilon(0.1)) + }) })); - let ranked = space.cost_sorted(&DefaultCostModel); - let root_ranked = ranked.iter().find(|g| Rc::ptr_eq(g.target, root)).unwrap(); - assert_eq!(root_ranked.candidates.len(), group.candidates.len()); - - let selection = space.global_selection(&DefaultCostModel); - let chosen = selection - .for_target(root) - .unwrap() - .chosen - .expect("a nested summary candidate wins"); - let Replacement::Summary(node) = &chosen.replacement else { - panic!() - }; - assert!(matches!(node.expr, SummaryExpr::SummaryEstimate { .. })); } #[test] fn root_target_check_removes_candidates_before_cost_ranking() { - let q = Rc::new(agg(vec![2], default_quantile(0.99), metric_scan(&["job"]))); + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); // A root target tighter than the node's own ε=0.01: every sketch // candidate misses it and is moved to `rejected`; nothing is left // for the cost model to rank. @@ -11014,14 +8152,12 @@ mod tests { assert!(group .candidates .iter() - .all(|c| matches!(c.replacement, Replacement::Rewrite(_)))); + .all(|c| matches!(&c.replacement, Replacement::SubDAG(n) if !n.contains_asap()))); assert!(group.rejected.iter().all(|r| matches!( r.error, AccuracyError::TargetNotSatisfied { target: AccuracyTarget::Epsilon(e), .. } if e == 0.001 ))); assert!(group.rejected.len() >= 2); - let selection = space.global_selection(&DefaultCostModel); - assert!(selection.for_target(root).unwrap().chosen.is_none()); // A root target the node's own sizing meets keeps every candidate. let space = search_workload_with_targets( @@ -11033,7 +8169,7 @@ mod tests { assert!(group .candidates .iter() - .any(|c| matches!(c.replacement, Replacement::Summary(_)))); + .any(|c| matches!(&c.replacement, Replacement::SubDAG(n) if n.contains_asap()))); // An `Exact` root target admits only exact candidates. let space = search_workload_with_targets( @@ -11043,11 +8179,11 @@ mod tests { ); let group = space.candidates_for_target(&space.roots[0].1).unwrap(); assert!(group.candidates.iter().all(|c| match &c.replacement { - Replacement::Summary(node) => node + Replacement::SubDAG(node) if node.contains_asap() => node .guarantee .as_ref() .is_some_and(ResultGuarantee::is_exact), - Replacement::Rewrite(_) => true, + Replacement::SubDAG(_) => true, Replacement::ExactComposition(_) => false, })); } @@ -11061,14 +8197,14 @@ mod tests { }, metric_scan(&["job"]), ); - let q = Rc::new(agg( + let q = agg( vec![], AggIntent::TopK { k: 10, accuracy: AccuracyTarget::Epsilon(0.01), }, inner, - )); + ); let space = search_workload_with_targets( vec![("q", Rc::clone(&q), Some(AccuracyTarget::Epsilon(0.01)))], &default_strategies(), @@ -11078,14 +8214,14 @@ mod tests { assert!(group.candidates.iter().any(|candidate| matches!( &candidate.replacement, - Replacement::Summary(node) if node.guarantee.as_ref().is_some_and(ResultGuarantee::has_unknown) + Replacement::SubDAG(node) if node.guarantee.as_ref().is_some_and(ResultGuarantee::has_unknown) ))); let candidate = group .candidates .iter() .find(|candidate| candidate.has_missing_accuracy_evidence()) .unwrap(); - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { unreachable!() }; let exported = asap_types::dag_export::export_summary(node); @@ -11093,29 +8229,19 @@ mod tests { .guarantee .as_ref() .is_some_and(ResultGuarantee::has_unknown)); - let selected = space.global_selection(&DefaultCostModel); - assert!(!selected - .for_target(&space.roots[0].1) - .unwrap() - .chosen - .is_some_and(ReplacementSubDAG::has_missing_accuracy_evidence)); - assert!(selected - .assemble_selected_dag(&space.roots[0].1) - .unwrap() - .is_some()); } // Source evidence alone must enable Planner-owned sizing and certification. #[test] fn scoped_hll_evidence_sizes_and_certifies_without_a_deployment_model() { use crate::accuracy::EstimatorContract; struct SourceEvidence { - expression: QueryExpr, + expression: OperatorNode, max_distinct: u32, } impl AccuracyEvidenceProvider for SourceEvidence { - fn estimator_contract(&self, expression: &QueryExpr) -> Option { + fn estimator_contract(&self, expression: &OperatorNode) -> Option { (expression == &self.expression).then_some(EstimatorContract::ClassicHll { - max_distinct_per_readout: self.max_distinct, + max_distinct_per_evaluation: self.max_distinct, }) } } @@ -11123,20 +8249,19 @@ mod tests { epsilon: 0.05, delta: 0.01, }; - let root = Rc::new(agg( + let root = agg( vec![], AggIntent::Cardinality { cols: vec![], accuracy: target.clone(), }, metric_scan(&[]), - )); + ); let evidence = SourceEvidence { expression: (*root).clone(), max_distinct: 128, }; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, @@ -11145,7 +8270,7 @@ mod tests { let hll = candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Summary(node) + Replacement::SubDAG(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll => { Some(node) @@ -11155,13 +8280,13 @@ mod tests { .expect("HLL candidate"); assert!(DefaultAccuracyModel .satisfies(hll.guarantee.as_ref().expect("HLL confidence"), &target)); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &hll.expr else { - panic!("readout") + let Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = &hll.operator else { + panic!("evaluation") }; - let SummaryExpr::SummaryAgg { + let Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &summary_input.expr + }) = &summary_input.operator else { panic!("HLL state") }; @@ -11175,9 +8300,8 @@ mod tests { precision: expected } ); - let absent = - SketchAlgorithmStrategy::default_cost_model().replacements(&TargetSubDAG::new(&root)); - assert!(!absent.iter().any(|candidate| matches!(&candidate.replacement, Replacement::Summary(node) + let absent = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); + assert!(!absent.iter().any(|candidate| matches!(&candidate.replacement, Replacement::SubDAG(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll && node.guarantee.as_ref().is_some_and(|g| DefaultAccuracyModel.satisfies(g, &target))))); // Invalid contracts, infeasible targets and evidence for another source // must never authorize a confidence-bearing HLL candidate. @@ -11191,94 +8315,47 @@ mod tests { epsilon: 0.05, delta, }; - let query = Rc::new(agg( + let query = agg( vec![], AggIntent::Cardinality { cols: vec![], accuracy: target.clone(), }, metric_scan(&[]), - )); + ); let evidence = SourceEvidence { expression: if wrong_scope { - metric_scan(&["other"]) + (*metric_scan(&["other"])).clone() } else { (*query).clone() }, max_distinct, }; - let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &DefaultCostModel, + let strategy = ASAPStrategies::new_with_planning_inputs_and_evidence( &DefaultAccuracyModel, &EqualSplitAllocator, &evidence, ); assert!(!strategy.replacements(&TargetSubDAG::new(&query)).iter().any(|candidate| - matches!(&candidate.replacement, Replacement::Summary(node) + matches!(&candidate.replacement, Replacement::SubDAG(node) if summary_family_algorithm(node) == SketchAlgorithm::Hll && node.guarantee.as_ref().is_some_and(|g| DefaultAccuracyModel.satisfies(g, &target))))); } } - // A value projection cannot consume an opaque exact accumulator edge. - #[test] - fn residual_projection_finalizes_selected_exact_state() { - let inner = Rc::new(agg(vec![], AggIntent::Sum { col: None }, metric_scan(&[]))); - let root = Rc::new(QueryExpr::Project { - cols: vec![asap_types::pre_asap::ProjectItem { - expr: QueryExpr::Column(0), - alias: Some("result".into()), - }], - qualifier: None, - child: inner.clone(), - }); - let space = search_workload_with_targets( - vec![("q", root.clone(), Some(AccuracyTarget::Exact))], - &default_strategies(), - &DefaultAccuracyModel, - ); - let selected = space.global_selection(&DefaultCostModel); - selected - .assembled_nodes - .borrow_mut() - .insert(Rc::as_ptr(&inner), realize(inner.as_ref()).unwrap()); - let node = selected.assemble_target(&root).unwrap(); - let SummaryExpr::ValueOperation { - child, - operation: ValueOperation::Project { .. }, - .. - } = &node.expr - else { - panic!("expected Project"); - }; - assert!(matches!( - child.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::FinalizeExactAccumulator, - .. - } - )); - assert!(child - .schema - .fields - .iter() - .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); - } // A numeric group key must not be mistaken for the ranked aggregate score. #[test] fn ranking_uses_aggregate_output_position_not_first_numeric_column() { let logical = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["id"])); - let mut values = lift(&logical.output_schema().unwrap()); + let mut values = logical.schema.clone(); values.fields[0].dtype = FieldDataType::Plain(DataType::Int64); assert_eq!(ranking_score_index(&logical, &values).unwrap(), 1); } // A heap's key schema is derived from its encoded item, not all label columns. #[test] - fn heap_readout_preserves_numeric_item_identity() { - let mut raw = metric_scan(&["id", "description"]); - let QueryExpr::Scan { schema, .. } = &mut raw else { - unreachable!() - }; + fn heap_evaluation_preserves_numeric_item_identity() { + let mut schema = metric_scan(&["id", "description"]).schema.clone(); schema.fields[2].dtype = FieldDataType::Plain(DataType::Int64); + let raw = crate::test_support::scan("m", schema); let node = agg( vec![], AggIntent::TopK { @@ -11288,7 +8365,7 @@ mod tests { agg(vec![2], AggIntent::Sum { col: None }, raw.clone()), ); let input = PhysicalSummaryInput { - child: Rc::new(raw), + child: raw, input: SummaryUpdate { item: Some(SummaryInputExpr::Column(ColumnRef::Named("id".into()))), weight: SummaryInputExpr::Constant(1.0), @@ -11297,7 +8374,7 @@ mod tests { }, }, }; - let schema = keyed_heap_readout_schema(&input, &node).unwrap(); + let schema = keyed_heap_evaluation_schema(&input, &node).unwrap(); assert_eq!( schema .fields @@ -11311,4 +8388,132 @@ mod tests { FieldDataType::Plain(DataType::Int64) ); } + + // Every SummaryAgg a strategy proposes declares whole-source coverage of + // the one source it reads (trusted, #570). + #[test] + fn proposed_summary_states_cover_their_whole_source() { + let root = agg( + vec![], + AggIntent::Quantile { + q: 0.9, + col: None, + accuracy: AccuracyTarget::Epsilon(0.01), + }, + metric_scan(&["job"]), + ); + let source = Source::TimeSeries { metric: "m".into() }; + let proposals = ASAPStrategies::default().propose(&TargetSubDAG::new(&root)); + let states: Vec<_> = proposals + .candidates + .iter() + .filter_map(|candidate| match &candidate.replacement { + Replacement::SubDAG(node) => Some(node), + _ => None, + }) + .flat_map(OperatorNode::reachable) + .filter(|node| matches!(node.asap(), Some(ASAPOp::SummaryAgg { .. }))) + .collect(); + assert!(!states.is_empty()); + for state in states { + let coverage = state.coverage.as_ref().expect("summary state has coverage"); + assert_eq!(coverage.source, source); + assert_eq!( + coverage.regions, + [CoverageRegion { + time_ms: None, + population: Default::default(), + }] + ); + } + } + + // Whole-source coverage names one source; over two it is not declared. + #[test] + fn whole_source_coverage_needs_exactly_one_source() { + let left = metric_scan(&["job"]); + let right = crate::test_support::scan("n", left.schema.clone()); + assert!(whole_source_coverage(&left).is_some()); + let join = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: asap_types::ir::operator::operator_properties::JoinKind::Inner, + pred: equi_pred(0, 2), + left, + right, + })) + .unwrap(); + assert_eq!(whole_source_coverage(&join), None); + } + + #[test] + fn relational_join_predicate_requires_and_normalizes_cross_input_columns() { + let forward = normalize_cross_input_equi_predicate(&equi_pred(1, 3), 2, 4) + .expect("left-to-right equality"); + let reverse = normalize_cross_input_equi_predicate(&equi_pred(3, 1), 2, 4) + .expect("right-to-left equality"); + assert_eq!(forward, reverse, "reverse equality must be canonicalized"); + assert!(normalize_cross_input_equi_predicate(&equi_pred(0, 1), 2, 4).is_none()); + assert!(normalize_cross_input_equi_predicate(&equi_pred(0, 4), 2, 4).is_none()); + } + + // A value projection cannot consume an opaque exact accumulator edge. + #[test] + fn residual_projection_finalizes_selected_exact_state() { + let inner = agg(vec![], AggIntent::Sum { col: None }, metric_scan(&[])); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols: vec![ProjectItem { + expr: ScalarExpr::Column(0), + alias: Some("result".into()), + }], + qualifier: None, + child: inner.clone(), + })) + .unwrap(); + let space = search_workload_with_targets( + vec![("q", root.clone(), Some(AccuracyTarget::Exact))], + &default_strategies(), + &DefaultAccuracyModel, + ); + // No site has a chosen candidate, so the root is assembled as a residual. + let groups = space + .target_subdag_candidates() + .map(|group| { + ( + Rc::as_ptr(&group.target), + TargetSubDAGSelection { + target: &group.target, + consumer_count: group.consumer_count, + effective_consumer_count: group.consumer_count, + chosen: None, + }, + ) + }) + .collect(); + let selected = GlobalSelection::new(space.order.clone(), groups, HashMap::new()); + // CSE re-interns the workload, so the space's root/child `Rc`s are not + // the fixture's. Assembly only assembles children that are discovered + // targets, so seed the memo under the space's own child pointer. + let root = Rc::clone(&space.roots[0].1); + let Some(NonASAPOp::Project { child: inner, .. }) = root.non_asap() else { + unreachable!() + }; + assert!(space.candidates_for_target(inner).is_some()); + selected + .assembled_nodes + .borrow_mut() + .insert(Rc::as_ptr(inner), realize(inner.as_ref()).unwrap()); + let node = selected.assemble_target(&root).unwrap(); + let Operator::NonASAP(NonASAPOp::Project { child, .. }) = &node.operator else { + panic!("expected Project"); + }; + assert!(matches!( + child.operator, + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { .. }) + )); + assert!(child + .schema + .fields + .iter() + .all(|field| matches!(field.dtype, FieldDataType::Plain(_)))); + } } diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/logical-optimizer/src/pass1/rewrite.rs similarity index 70% rename from crates/asap-aware-mapping/src/rewrite.rs rename to crates/logical-optimizer/src/pass1/rewrite.rs index 94a3d638c..d8a1afaa7 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/logical-optimizer/src/pass1/rewrite.rs @@ -35,7 +35,7 @@ //! - **`without(...)` grouping** leaves an `Aggregate`'s own output schema //! *open* (`closed: false`, see `without_output_schema`), while the //! `Project` this strategy always wraps the rewrite in forces -//! `closed: true` (see `QueryExpr::output_schema`'s `Project` arm). Under +//! `closed: true` (see `NonASAPOp::output_schema`'s `Project` arm). Under //! `without(...)` the rewritten form's `closed` flag would silently flip //! relative to the original — exactly the kind of schema drift this //! module exists to avoid. @@ -43,7 +43,7 @@ //! Both are follow-ups (issue #253 itself scopes to "the concrete case in //! Peilin's comment"), not correctness bugs in what ships here — a node //! outside this scope simply doesn't `match`, the same "safe but -//! uninformative" fallback [`SketchAlgorithmStrategy`]/[`SharedSubDAGStrategy`] +//! uninformative" fallback [`ASAPStrategies`]/[`SharedSubDAGStrategy`] //! already use for shapes they don't have an opinion on. //! //! ## Non-goals (mirrors [`replacement`]'s own discipline) @@ -57,32 +57,55 @@ //! the rewritten form is actually worth picking, by letting the original //! and rewritten forms compete on cost — not this strategy. +use asap_types::ir::operator::non_asap::any_measure_filtered; use std::rc::Rc; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::expr_ir::ArithmeticOpKind; -use asap_types::pre_asap::query_expr::{ - any_measure_filtered, BinaryOpKind, ProjectItem, QueryExpr, Reduction, -}; -use asap_types::pre_asap::schema::{ColumnId, DataType}; +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::{BinaryOpKind, Reduction}; +use asap_types::ir::scalar::ArithmeticOpKind; +use asap_types::ir::schema::{ColumnId, DataType}; +use asap_types::ir::{BinaryOperator, NonASAPOp, OperatorNode, ProjectItem, ScalarExpr}; + use asap_types::types::AccuracyTarget; -use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; +use crate::pass1::replacement::{ + Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +}; /// The shape [`AvgToSumOverCountStrategy`] rewrites: a single `Avg{col}` /// measure, no `HAVING`, grouped with an ordinary `by(...)` reduction (see /// the module docs' "Scope" for why `without(...)`/`PerEntity` are /// excluded). Returns the grouping key count and the summed column so /// [`build_rewrite`] doesn't have to re-match. -fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { - let QueryExpr::Aggregate { +/// `a / b` with PromQL arithmetic semantics and no vector matching. +fn arithmetic( + op: ArithmeticOpKind, + lhs: Rc, + rhs: Rc, +) -> Option> { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Arithmetic(op), + vector_match: None, + }, + return_bool: false, + lhs, + rhs, + })) + .ok() +} + +fn avg_rewrite_target(node: &OperatorNode) -> Option<(usize, Option)> { + let Some(NonASAPOp::Aggregate { reduction, measures, filters, having: None, child, .. - } = node + }) = node.non_asap() else { return None; }; @@ -102,7 +125,7 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { // therefore be decomposed through it only when the averaged input is // provably non-null; otherwise NULL rows would incorrectly contribute to // the denominator. - let input_schema = child.output_schema().ok()?; + let input_schema = &child.schema; let value_col = col .or_else(|| input_schema.column_id("value")) .or_else(|| (0..input_schema.fields.len()).find(|i| !by.contains(i)))?; @@ -129,21 +152,21 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { /// exactly regardless of the summed column's own type (integer division /// would otherwise silently reappear whenever the input column is itself /// integer-typed: `Sum`'s output type tracks its input, `Count`'s is always -/// `Int64`, and `QueryExpr::output_schema`'s own `Arithmetic` type inference +/// `Int64`, and `ScalarExpr::scalar_type`'s own `Arithmetic` type inference /// types a `Div` of two `Int64` operands as `Int64` — the explicit operand /// `Cast` is what keeps both the division and rewritten `avg` column /// `Float64` the way the original always was, not an incidental extra step). // These are conditional physical components, never an unconditional Rewrite. // The caller must attach the finite-division execution guard before admission. -pub(crate) fn temporal_average_components(root: &Rc) -> Option> { - let QueryExpr::Aggregate { +pub(crate) fn temporal_average_components(root: &Rc) -> Option> { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures, filters, child, having: None, .. - } = root.as_ref() + }) = root.non_asap() else { return None; }; @@ -153,10 +176,10 @@ pub(crate) fn temporal_average_components(root: &Rc) -> Option) -> Option) -> Option> { +fn build_rewrite(root: &Rc) -> Option> { let (group_count, col) = avg_rewrite_target(root)?; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, output_names, child, .. - } = root.as_ref() + }) = root.non_asap() else { unreachable!("avg_rewrite_target already confirmed an Aggregate shape"); }; // The original `avg` column's own name: `output_names[0]` if the // producing front end overrode it (SQL threading DataFusion's own - // generated name — see `QueryExpr::Aggregate::output_names`'s docs), + // generated name — see `NonASAPOp::Aggregate::output_names`'s docs), // else `AggIntent::Avg`'s synthetic default. Either way this is the // *only* thing about the original output column this rewrite needs to // reproduce — `AggIntent::Avg::output_column`'s `(Float64, nullable: @@ -210,77 +233,78 @@ fn build_rewrite(root: &Rc) -> Option> { .cloned() .unwrap_or_else(|| "avg".to_string()); - let sum_agg = Rc::new(QueryExpr::Aggregate { - reduction: reduction.clone(), - measures: vec![AggIntent::Sum { col }], - output_names: Vec::new(), - filters: vec![], - having: None, - child: Rc::clone(child), - }); - let count_agg = Rc::new(QueryExpr::Aggregate { - reduction: reduction.clone(), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: Vec::new(), - filters: vec![], - having: None, - child: Rc::clone(child), - }); + let sum_agg = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: reduction.clone(), + measures: vec![AggIntent::Sum { col }], + output_names: Vec::new(), + filters: vec![], + having: None, + child: Rc::clone(child), + })) + .ok()?; + let count_agg = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: reduction.clone(), + measures: vec![AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }], + output_names: Vec::new(), + filters: vec![], + having: None, + child: Rc::clone(child), + })) + .ok()?; let sum_idx = group_count; let mut cols: Vec = (0..group_count) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: ScalarExpr::Column(i), }) .collect(); cols.push(ProjectItem { alias: Some(avg_name), - expr: QueryExpr::Cast { - expr: Rc::new(QueryExpr::Column(sum_idx)), + expr: ScalarExpr::Cast { + expr: Box::new(ScalarExpr::Column(sum_idx)), to: DataType::Float64, try_cast: false, }, }); - let float_sum = Rc::new(QueryExpr::Project { - cols, - qualifier: None, - child: sum_agg, - }); - Some(Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: float_sum, - rhs: count_agg, - vector_match: None, - })) + let float_sum = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols, + qualifier: None, + child: sum_agg, + })) + .ok()?; + arithmetic(ArithmeticOpKind::Div, float_sum, count_agg) } /// Compose adjacent per-entity and cross-entity accumulators when their /// algebra, rather than a query-language spelling, proves equivalence. -pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { - let original_schema = root.output_schema().ok()?; - let QueryExpr::Aggregate { +pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option> { + let original_schema = &root.schema; + let Some(NonASAPOp::Aggregate { reduction: outer_reduction @ Reduction::Reduce(_), measures: outer_measures, output_names, filters: outer_filters, having: None, child, - } = root.as_ref() + }) = root.non_asap() else { return None; }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: inner_measures, filters: inner_filters, having: None, child: inner_child, .. - } = child.as_ref() + }) = child.non_asap() else { return None; }; @@ -297,14 +321,16 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option inner.clone(), _ => return None, }; - let aggregate = Rc::new(QueryExpr::Aggregate { - reduction: outer_reduction.clone(), - measures: vec![composed], - output_names: output_names.clone(), - filters: vec![], - having: None, - child: Rc::clone(inner_child), - }); + let aggregate = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: outer_reduction.clone(), + measures: vec![composed], + output_names: output_names.clone(), + filters: vec![], + having: None, + child: Rc::clone(inner_child), + })) + .ok()?; // The outer Sum sees PromQL's Float64 sample value, whereas the composed // Count accumulator is Int64. Keep the original observable type. @@ -321,7 +347,7 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option = (0..by.keys().len()) .map(|i| ProjectItem { alias: None, - expr: QueryExpr::Column(i), + expr: ScalarExpr::Column(i), }) .collect(); cols.push(ProjectItem { @@ -332,17 +358,18 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option) -> Option) -> Option QueryExpr { - let mut columns = vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ]; - columns.extend( - labels - .iter() - .map(|n| Field::plain(*n, DataType::Utf8, true)), - ); - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index(columns, 0, vec![]), - } + use crate::test_support::metric_scan; + use asap_types::ir::TimeRangeKind; + + fn avg_agg( + by: Vec, + col: Option, + child: Rc, + ) -> Rc { + avg_agg_with(by, col, vec![], None, child) } - fn avg_agg(by: Vec, col: Option, child: QueryExpr) -> QueryExpr { - QueryExpr::Aggregate { + fn avg_agg_with( + by: Vec, + col: Option, + output_names: Vec, + having: Option, + child: Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![AggIntent::Avg { col }], - output_names: vec![], + output_names, filters: vec![], - having: None, - child: Rc::new(child), - } + having, + child, + })) + .unwrap() } // Temporal averages expose two single-measure children without closing labels. #[test] fn temporal_average_components_preserves_schema_and_exposes_sum_count() { - let root = Rc::new(lower_promql( - "avg_over_time(a{job=\"api\"}[5m])", - AccuracyTarget::Exact, - )); + let root = lower_promql("avg_over_time(a{job=\"api\"}[5m])", AccuracyTarget::Exact); assert!(SemanticEquivalentRewriteStrategy .replacements(&TargetSubDAG::new(&root)) .is_empty()); let rewritten = temporal_average_components(&root).expect("conditional sum/count components"); - assert_eq!( - root.output_schema().unwrap(), - rewritten.output_schema().unwrap() - ); - assert!(matches!(rewritten.as_ref(), QueryExpr::BinaryOp { .. })); + assert_eq!(root.schema.clone(), rewritten.schema.clone()); + assert!(matches!( + rewritten.non_asap(), + Some(NonASAPOp::BinaryOp { .. }) + )); } // ── matches ────────────────────────────────────────────────────────── #[test] fn matches_a_bare_avg_aggregate() { - let q = Rc::new(avg_agg(vec![], None, metric_scan(&[]))); + let q = avg_agg(vec![], None, metric_scan(&[])); let target = TargetSubDAG::new(&q); assert!(AvgToSumOverCountStrategy.matches(&target)); } #[test] fn matches_a_grouped_avg_aggregate() { - let q = Rc::new(avg_agg(vec![2], None, metric_scan(&["job"]))); + let q = avg_agg(vec![2], None, metric_scan(&["job"])); let target = TargetSubDAG::new(&q); assert!(AvgToSumOverCountStrategy.matches(&target)); } #[test] fn does_not_match_a_multi_measure_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![AggIntent::Sum { col: None }, AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -494,13 +520,15 @@ mod tests { #[test] fn does_not_match_a_having_bearing_avg_aggregate() { - let mut q = avg_agg(vec![2], None, metric_scan(&["job"])); - if let QueryExpr::Aggregate { having, .. } = &mut q { - *having = Some(asap_types::pre_asap::query_expr::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::expr_ir::ScalarValue::Boolean(true)), - ))); - } - let q = Rc::new(q); + let q = avg_agg_with( + vec![2], + None, + vec![], + Some(asap_types::ir::Predicate(ScalarExpr::Literal( + asap_types::ir::scalar::ScalarValue::Boolean(true), + ))), + metric_scan(&["job"]), + ); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -515,14 +543,16 @@ mod tests { }, AggIntent::Min { col: None }, ] { - let q = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![intent.clone()], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(metric_scan(&["job"])), - }); + let q = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![intent.clone()], + output_names: vec![], + filters: vec![], + having: None, + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!( !AvgToSumOverCountStrategy.matches(&target), @@ -534,16 +564,17 @@ mod tests { #[test] fn does_not_match_a_without_grouped_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(asap_types::pre_asap::query_expr::GroupKeys::without( - vec![2], - )), + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::Reduce( + asap_types::ir::operator::operator_properties::GroupKeys::without(vec![2]), + ), measures: vec![AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&["job"])), - }); + child: metric_scan(&["job"]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -551,14 +582,15 @@ mod tests { #[test] fn does_not_match_a_per_entity_avg_aggregate() { - let q = Rc::new(QueryExpr::Aggregate { + let q = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::PerEntity, measures: vec![AggIntent::Avg { col: None }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(metric_scan(&[])), - }); + child: metric_scan(&[]), + })) + .unwrap(); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -566,7 +598,7 @@ mod tests { #[test] fn does_not_match_a_non_aggregate_node() { - let scan = Rc::new(metric_scan(&["job"])); + let scan = metric_scan(&["job"]); let target = TargetSubDAG::new(&scan); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); @@ -579,7 +611,7 @@ mod tests { #[test] fn avg_rewrites_and_schema_matches_exactly_when_ungrouped() { let original = avg_agg(vec![], None, metric_scan(&[])); - let original_rc = Rc::new(original.clone()); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); @@ -587,28 +619,26 @@ mod tests { assert!(!replacements[0].rationale.is_empty()); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = rewritten.non_asap() else { panic!("expected sum/count BinaryOp, got {rewritten:?}"); }; - let QueryExpr::Project { child: sum, .. } = lhs.as_ref() else { + let Some(NonASAPOp::Project { child: sum, .. }) = lhs.non_asap() else { panic!("expected cast Project above Sum, got {lhs:?}"); }; - assert!(matches!( - sum.as_ref(), - QueryExpr::Aggregate { measures, .. } + assert!(matches!(sum.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Sum { col: None }]) )); - assert!(matches!( - rhs.as_ref(), - QueryExpr::Aggregate { measures, .. } + assert!(matches!(rhs.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if matches!(measures.as_slice(), [AggIntent::Count { accuracy: AccuracyTarget::Exact }]) )); - let original_schema = original.output_schema().unwrap(); - let rewritten_schema = rewritten.output_schema().unwrap(); + let original_schema = original.schema.clone(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!( original_schema, rewritten_schema, "the rewritten DAG must report exactly the same output schema as the original avg" @@ -620,20 +650,22 @@ mod tests { /// synthetic `"avg"` default. #[test] fn preserves_an_explicit_output_name_override() { - let mut q = avg_agg(vec![], None, metric_scan(&[])); - if let QueryExpr::Aggregate { output_names, .. } = &mut q { - *output_names = vec!["avg_latency".to_string()]; - } - let original_schema = q.output_schema().unwrap(); - let q = Rc::new(q); + let q = avg_agg_with( + vec![], + None, + vec!["avg_latency".to_string()], + None, + metric_scan(&[]), + ); + let original_schema = q.schema.clone(); let target = TargetSubDAG::new(&q); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!(original_schema, rewritten_schema); assert_eq!(rewritten_schema.fields[0].name, "avg_latency"); } @@ -643,36 +675,36 @@ mod tests { #[test] fn grouped_avg_rewrite_preserves_the_whole_schema() { let original = avg_agg(vec![2], None, metric_scan(&["job"])); - let original_schema = original.output_schema().unwrap(); - let original_rc = Rc::new(original); + let original_schema = original.schema.clone(); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); assert_eq!(rewritten_schema, original_schema); } #[test] fn default_search_discovers_bindable_sum_and_count_targets() { - let root = Rc::new(avg_agg(vec![2], None, metric_scan(&["job"]))); - let space = crate::replacement::search_workload(vec![("avg", Rc::clone(&root))]); + let root = avg_agg(vec![2], None, metric_scan(&["job"])); + let space = crate::pass1::replacement::search_workload(vec![("avg", Rc::clone(&root))]); let avg_group = space .candidates_for_target(&space.roots[0].1) .expect("avg group"); assert!(avg_group.candidates.iter().any(|candidate| { - candidate.provenance == crate::replacement::ReplacementProvenance::LogicalRewrite + candidate.provenance == crate::pass1::replacement::ReplacementProvenance::LogicalRewrite })); let mut found_sum = false; let mut found_count = false; for group in space.target_subdag_candidates() { - let QueryExpr::Aggregate { measures, .. } = group.target.as_ref() else { + let Some(NonASAPOp::Aggregate { measures, .. }) = group.target.non_asap() else { continue; }; let expected = matches!(measures.as_slice(), [AggIntent::Sum { .. }]) @@ -689,7 +721,8 @@ mod tests { group .candidates .iter() - .any(|candidate| matches!(candidate.replacement, Replacement::Summary(_))), + .any(|candidate| matches!(&candidate.replacement, + Replacement::SubDAG(node) if node.contains_asap())), "rewritten accumulator must be independently bindable: {measures:?}" ); found_sum |= matches!(measures.as_slice(), [AggIntent::Sum { .. }]); @@ -710,25 +743,26 @@ mod tests { Field::plain("job", DataType::Utf8, true), Field::plain("bytes", DataType::Int64, false), ]; - let child = QueryExpr::Scan { + let child = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: { let cols = std::mem::take(&mut schema_cols); Schema::with_time_index(cols, 0, vec![]) }, - }; + })) + .unwrap(); let original = avg_agg(vec![1], Some(2), child); - let original_schema = original.output_schema().unwrap(); - let original_rc = Rc::new(original); + let original_schema = original.schema.clone(); + let original_rc = Rc::clone(&original); let target = TargetSubDAG::new(&original_rc); let replacements = AvgToSumOverCountStrategy.replacements(&target); let rewritten = match &replacements[0].replacement { - Replacement::Rewrite(rc) => rc, + Replacement::SubDAG(rc) => rc, other => panic!("expected a Rewrite replacement, got {other:?}"), }; - let rewritten_schema = rewritten.output_schema().unwrap(); + let rewritten_schema = rewritten.schema.clone(); // The whole reason for the explicit `Cast` in `build_rewrite`: an // `Int64` input column (`bytes`) makes `Sum`'s own output `Int64` @@ -741,15 +775,15 @@ mod tests { DataType::Float64 ); - let QueryExpr::BinaryOp { lhs, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::BinaryOp { lhs, .. }) = rewritten.non_asap() else { panic!("expected sum/count BinaryOp"); }; - let QueryExpr::Project { cols, .. } = lhs.as_ref() else { + let Some(NonASAPOp::Project { cols, .. }) = lhs.non_asap() else { panic!("expected cast Project above Sum"); }; assert!(matches!( &cols.last().unwrap().expr, - QueryExpr::Cast { + ScalarExpr::Cast { to: DataType::Float64, .. } @@ -758,7 +792,7 @@ mod tests { #[test] fn does_not_rewrite_avg_of_a_nullable_column_via_count_star() { - let child = QueryExpr::Scan { + let child = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -770,27 +804,34 @@ mod tests { 0, vec![], ), - }; - let q = Rc::new(avg_agg(vec![], Some(2), child)); + })) + .unwrap(); + let q = avg_agg(vec![], Some(2), child); let target = TargetSubDAG::new(&q); assert!(!AvgToSumOverCountStrategy.matches(&target)); assert!(AvgToSumOverCountStrategy.replacements(&target).is_empty()); } - fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { - let temporal = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![inner], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(metric_scan(&["service"])), - }), - }; - Rc::new(QueryExpr::Aggregate { + fn nested_aggregate(outer: AggIntent, inner: AggIntent) -> Rc { + let temporal = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![inner], + output_names: vec![], + filters: vec![], + having: None, + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + range: Duration::from_secs(300), + child: metric_scan(&["service"]), + }, + )) + .unwrap(), + })) + .unwrap(); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![2]), measures: vec![outer], // Match the PromQL front end: an empty entry selects the intent's @@ -798,8 +839,9 @@ mod tests { output_names: vec![String::new()], filters: vec![], having: None, - child: Rc::new(temporal), - }) + child: temporal, + })) + .unwrap() } #[test] @@ -820,34 +862,30 @@ mod tests { let [candidate] = candidates.as_slice() else { panic!("supported pair should produce exactly one rewrite") }; - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDAG(rewritten) = &candidate.replacement else { panic!("expected a logical rewrite") }; - assert_eq!( - original.output_schema().unwrap(), - rewritten.output_schema().unwrap() - ); - let aggregate = match rewritten.as_ref() { - QueryExpr::Aggregate { .. } => rewritten.as_ref(), - QueryExpr::Project { child, .. } => child.as_ref(), + assert_eq!(original.schema.clone(), rewritten.schema.clone()); + let aggregate = match rewritten.non_asap() { + Some(NonASAPOp::Aggregate { .. }) => rewritten.as_ref(), + Some(NonASAPOp::Project { child, .. }) => child.as_ref(), other => panic!("expected Aggregate or cast Project, got {other:?}"), }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), measures, child, .. - } = aggregate + }) = aggregate.non_asap() else { panic!("expected composed cross-entity aggregate") }; assert_eq!(by.keys(), &[2]); assert_eq!(measures, &[expected]); - assert!(matches!( - child.as_ref(), - QueryExpr::TimeRange { range, child } + assert!(matches!(child.non_asap(), + Some(NonASAPOp::TimeRange { range, child, .. }) if *range == Duration::from_secs(300) - && matches!(child.as_ref(), QueryExpr::Scan { .. }) + && matches!(child.non_asap(), Some(NonASAPOp::Scan { .. })) )); } } @@ -872,7 +910,7 @@ mod tests { accuracy: AccuracyTarget::Exact, }, ); - let space = crate::replacement::search_workload(vec![("sum-count", root)]); + let space = crate::pass1::replacement::search_workload(vec![("sum-count", root)]); let root = &space.roots[0].1; let group = space.candidates_for_target(root).expect("root memo group"); let candidate = group @@ -880,9 +918,9 @@ mod tests { .iter() .find(|candidate| candidate.strategy == "SemanticEquivalentRewriteStrategy") .expect("default search should run semantic rewrites"); - let Replacement::Rewrite(rewritten) = &candidate.replacement else { + let Replacement::SubDAG(rewritten) = &candidate.replacement else { panic!("expected logical rewrite") }; - assert_eq!(rewritten.output_schema().unwrap().fields[1].name, "sum"); + assert_eq!(rewritten.schema.clone().fields[1].name, "sum"); } } diff --git a/crates/asap-aware-mapping/src/rollup.rs b/crates/logical-optimizer/src/pass1/rollup.rs similarity index 84% rename from crates/asap-aware-mapping/src/rollup.rs rename to crates/logical-optimizer/src/pass1/rollup.rs index ab6eff3d9..5b8bb59e4 100644 --- a/crates/asap-aware-mapping/src/rollup.rs +++ b/crates/logical-optimizer/src/pass1/rollup.rs @@ -20,7 +20,7 @@ //! "Rolling up aggregations on a fine-grained group by to get a //! coarse-grained group by (like AHA)," alongside "CSE across aggregations, //! and group by key management" — this strategy is the *cross-aggregate* -//! sibling of `pre_asap::cse::share_common_sub_dags`'s *identical*-sub-DAG +//! sibling of `ir::cse::share_common_sub_dags`'s *identical*-sub-DAG //! sharing: CSE shares two structurally-*equal* aggregates onto one `Rc`; //! this strategy relates two structurally-*different* (differently grouped) //! aggregates over the same shared source. @@ -66,7 +66,7 @@ //! //! ## `ColumnId` comparability — only sound for identical child IR //! -//! A `ColumnId` is a *position* into a specific `Schema` (`crates/types/src/pre_asap/schema.rs`'s +//! A `ColumnId` is a *position* into a specific `Schema` (`crates/types/src/ir/schema/mod.rs`'s //! own doc: "the same edge, the same schema, the same positional numbering"). //! Comparing the coarser aggregate's `by` positions against the finer //! aggregate's `by` positions is only meaningful when both aggregates have @@ -87,8 +87,8 @@ //! own set through [`RollupStrategy::new`]. //! - **No materialized roll-up operator.** Actually building a pre-aggregated //! summary/scan leaf at execution time is separate, larger work outside -//! `asap-aware-mapping`'s scope (see issue #254's own "Non-goal" section) -//! — this module only constructs the pre-ASAP [`QueryExpr::Aggregate`] +//! `asap-logical-optimizer`'s scope (see issue #254's own "Non-goal" section) +//! — this module only constructs the pre-ASAP `NonASAPOp::Aggregate` //! rewrite; a `CostModel`/search engine decides whether to prefer it. //! - **No cross-schema reconciliation** (see "`ColumnId` comparability" //! above) and **no `without(...)` grouping support** — `without`'s kept @@ -97,34 +97,39 @@ //! against a superset/subset relationship at all; [`is_legal_rollup_source`] //! declines both directions. +use asap_types::ir::operator::non_asap::any_measure_filtered; use std::collections::HashSet; use std::rc::Rc; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, GroupKeys, QueryExpr, Reduction}; -use asap_types::pre_asap::schema::{ColumnId, Schema}; +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::{GroupKeys, Reduction}; +use asap_types::ir::schema::{ColumnId, Schema}; +use asap_types::ir::{NonASAPOp, OperatorNode}; + use asap_types::types::AccuracyTarget; -use crate::replacement::{Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG}; +use crate::pass1::replacement::{ + Replacement, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, +}; /// The `(by, intent, child)` shape this strategy operates on: a single /// measure, no `HAVING` — the same bindable shape -/// [`crate::replacement::SketchAlgorithmStrategy`] requires (see that module's +/// [`crate::pass1::replacement::ASAPStrategies`] requires (see that module's /// private `bindable_intent`) — **plus** a genuine [`Reduction::Reduce`] /// grouping to compare (not [`Reduction::PerEntity`], which has no `by` set /// at all). `None` for anything else, including a multi-measure or `HAVING` /// aggregate, a non-`Aggregate` node, or a `PerEntity` reduction. fn bindable_grouped_aggregate( - node: &QueryExpr, -) -> Option<(&GroupKeys, &AggIntent, &Rc)> { - let QueryExpr::Aggregate { + node: &OperatorNode, +) -> Option<(&GroupKeys, &AggIntent, &Rc)> { + let Some(NonASAPOp::Aggregate { reduction, measures, filters, having, child, .. - } = node + }) = node.non_asap() else { return None; }; @@ -204,7 +209,7 @@ fn rollup_combinator(intent: &AggIntent, finer_measure_col: ColumnId) -> Option< /// 4. `finer_output_schema` (the finer aggregate's own *output* schema, not /// the shared child's) carries a provable unique key /// ([`Schema::has_unique_key`]) — **the exact legality gate -/// `pre_asap::cse::share_common_sub_dags` already applies to its own +/// `ir::cse::share_common_sub_dags` already applies to its own /// sharing decisions**, reused verbatim here rather than re-invented: /// `share_common_sub_dags`'s own doc ("Legality: gated by /// `Schema::unique_keys`") states a producer's output is only safely @@ -263,7 +268,7 @@ fn is_strict_column_superset(finer: &[ColumnId], coarser: &[ColumnId]) -> bool { /// docs' "Non-goals" on why finding the full sibling set across a workload /// is a workload-wide traversal this strategy does not own. pub struct RollupStrategy { - siblings: Vec>, + siblings: Vec>, } impl RollupStrategy { @@ -271,7 +276,7 @@ impl RollupStrategy { /// each as a candidate roll-up source (or target) — typically the full set of `Aggregate` /// nodes a workload-wide discovery pass (issue #252) already found /// sharing at least one child `Rc` with something else. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -280,7 +285,7 @@ impl RollupStrategy { /// Every sibling that is a legal, strictly finer roll-up source for /// `target` — shared between `matches` and `replacements` so the two /// can never disagree about which siblings qualify. - fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { + fn finer_sources(&self, target: &TargetSubDAG<'_>) -> Vec<&Rc> { let Some((coarser_by, coarser_intent, coarser_child)) = bindable_grouped_aggregate(target.root) else { @@ -301,12 +306,9 @@ impl RollupStrategy { if !Rc::ptr_eq(finer_child, coarser_child) && finer_child != coarser_child { return false; } - let Ok(finer_schema) = candidate.output_schema() else { - return false; - }; is_legal_rollup_source( finer_by, - &finer_schema, + &candidate.schema, finer_intent, coarser_by, coarser_intent, @@ -325,7 +327,7 @@ impl ReplacementStrategy for RollupStrategy { let Some((coarser_by, coarser_intent, _)) = bindable_grouped_aggregate(target.root) else { return Vec::new(); }; - let QueryExpr::Aggregate { output_names, .. } = target.root.as_ref() else { + let Some(NonASAPOp::Aggregate { output_names, .. }) = target.root.non_asap() else { unreachable!("bindable_grouped_aggregate already confirmed Aggregate"); }; self.finer_sources(target) @@ -335,7 +337,7 @@ impl ReplacementStrategy for RollupStrategy { } } -/// Build the coarser replacement: a new `QueryExpr::Aggregate` grouped by +/// Build the coarser replacement: a new `NonASAPOp::Aggregate` grouped by /// `coarser_by`'s columns (repositioned into `finer`'s own output schema — /// see below), computing `rollup_combinator(intent, ..)` over `finer`'s own /// measure column, with `child = finer` instead of the original shared @@ -350,7 +352,7 @@ impl ReplacementStrategy for RollupStrategy { /// position in the shared child to its position in `finer`'s output: the /// index its `ColumnId` occupies within `finer_by`'s own ordered list. fn build_rollup( - finer: &Rc, + finer: &Rc, coarser_by: &GroupKeys, intent: &AggIntent, output_names: &[String], @@ -367,19 +369,21 @@ fn build_rollup( .map(|id| finer_by.keys().iter().position(|f| f == id)) .collect::>>()?; - let rewritten = QueryExpr::Aggregate { - reduction: Reduction::by(remapped_by), - measures: vec![combinator], - output_names: output_names.to_vec(), - filters: vec![], - having: None, - child: Rc::clone(finer), - }; + let rewritten = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(remapped_by), + measures: vec![combinator], + output_names: output_names.to_vec(), + filters: vec![], + having: None, + child: Rc::clone(finer), + })) + .ok()?; Some(ReplacementSubDAG { strategy: "RollupStrategy", - replacement: Replacement::Rewrite(Rc::new(rewritten)), - provenance: crate::replacement::ReplacementProvenance::LogicalRewrite, + replacement: Replacement::SubDAG(rewritten), + provenance: crate::pass1::replacement::ReplacementProvenance::LogicalRewrite, rationale: format!( "rolls up from the finer Aggregate grouped by {:?} (a strict superset of this \ node's own {:?} grouping over the same shared source) instead of an independent \ @@ -394,13 +398,13 @@ fn build_rollup( #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::{DataType, Field}; + use asap_types::ir::operator::operator_properties::Source; + use asap_types::ir::schema::{DataType, Field}; use asap_types::types::AccuracyTarget; /// `[ts(0), value(1), job(2), region(3)]`. - fn metric_scan() -> QueryExpr { - QueryExpr::Scan { + fn metric_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -413,33 +417,36 @@ mod tests { 0, vec![], ), - } + })) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } fn without_agg( excluded: Vec, intent: AggIntent, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![intent], output_names: vec![], filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } // ── is_legal_rollup_source (the standalone predicate) ─────────────── @@ -569,7 +576,7 @@ mod tests { #[test] fn superset_by_over_identical_mergeable_intent_and_shared_child_rolls_up() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); @@ -581,16 +588,16 @@ mod tests { let replacements = strategy.replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDAG(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { + let Some(NonASAPOp::Aggregate { reduction, measures, child, having, .. - } = rewritten.as_ref() + }) = rewritten.non_asap() else { panic!("expected an Aggregate rewrite, got {rewritten:?}"); }; @@ -620,7 +627,7 @@ mod tests { // Count is not self-combining (see the module docs) — the rewritten // measure must be Sum over the finer Count's own output column, not // Count reapplied. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg( vec![2, 3], AggIntent::Count { @@ -642,10 +649,10 @@ mod tests { let replacements = strategy.replacements(&target); assert_eq!(replacements.len(), 1, "{replacements:?}"); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDAG(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - let QueryExpr::Aggregate { measures, .. } = rewritten.as_ref() else { + let Some(NonASAPOp::Aggregate { measures, .. }) = rewritten.non_asap() else { panic!("expected an Aggregate rewrite"); }; assert_eq!( @@ -657,7 +664,7 @@ mod tests { #[test] fn approximate_count_does_not_roll_up_via_sum() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let intent = AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), }; @@ -673,21 +680,21 @@ mod tests { #[test] fn default_workload_search_adds_rollup_for_two_query_workload() { - let fine_scan = Rc::new(metric_scan()); - let coarse_scan = Rc::new(metric_scan()); + let fine_scan = metric_scan(); + let coarse_scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &fine_scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &coarse_scan); - let space = crate::replacement::search_workload(vec![("fine", fine), ("coarse", coarse)]); + let space = + crate::pass1::replacement::search_workload(vec![("fine", fine), ("coarse", coarse)]); let coarse_group = space .target_subdag_candidates() .find(|group| { - matches!( - group.target.as_ref(), - QueryExpr::Aggregate { + matches!(group.target.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2] + }) if by.keys() == [2] ) }) .expect("coarser aggregate group"); @@ -696,19 +703,19 @@ mod tests { .candidates .iter() .find_map(|candidate| match &candidate.replacement { - Replacement::Rewrite(rewrite) => Some(rewrite), - Replacement::Summary(_) | Replacement::ExactComposition(_) => None, + // Old `Replacement::Rewrite`: a pure pre-ASAP sub-DAG. + Replacement::SubDAG(rewrite) if !rewrite.contains_asap() => Some(rewrite), + Replacement::SubDAG(_) | Replacement::ExactComposition(_) => None, }) .expect("default search must include the roll-up rewrite"); - let QueryExpr::Aggregate { child, .. } = rewrite.as_ref() else { + let Some(NonASAPOp::Aggregate { child, .. }) = rewrite.non_asap() else { panic!("expected aggregate rewrite, got {rewrite:?}"); }; - assert!(matches!( - child.as_ref(), - QueryExpr::Aggregate { + assert!(matches!(child.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2, 3] + }) if by.keys() == [2, 3] )); } @@ -717,18 +724,18 @@ mod tests { let intent = AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), }; - let fine = agg(vec![2, 3], intent.clone(), &Rc::new(metric_scan())); - let coarse = agg(vec![2], intent, &Rc::new(metric_scan())); - let space = crate::replacement::search_workload(vec![("fine", fine), ("coarse", coarse)]); + let fine = agg(vec![2, 3], intent.clone(), &metric_scan()); + let coarse = agg(vec![2], intent, &metric_scan()); + let space = + crate::pass1::replacement::search_workload(vec![("fine", fine), ("coarse", coarse)]); let coarse_group = space .target_subdag_candidates() .find(|group| { - matches!( - group.target.as_ref(), - QueryExpr::Aggregate { + matches!(group.target.non_asap(), + Some(NonASAPOp::Aggregate { reduction: Reduction::Reduce(by), .. - } if by.keys() == [2] + }) if by.keys() == [2] ) }) .expect("coarser aggregate group"); @@ -736,32 +743,35 @@ mod tests { assert!(coarse_group .candidates .iter() - .all(|candidate| !matches!(candidate.replacement, Replacement::Rewrite(_)))); + .all(|candidate| !matches!(&candidate.replacement, + Replacement::SubDAG(rewrite) if !rewrite.contains_asap()))); } #[test] fn rollup_preserves_the_coarser_output_name() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let coarse = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![AggIntent::Sum { col: Some(1) }], - output_names: vec!["total_requests".into()], - filters: vec![], - having: None, - child: Rc::clone(&scan), - }); - let original_schema = coarse.output_schema().unwrap(); + let coarse = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![AggIntent::Sum { col: Some(1) }], + output_names: vec!["total_requests".into()], + filters: vec![], + having: None, + child: Rc::clone(&scan), + })) + .unwrap(); + let original_schema = coarse.schema.clone(); let siblings = vec![Rc::clone(&fine), Rc::clone(&coarse)]; let strategy = RollupStrategy::new(&siblings); let replacements = strategy.replacements(&TargetSubDAG::new(&coarse)); - let Replacement::Rewrite(rewritten) = &replacements[0].replacement else { + let Replacement::SubDAG(rewritten) = &replacements[0].replacement else { panic!("expected a Rewrite replacement"); }; - assert_eq!(rewritten.output_schema().unwrap(), original_schema); - let QueryExpr::Aggregate { output_names, .. } = rewritten.as_ref() else { + assert_eq!(rewritten.schema.clone(), original_schema); + let Some(NonASAPOp::Aggregate { output_names, .. }) = rewritten.non_asap() else { unreachable!(); }; assert_eq!(output_names, &vec!["total_requests".to_string()]); @@ -769,7 +779,7 @@ mod tests { #[test] fn non_mergeable_intent_does_not_roll_up() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Avg { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Avg { col: Some(1) }, &scan); @@ -789,12 +799,12 @@ mod tests { // numerically a superset of the coarser side's *kept* positions — // `is_legal_rollup_source` rejects any `without` grouping outright, // and would reject on the missing unique key regardless. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = without_agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); assert!( - !fine.output_schema().unwrap().has_unique_key(), + !fine.schema.clone().has_unique_key(), "fixture sanity: a without(...) aggregate has no provable unique key" ); @@ -809,7 +819,7 @@ mod tests { #[test] fn unrelated_by_sets_do_not_roll_up() { // Neither `[job]` nor `[region]` is a superset of the other. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let a = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); let b = agg(vec![3], AggIntent::Sum { col: Some(1) }, &scan); @@ -830,7 +840,7 @@ mod tests { // Equal groupings are `SharedSubDAGStrategy`'s CSE-sharing // question (build once and share, or build independently) — a // roll-up requires a *strict* superset, not equality. - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let a = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); let b = agg(vec![2], AggIntent::Sum { col: Some(1) }, &scan); @@ -845,16 +855,8 @@ mod tests { // Scans without unique keys are deliberately not pointer-aliased by // CSE. Structural equality still proves identical schemas and makes // the two aggregates' positional ColumnIds comparable. - let fine = agg( - vec![2, 3], - AggIntent::Sum { col: Some(1) }, - &Rc::new(metric_scan()), - ); - let coarse = agg( - vec![2], - AggIntent::Sum { col: Some(1) }, - &Rc::new(metric_scan()), - ); + let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &metric_scan()); + let coarse = agg(vec![2], AggIntent::Sum { col: Some(1) }, &metric_scan()); let siblings = vec![Rc::clone(&fine), Rc::clone(&coarse)]; let strategy = RollupStrategy::new(&siblings); @@ -865,21 +867,23 @@ mod tests { #[test] fn does_not_match_a_multi_measure_or_having_aggregate() { - let scan = Rc::new(metric_scan()); + let scan = metric_scan(); let fine = agg(vec![2, 3], AggIntent::Sum { col: Some(1) }, &scan); - let multi = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![ - AggIntent::Sum { col: Some(1) }, - AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }, - ], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::clone(&scan), - }); + let multi = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![2]), + measures: vec![ + AggIntent::Sum { col: Some(1) }, + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }, + ], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::clone(&scan), + })) + .unwrap(); let siblings = vec![Rc::clone(&fine), Rc::clone(&multi)]; let strategy = RollupStrategy::new(&siblings); diff --git a/crates/logical-optimizer/src/pass2/identical_expressions.rs b/crates/logical-optimizer/src/pass2/identical_expressions.rs new file mode 100644 index 000000000..2baefe7b9 --- /dev/null +++ b/crates/logical-optimizer/src/pass2/identical_expressions.rs @@ -0,0 +1,142 @@ +//! Pass 2's identical-expression rule (#509): structurally identical sub-DAGs +//! across queries may be computed once. Sharing changes cost +//! non-additively (a shared node is priced once), so Stage 1 keeps both the +//! independent and the shared form, and Stage 3 chooses by cost. +//! +//! Each form is a Pass 1 inventory: Pass 1 over the queries as written, and +//! Pass 1 over the queries with identical sub-DAGs merged by +//! [`share_common_sub_dags`]. A target that sharing merges is one target in +//! the shared form, so its queries take the same alternative. + +use std::collections::HashSet; +use std::rc::Rc; + +use asap_types::ir::cse::share_common_sub_dags; +use asap_types::ir::{OperatorNode, QueryRoot}; + +use crate::pass1::logical_candidates::{ + enumerate_local_logical_candidates, LocalLogicalCandidates, LogicalCandidateError, +}; + +/// One input-sharing form of the workload, with its Pass 1 alternatives. +#[derive(Debug, Clone)] +pub struct SharingVariant { + /// Whether identical sub-DAGs across queries are merged. + pub shared: bool, + pub inventory: LocalLogicalCandidates, +} + +/// Stage 1 = Pass 1 + Pass 2's identical-expression rule: the independent +/// variant first, then the shared one when sharing merges at least one node. +pub fn stage1_logical_candidates( + roots: Vec<(Id, QueryRoot)>, +) -> Result>, LogicalCandidateError> { + let shared = share_identical_expressions(&roots); + let mut variants = vec![SharingVariant { + shared: false, + inventory: enumerate_local_logical_candidates(roots)?, + }]; + if let Some(roots) = shared { + variants.push(SharingVariant { + shared: true, + inventory: enumerate_local_logical_candidates(roots)?, + }); + } + Ok(variants) +} + +/// `roots` with identical operator sub-DAGs merged across queries, or `None` +/// when that merges nothing. Scalar roots are kept as written. +pub fn share_identical_expressions( + roots: &[(Id, QueryRoot)], +) -> Option> { + let operators: Vec<(usize, Rc)> = roots + .iter() + .enumerate() + .filter_map(|(i, (_, root))| match root { + QueryRoot::Operator(node) => Some((i, node.clone())), + QueryRoot::Scalar(_) => None, + }) + .collect(); + let before = distinct_nodes(operators.iter().map(|(_, node)| node)); + let merged = share_common_sub_dags(operators); + if distinct_nodes(merged.iter().map(|(_, node)| node)) == before { + return None; + } + let mut out = roots.to_vec(); + for (i, node) in merged { + out[i].1 = QueryRoot::Operator(node); + } + Some(out) +} + +/// Distinct nodes (by identity) reachable from `roots`. +pub fn distinct_nodes<'a>(roots: impl IntoIterator>) -> usize { + roots + .into_iter() + .flat_map(OperatorNode::reachable) + .map(|node| Rc::as_ptr(&node)) + .collect::>() + .len() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::test_support::lower_promql; + use asap_types::types::AccuracyTarget; + + fn roots(queries: &[&str]) -> Vec<(usize, QueryRoot)> { + queries + .iter() + .enumerate() + .map(|(i, q)| { + let root = lower_promql(q, AccuracyTarget::Exact); + let root = + asap_types::ir::schema_support::with_promql_series_identity(&root).unwrap(); + (i, QueryRoot::Operator(root)) + }) + .collect() + } + + /// Two queries over the same range selector get an independent and a + /// shared variant, with the same targets in each. + #[test] + fn identical_input_adds_a_shared_variant() { + let variants = stage1_logical_candidates(roots(&[ + "sum by (job) (rate(m[1m]))", + "topk by (job) (10, sum_over_time(m[1m]))", + ])) + .unwrap(); + assert_eq!( + variants.iter().map(|v| v.shared).collect::>(), + [false, true] + ); + assert_eq!( + variants[0].inventory.targets.len(), + variants[1].inventory.targets.len() + ); + let operators = |v: &SharingVariant| -> Vec> { + v.inventory + .roots + .iter() + .map(|(_, r)| match r { + QueryRoot::Operator(n) => n.clone(), + QueryRoot::Scalar(_) => unreachable!(), + }) + .collect() + }; + assert!( + distinct_nodes(&operators(&variants[1])) < distinct_nodes(&operators(&variants[0])) + ); + } + + /// Queries with nothing in common have no shared variant. + #[test] + fn nothing_identical_adds_no_variant() { + let variants = + stage1_logical_candidates(roots(&["sum(rate(a[1m]))", "sum(rate(b[1m]))"])).unwrap(); + assert_eq!(variants.len(), 1); + assert!(!variants[0].shared); + } +} diff --git a/crates/logical-optimizer/src/pass2/mod.rs b/crates/logical-optimizer/src/pass2/mod.rs new file mode 100644 index 000000000..520b774c0 --- /dev/null +++ b/crates/logical-optimizer/src/pass2/mod.rs @@ -0,0 +1,7 @@ +//! Pass 2: ASAP-aware sharing across targets. A shared summary must meet the +//! strictest accuracy requirement of its readers. The stage pipeline applies +//! the identical-expression rule ([`identical_expressions`]). + +pub mod identical_expressions; +pub mod reconciliation; +pub mod topk_reuse; diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/logical-optimizer/src/pass2/reconciliation.rs similarity index 66% rename from crates/asap-aware-mapping/src/accuracy/reconciliation.rs rename to crates/logical-optimizer/src/pass2/reconciliation.rs index 6014d3ba3..051bb6425 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/logical-optimizer/src/pass2/reconciliation.rs @@ -3,14 +3,14 @@ //! //! ## The gap this closes //! -//! `asap_types::pre_asap::cse::share_common_sub_dags` (pre-ASAP CSE) only +//! `asap_types::ir::cse::share_common_sub_dags` (pre-ASAP CSE) only //! ever merges two sub-DAGs that are *exactly* [`PartialEq`]-equal, //! including their [`AggIntent`]'s `accuracy: AccuracyTarget` field. Two //! otherwise-identical aggregates that differ *only* in how tight an //! accuracy bound they ask for — `quantile(0.99, x)` at `epsilon=0.01` for //! one consumer, the same `quantile(0.99, x)` at `epsilon=0.05` for //! another — are therefore never the same `Rc`, never collapse into one -//! [`crate::replacement::TargetSubDAGCandidates`], and [`crate::replacement::SharedSubDAGStrategy`] +//! [`crate::pass1::replacement::TargetSubDAGCandidates`], and [`crate::pass1::replacement::SharedSubDAGStrategy`] //! never even gets a `TargetSubDAG` with `consumer_count >= 2` to propose //! sharing for. This crate would build two entirely independent sketches //! for what is conceptually one computation, even though a single sketch @@ -22,26 +22,26 @@ //! consumer that pattern-matches on a specific `AccuracyTarget` — depends on //! that). What this module adds is a *second*, narrower notion of "close //! enough to share" that sits entirely inside the [`ReplacementStrategy`] -//! extension point: one more candidate a [`crate::cost_model::CostModel`] +//! extension point: one more candidate a `cost_model::CostModel` //! may or may not prefer, never a forced rewrite and never a change to what //! `share_common_sub_dags` itself merges. //! //! ## What counts as a "near-duplicate", and why //! -//! Two [`QueryExpr::Aggregate`] nodes are accuracy-near-duplicates here iff, +//! Two `NonASAPOp::Aggregate` nodes are accuracy-near-duplicates here iff, //! **in this order**: //! -//! 1. Both are the same bindable shape [`crate::replacement::SketchAlgorithmStrategy`] +//! 1. Both are the same bindable shape [`crate::pass1::replacement::ASAPStrategies`] //! itself targets — a single measure, no `HAVING` (`bindable_intent`'s own //! scope) — **and** that one measure is one of the four accuracy-bearing -//! [`AggIntent`] variants ([`crate::replacement::accuracy_target`]'s own +//! [`AggIntent`] variants ([`crate::pass1::replacement::accuracy_target`]'s own //! scope: `Count` / `Quantile` / `Cardinality` / `TopK`). Every other //! intent has no `AccuracyTarget` to reconcile in the first place. //! 2. Same `reduction` (grouping), same `output_names`, and the same shared //! `child` (`Rc::ptr_eq`, or value-equal for two independently-built but //! identical sub-DAGs CSE conservatively declined to alias) — the same -//! "identical everything else" bar [`crate::rollup::RollupStrategy`] and -//! [`crate::topk_reuse::TopKLimitReuseStrategy`] already hold their own +//! "identical everything else" bar [`crate::pass1::rollup::RollupStrategy`] and +//! [`crate::pass2::topk_reuse::TopKLimitReuseStrategy`] already hold their own //! sibling-reuse candidates to. //! 3. The one measure is identical **except** for `accuracy` — same variant, //! same `col`/`q`/`k` (see [`same_intent_except_accuracy`]). @@ -51,7 +51,7 @@ //! 5. The tighter candidate's own **output** schema carries a provable //! unique key (`Schema::has_unique_key`) — the exact legality gate //! `share_common_sub_dags` itself applies (see `cse.rs`'s "Legality" -//! section) and [`crate::rollup::RollupStrategy::is_legal_rollup_source`] +//! section) and [`crate::pass1::rollup::RollupStrategy::is_legal_rollup_source`] //! already reuses verbatim for the identical reason: a producer's output //! is only safely reusable across a second, independent consumer when //! its row identity is provably stable across reads. A global or @@ -61,15 +61,13 @@ //! //! ## Safety of tightening: why reading the tighter build is always sound //! -//! [`crate::replacement::accuracy_budget`] resolves *every* `AccuracyTarget` +//! [`crate::pass1::replacement::accuracy_budget`] resolves *every* `AccuracyTarget` //! (`Epsilon`/`EpsilonDelta`) to the literal `(eps, delta)` pair -//! `realizations_for_intent`'s `sketch_realizations` feeds into -//! `CostModel::size_params` — the same numbers `default_size_params`' +//! `realizations_for_intent`'s `sketch_realizations` feeds into the +//! analytical sizing — the same numbers `default_size_params`' //! `kll_k` / `cms_width` / `cms_depth` / `hll_precision` / `kmv_k` / DDSketch's //! own `alpha == eps` invert. Every shipped formula is monotonic in its -//! input, and custom [`crate::cost_model::CostModel::size_params`] -//! implementations are contractually required to return parameters that -//! satisfy their supplied budget. So a sketch satisfying budget `(e1, d1)` +//! input. So a sketch satisfying budget `(e1, d1)` //! also satisfies any //! requirement `(e2, d2)` with `e1 <= e2 && d1 <= d2` — [`dominates`]'s exact //! check — regardless of which of `Epsilon`/`EpsilonDelta` either side is @@ -105,20 +103,20 @@ //! //! Like every [`ReplacementStrategy`], this only ever *proposes* — the //! looser-accuracy consumer's own independently-sized candidate (from -//! [`crate::replacement::SketchAlgorithmStrategy`]) stays in its -//! [`crate::replacement::TargetSubDAGCandidates`] right alongside this strategy's +//! [`crate::pass1::replacement::ASAPStrategies`]) stays in its +//! [`crate::pass1::replacement::TargetSubDAGCandidates`] right alongside this strategy's //! "read the tighter sibling instead" [`Replacement::Rewrite`] candidate; -//! [`crate::cost_model::CostModel`]-driven ranking picks between them; +//! `cost_model::CostModel`-driven ranking picks between them; //! nothing here removes or filters the independent candidate. //! //! ## Costing this candidate shape: a dedicated arm, not a reused one //! //! This strategy's candidates carry their own -//! [`crate::replacement::ReplacementProvenance::AccuracyReconciliation`] +//! [`crate::pass1::replacement::ReplacementProvenance::AccuracyReconciliation`] //! rather than reusing `LogicalRewrite` -//! ([`crate::rollup::RollupStrategy`]/[`crate::topk_reuse::TopKLimitReuseStrategy`]'s +//! ([`crate::pass1::rollup::RollupStrategy`]/[`crate::pass2::topk_reuse::TopKLimitReuseStrategy`]'s //! tag), because it needs its own cost treatment in -//! [`crate::cost_model::DefaultCostModel::estimate_cost`], not just its own +//! `cost_model::DefaultCostModel::estimate_cost`, not just its own //! label. Every other `Replacement::Rewrite` shape that reaches //! `estimate_cost` (`SharedSubDAGStrategy`'s `CseRecompute`, `Rollup`'s and //! `TopKLimitReuse`'s `LogicalRewrite`) really does rebuild `target` from a @@ -135,13 +133,13 @@ //! the literal inversion this module's tests //! (`estimate_cost_does_not_scale_with_the_readers_own_consumer_count`) //! pin against. `estimate_cost` instead prices this shape as a -//! [`crate::cost_model::CostModel::cse_shared_maintenance_cost`] read +//! `cost_model::CostModel::cse_shared_maintenance_cost` read //! against `rc`'s **own** bound summary — the same order-of-magnitude, //! per-family cost `SharedSubDAGStrategy`'s own `CseShare` candidate is //! priced with, reflecting "one more reference into a structure that's //! already being maintained" rather than "build a whole new one." //! -//! `CandidateLogicalASAPDAGs::global_selection` treats this rewrite as a cross-group edge: +//! `candidate_selection::global_selection` treats this rewrite as a cross-group edge: //! selecting it increments `rc`'s own `effective_consumer_count`, then lets //! that sibling group propagate the uses through its selected implementation. //! Accuracy edges are directed strictly from looser to tighter budgets, so @@ -149,14 +147,16 @@ //! same structural child, so adding the edge preserves the reference DAG's //! parent-before-child topological ordering. +use asap_types::ir::operator::non_asap::any_measure_filtered; use std::cmp::Ordering; use std::rc::Rc; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::query_expr::{any_measure_filtered, QueryExpr, Reduction}; +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::Reduction; +use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::types::AccuracyTarget; -use crate::replacement::{ +use crate::pass1::replacement::{ accuracy_budget, accuracy_target, Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; @@ -169,27 +169,27 @@ type BindableAccuracyAggregate<'a> = ( &'a AggIntent, &'a AccuracyTarget, &'a [String], - &'a Rc, + &'a Rc, ); /// The `(reduction, intent, accuracy, output_names, child)` shape this /// module operates on: the same single-measure, no-`HAVING` bindable shape -/// [`crate::replacement::SketchAlgorithmStrategy`] targets (see that +/// [`crate::pass1::replacement::ASAPStrategies`] targets (see that /// module's private `bindable_intent`), further narrowed to a measure whose /// intent actually carries an [`AccuracyTarget`] -/// ([`crate::replacement::accuracy_target`]'s own scope: `Count` / +/// ([`crate::pass1::replacement::accuracy_target`]'s own scope: `Count` / /// `Quantile` / `Cardinality` / `TopK`). `None` for anything else, including /// a multi-measure or `HAVING` aggregate, a non-`Aggregate` node, or an /// accuracy-free intent (`Sum`, `Avg`, …). -fn bindable_accuracy_aggregate(node: &QueryExpr) -> Option> { - let QueryExpr::Aggregate { +fn bindable_accuracy_aggregate(node: &OperatorNode) -> Option> { + let Some(NonASAPOp::Aggregate { reduction, measures, output_names, filters, having, child, - } = node + }) = node.non_asap() else { return None; }; @@ -207,7 +207,7 @@ fn bindable_accuracy_aggregate(node: &QueryExpr) -> Option bool { match (a, b) { @@ -268,17 +268,17 @@ fn strictly_tighter(a: &AccuracyTarget, b: &AccuracyTarget) -> bool { /// tighter accuracy, and proposes reading that sibling's own (to-be-built) /// result instead of building an independent, looser copy — "build once at /// the tightest of the group's accuracy requirements, all consumers read -/// from it," ranked by [`crate::cost_model::CostModel`] like any other +/// from it," ranked by `cost_model::CostModel` like any other /// candidate, never forced. See the module docs for the full design. /// /// `siblings` is **caller-supplied, not discovered here** — the identical /// "workload-wide discovery isn't this strategy's job" split -/// [`crate::rollup::RollupStrategy`] and [`crate::topk_reuse::TopKLimitReuseStrategy`] -/// already draw; [`crate::replacement::search_workload_with`] constructs +/// [`crate::pass1::rollup::RollupStrategy`] and [`crate::pass2::topk_reuse::TopKLimitReuseStrategy`] +/// already draw; [`crate::pass1::replacement::search_workload_with`] constructs /// this strategy from the same post-CSE `Aggregate` sibling set it already /// builds for `RollupStrategy`. pub struct AccuracyReconciliationStrategy { - siblings: Vec>, + siblings: Vec>, } impl AccuracyReconciliationStrategy { @@ -286,7 +286,7 @@ impl AccuracyReconciliationStrategy { /// each as a candidate tighter-accuracy source (or looser-accuracy /// target) — typically the full set of `Aggregate` nodes a workload-wide /// discovery pass already found. - pub fn new(siblings: &[Rc]) -> Self { + pub fn new(siblings: &[Rc]) -> Self { Self { siblings: siblings.to_vec(), } @@ -301,8 +301,8 @@ impl AccuracyReconciliationStrategy { /// /// Also requires the candidate's own *output* schema to carry a provable /// unique key ([`Schema::has_unique_key`]) — the exact legality gate - /// `pre_asap::cse::share_common_sub_dags` already applies to its own - /// sharing decisions, and [`crate::rollup::RollupStrategy`] already + /// `ir::cse::share_common_sub_dags` already applies to its own + /// sharing decisions, and [`crate::pass1::rollup::RollupStrategy`] already /// reuses verbatim for the identical reason (see that module's /// `is_legal_rollup_source` doc, point 4): a producer's output is only /// safely reusable across a second, independent consumer when its row @@ -311,14 +311,14 @@ impl AccuracyReconciliationStrategy { /// reports no unique key — see `cse.rs`'s "Legality" section) would get /// proposed for reconciliation even though nothing guarantees a second /// read of it lines up row-for-row with the first. - fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + fn tighter_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { let Some((target_reduction, target_intent, target_accuracy, target_names, target_child)) = bindable_accuracy_aggregate(target.root) else { return Vec::new(); }; - let mut sources: Vec<&Rc> = self + let mut sources: Vec<&Rc> = self .siblings .iter() .filter(|candidate| { @@ -335,9 +335,7 @@ impl AccuracyReconciliationStrategy { && (Rc::ptr_eq(child, target_child) || child == target_child) && same_intent_except_accuracy(intent, target_intent) && strictly_tighter(accuracy, target_accuracy) - && candidate - .output_schema() - .is_ok_and(|schema| schema.has_unique_key()) + && candidate.schema.has_unique_key() }) .collect(); sources.sort_by(|a, b| { @@ -369,7 +367,7 @@ impl ReplacementStrategy for AccuracyReconciliationStrategy { .expect("tighter_sources only returns bindable_accuracy_aggregate matches"); ReplacementSubDAG { strategy: self.name(), - replacement: Replacement::Rewrite(Rc::clone(source)), + replacement: Replacement::SubDAG(Rc::clone(source)), provenance: ReplacementProvenance::AccuracyReconciliation, rationale: format!( "reuses a near-duplicate sibling aggregate — identical intent and grouping \ @@ -390,19 +388,17 @@ impl ReplacementStrategy for AccuracyReconciliationStrategy { #[cfg(test)] mod tests { use super::*; - use crate::cost_model::{CostModel, DefaultCostModel}; - use asap_types::post_asap::SketchAlgorithm; - use asap_types::pre_asap::cse::share_common_sub_dags; - use asap_types::pre_asap::query_expr::{GroupKeys, Source}; - use asap_types::pre_asap::schema::{ColumnId, DataType, Field, Schema}; + use asap_types::ir::cse::share_common_sub_dags; + use asap_types::ir::operator::operator_properties::{GroupKeys, Source}; + use asap_types::ir::schema::{ColumnId, DataType, Field, Schema}; /// `[ts(0), value(1), job(2)]`. /// A unique-keyed scan (`[ts]`) so `share_common_sub_dags` is actually /// willing to hoist it — see `Schema::has_unique_key`/`cse.rs`'s own /// "Legality" section: a producer with no provable unique key is always /// inserted fresh, never hoisted, regardless of structural equality. - fn metric_scan() -> Rc { - Rc::new(QueryExpr::Scan { + fn metric_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -414,21 +410,23 @@ mod tests { 0, vec![vec![0]], ), - }) + })) + .unwrap() } - fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { - Rc::new(QueryExpr::Aggregate { + fn agg(by: Vec, intent: AggIntent, child: &Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(by), measures: vec![intent], output_names: vec![], filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } - fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { agg( vec![2], AggIntent::Quantile { @@ -441,10 +439,14 @@ mod tests { } /// A globally-grouped (`by(vec![])`) quantile — `aggregate_output_schema` - /// reports no unique key for an empty `by` (see `query_expr.rs`'s own + /// reports no unique key for an empty `by` (see `aggregate_schema.rs`'s own /// `unique_keys = if by.is_empty() || has_count_values { vec![] } else /// { .. }`). - fn global_quantile(q: f64, accuracy: AccuracyTarget, child: &Rc) -> Rc { + fn ungrouped_quantile( + q: f64, + accuracy: AccuracyTarget, + child: &Rc, + ) -> Rc { agg( vec![], AggIntent::Quantile { @@ -463,9 +465,9 @@ mod tests { q: f64, accuracy: AccuracyTarget, excluded: Vec, - child: &Rc, - ) -> Rc { - Rc::new(QueryExpr::Aggregate { + child: &Rc, + ) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::Reduce(GroupKeys::without(excluded)), measures: vec![AggIntent::Quantile { col: None, @@ -476,7 +478,8 @@ mod tests { filters: vec![], having: None, child: Rc::clone(child), - }) + })) + .unwrap() } // ── dominates / strictly_tighter ───────────────────────────────────── @@ -527,7 +530,7 @@ mod tests { let epsilon_only = AccuracyTarget::Epsilon(0.01); let equivalent_epsilon_delta = AccuracyTarget::EpsilonDelta { epsilon: 0.01, - delta: crate::replacement::DEFAULT_DELTA, + delta: crate::pass1::replacement::DEFAULT_DELTA, }; assert!(dominates(&epsilon_only, &equivalent_epsilon_delta)); assert!(dominates(&equivalent_epsilon_delta, &epsilon_only)); @@ -548,7 +551,7 @@ mod tests { assert!(strategy.matches(&TargetSubDAG::new(&loose))); let replacements = strategy.replacements(&TargetSubDAG::new(&loose)); assert_eq!(replacements.len(), 1); - let Replacement::Rewrite(rc) = &replacements[0].replacement else { + let Replacement::SubDAG(rc) = &replacements[0].replacement else { panic!("expected a Rewrite candidate"); }; assert!(Rc::ptr_eq(rc, &tight)); @@ -571,7 +574,8 @@ mod tests { let tight = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); let loose = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); - let space = crate::replacement::search_workload(vec![("tight", tight), ("loose", loose)]); + let space = + crate::pass1::replacement::search_workload(vec![("tight", tight), ("loose", loose)]); let loose_root = &space.roots[1].1; let loose_group = space @@ -580,7 +584,7 @@ mod tests { assert!( loose_group.candidates.iter().any(|candidate| { candidate.strategy == "AccuracyReconciliationStrategy" - && matches!(candidate.replacement, Replacement::Rewrite(_)) + && matches!(candidate.replacement, Replacement::SubDAG(_)) }), "expected an AccuracyReconciliationStrategy candidate for the looser consumer, got: \ {:?}", @@ -661,7 +665,7 @@ mod tests { // ── exact structural equality / share_common_sub_dags is unchanged ──── #[test] - fn share_common_sub_dags_still_never_merges_differing_accuracy() { + fn share_common_subdags_still_never_merges_differing_accuracy() { // The additive guarantee this issue explicitly must not violate: // pre-ASAP CSE's own exact-equality merge stays exact. Two // aggregates differing only in `accuracy` must come back as two @@ -669,8 +673,8 @@ mod tests { // is the *only* place cross-accuracy sharing gets proposed, never // `share_common_sub_dags` itself. let scan = metric_scan(); - let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); + let a = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let b = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!( @@ -684,10 +688,10 @@ mod tests { // The identical scan child, though, is still shared exactly as // before — this module changes nothing about that. - let QueryExpr::Aggregate { child: child_a, .. } = roots[0].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child: child_a, .. }) = roots[0].1.non_asap() else { panic!("expected an Aggregate root"); }; - let QueryExpr::Aggregate { child: child_b, .. } = roots[1].1.as_ref() else { + let Some(NonASAPOp::Aggregate { child: child_b, .. }) = roots[1].1.non_asap() else { panic!("expected an Aggregate root"); }; assert!(Rc::ptr_eq(child_a, child_b)); @@ -700,8 +704,8 @@ mod tests { // exact equality — unrelated to this module, but pins the contrast // with the test above. let scan = metric_scan(); - let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); + let a = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let b = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); @@ -717,11 +721,11 @@ mod tests { // `share_common_sub_dags`/`RollupStrategy` apply, which this // strategy must not bypass (module docs, point 5). let scan = metric_scan(); - let tight = global_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); - let loose = global_quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); + let tight = ungrouped_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let loose = ungrouped_quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); assert!( - !tight.output_schema().unwrap().has_unique_key(), + !tight.schema.clone().has_unique_key(), "fixture sanity: a globally-grouped aggregate has no provable unique key" ); @@ -741,7 +745,7 @@ mod tests { let loose = without_quantile(0.99, AccuracyTarget::Epsilon(0.05), vec![2], &scan); assert!( - !tight.output_schema().unwrap().has_unique_key(), + !tight.schema.clone().has_unique_key(), "fixture sanity: a without(...) aggregate has no provable unique key" ); @@ -749,229 +753,4 @@ mod tests { assert!(!strategy.matches(&TargetSubDAG::new(&loose))); assert!(strategy.replacements(&TargetSubDAG::new(&loose)).is_empty()); } - - // ── cost: reading the sibling must not be priced like recomputing - // `target` independently per consumer ──────────────────────────────── - - #[test] - fn estimate_cost_does_not_scale_with_the_readers_own_consumer_count() { - // Regression guard for the review-reported sign inversion: pricing - // this candidate like `CseRecompute` ("rebuild `target`, once per - // consumer") made it artificially *more* expensive exactly as more - // of `target`'s own consumers stood to benefit from reading the - // already-necessary tighter sibling instead — the literal opposite - // of the intended incentive. The real cost is "one more read against - // `rc`'s own build," which must not scale with `target`'s own - // `consumer_count`. - let scan = metric_scan(); - let tight = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); - let loose = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); - let strategy = AccuracyReconciliationStrategy::new(&[Rc::clone(&tight), Rc::clone(&loose)]); - let candidate = strategy - .replacements(&TargetSubDAG::new(&loose)) - .into_iter() - .next() - .expect("loose has a reconciliation candidate reading the tight sibling"); - - let cost_model = DefaultCostModel; - let single_consumer = TargetSubDAG::with_consumer_count(&loose, 1); - let many_consumers = TargetSubDAG::with_consumer_count(&loose, 5); - - let cost_single = cost_model.estimate_cost(&candidate, &single_consumer); - let cost_many = cost_model.estimate_cost(&candidate, &many_consumers); - - assert!( - cost_single.is_finite(), - "expected a real cost, not the NaN placeholder: {cost_single}" - ); - assert_eq!( - cost_single, cost_many, - "AccuracyReconciliation's estimate_cost must price 'read the sibling', not scale \ - with the reader's own consumer_count the way CseRecompute's 'rebuild independently \ - per consumer' formula does (single-consumer: {cost_single}, 5 consumers: \ - {cost_many})" - ); - } - - // ── cost_sorted / global_selection: single-consumer and shared-consumer ─ - - #[test] - fn cost_sorted_and_global_selection_handle_a_single_consumer_looser_target() { - let scan = metric_scan(); - let tight = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); - let loose = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); - let space = crate::replacement::search_workload(vec![("tight", tight), ("loose", loose)]); - let loose_root = &space.roots[1].1; - - let cost_model = DefaultCostModel; - let ranked = space.cost_sorted(&cost_model); - let loose_ranked = ranked - .iter() - .find(|group| Rc::ptr_eq(group.target, loose_root)) - .expect("loose has its own ranked group"); - assert!( - loose_ranked.costs.iter().all(|cost| cost.is_finite()), - "no candidate should cost NaN under DefaultCostModel: {:?}", - loose_ranked.costs - ); - assert!( - loose_ranked - .candidates - .iter() - .any(|c| c.strategy == "AccuracyReconciliationStrategy"), - "the reconciliation candidate must still be present, ranked, not filtered" - ); - - let selected = space.global_selection(&cost_model); - let chosen = selected - .for_target(loose_root) - .and_then(|group| group.chosen); - assert!( - chosen.is_some(), - "global_selection must commit to some candidate for a single-consumer looser target" - ); - // With no recompute term at all (it never rebuilds `target`), this - // candidate strictly undercuts every SketchAlgorithmStrategy - // candidate (which each pay a recompute term on top of their own - // maintenance term) under DefaultCostModel's numbers — the sane - // direction: reading an already-necessary sibling should be able to - // win on its own merit, not just fail to lose as badly as before. - assert_eq!( - chosen.map(|c| c.provenance), - Some(crate::replacement::ReplacementProvenance::AccuracyReconciliation) - ); - } - - #[test] - fn cost_sorted_and_global_selection_handle_a_shared_looser_target() { - // The loose accuracy target itself has 2 direct consumers (two - // independently-built but structurally identical loose queries - // merge onto one Rc via ordinary CSE), *and* a separate, - // single-consumer tight sibling exists over the same input — the - // scenario the issue itself targets: `SharedSubDAGStrategy`'s own - // CseShare/CseRecompute pair is on the table for the loose target's - // own 2 consumers at the same time as this strategy's "read the - // tight sibling instead" candidate. - let scan = metric_scan(); - let loose_a = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); - let loose_b = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); - let tight = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - - let space = crate::replacement::search_workload(vec![ - ("loose_a", Rc::new(loose_a)), - ("loose_b", Rc::new(loose_b)), - ("tight", Rc::new(tight)), - ]); - - // Fixture sanity: the two loose roots really did merge onto one Rc. - assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); - let loose_group = space - .candidates_for_target(&space.roots[0].1) - .expect("the merged loose target has a group"); - assert_eq!(loose_group.consumer_count, 2); - assert!( - loose_group - .candidates - .iter() - .any(|c| c.strategy == "AccuracyReconciliationStrategy"), - "the reconciliation candidate must still be proposed alongside the CSE share/recompute \ - pair, not crowded out: {:?}", - loose_group - .candidates - .iter() - .map(|c| (c.strategy, c.provenance)) - .collect::>() - ); - - let cost_model = DefaultCostModel; - let ranked = space.cost_sorted(&cost_model); - let loose_ranked = ranked - .iter() - .find(|group| Rc::ptr_eq(group.target, &space.roots[0].1)) - .expect("loose has its own ranked group"); - assert!( - loose_ranked.costs.iter().all(|cost| cost.is_finite()), - "no candidate should cost NaN under DefaultCostModel, shared or not: {:?}", - loose_ranked.costs - ); - - let selected = space.global_selection(&cost_model); - let chosen = selected - .for_target(&space.roots[0].1) - .and_then(|group| group.chosen); - assert!( - chosen.is_some(), - "global_selection must commit to some candidate for the shared looser target" - ); - // Under `DefaultCostModel`'s numbers, `CseShare` (flat maintenance, - // no recompute term) and this strategy's own candidate (also a - // flat, non-scaling read cost after the fix) land tied, and - // `global_selection` breaks ties in `CseShare`'s favor (it only ever - // overrides the CSE choice on a *strict* `<`, not `<=`) — a sane, - // deliberate tie-break, not the "reconciliation always loses to - // CseShare regardless of its own real merit" bug this test guards - // against (see `estimate_cost_does_not_scale_with_the_readers_own_consumer_count` - // for the direct regression check that the old `* consumer_count` - // scaling — which made this an unfair, ever-widening loss instead - // of a tie — is gone). - assert_eq!( - chosen.map(|c| c.provenance), - Some(crate::replacement::ReplacementProvenance::CseShare) - ); - } - - #[test] - fn global_selection_propagates_reconciled_consumers_to_the_tighter_group() { - struct PreferReconciliation; - - impl CostModel for PreferReconciliation { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn estimate_cost( - &self, - candidate: &ReplacementSubDAG, - _target: &TargetSubDAG<'_>, - ) -> f64 { - if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { - 0.0 - } else { - 100.0 - } - } - } - - let scan = metric_scan(); - let tight = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); - let loose = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); - let space = crate::replacement::search_workload(vec![ - ("tight", Rc::clone(&tight)), - ("loose", Rc::clone(&loose)), - ]); - - let selected = space.global_selection(&PreferReconciliation); - let tight_root = &space.roots[0].1; - let loose_root = &space.roots[1].1; - assert_eq!( - selected - .for_target(loose_root) - .and_then(|group| group.chosen) - .map(|candidate| candidate.provenance), - Some(ReplacementProvenance::AccuracyReconciliation), - "fixture must select the cross-sibling rewrite" - ); - assert_eq!( - selected - .for_target(tight_root) - .expect("the tighter sibling is a discovered memo group") - .effective_consumer_count, - 2, - "the tighter build serves its original root and the reconciled looser root" - ); - } } diff --git a/crates/asap-aware-mapping/src/topk_reuse.rs b/crates/logical-optimizer/src/pass2/topk_reuse.rs similarity index 58% rename from crates/asap-aware-mapping/src/topk_reuse.rs rename to crates/logical-optimizer/src/pass2/topk_reuse.rs index 8329569d1..f58c35d79 100644 --- a/crates/asap-aware-mapping/src/topk_reuse.rs +++ b/crates/logical-optimizer/src/pass2/topk_reuse.rs @@ -7,30 +7,31 @@ use std::rc::Rc; -use asap_types::pre_asap::QueryExpr; +use asap_types::ir::{NonASAPOp, OperatorNode}; -use crate::replacement::{ +use crate::pass1::replacement::{ Replacement, ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, TargetSubDAG, }; /// Derives a smaller top-k result from a compatible larger top-k sibling. pub struct TopKLimitReuseStrategy { - limits: Vec>, + limits: Vec>, } impl TopKLimitReuseStrategy { - pub fn new(limits: &[Rc]) -> Self { + pub fn new(limits: &[Rc]) -> Self { Self { limits: limits.to_vec(), } } - fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { - let QueryExpr::Limit { - n: target_n, + fn larger_sources<'a>(&'a self, target: &TargetSubDAG<'_>) -> Vec<&'a Rc> { + let Some(NonASAPOp::Limit { + n: Some(target_n), offset: 0, child: target_child, - } = target.root.as_ref() + .. + }) = target.root.non_asap() else { return Vec::new(); }; @@ -42,11 +43,12 @@ impl TopKLimitReuseStrategy { if Rc::ptr_eq(candidate, target.root) { return false; } - let QueryExpr::Limit { - n, + let Some(NonASAPOp::Limit { + n: Some(n), offset: 0, child, - } = candidate.as_ref() + .. + }) = candidate.non_asap() else { return false; }; @@ -56,8 +58,8 @@ impl TopKLimitReuseStrategy { .collect(); // Prefer the smallest sufficient materialized top-k when several // larger siblings are available. - sources.sort_by_key(|source| match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + sources.sort_by_key(|source| match source.non_asap() { + Some(NonASAPOp::Limit { n: Some(n), .. }) => *n, _ => unreachable!(), }); sources @@ -70,34 +72,38 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { } fn replacements(&self, target: &TargetSubDAG<'_>) -> Vec { - let QueryExpr::Limit { - n: target_n, + let Some(NonASAPOp::Limit { + n: Some(target_n), offset: 0, + partition_by, .. - } = target.root.as_ref() + }) = target.root.non_asap() else { return Vec::new(); }; self.larger_sources(target) .into_iter() - .map(|source| { - let source_n = match source.as_ref() { - QueryExpr::Limit { n, .. } => *n, + .filter_map(|source| { + let source_n = match source.non_asap() { + Some(NonASAPOp::Limit { n: Some(n), .. }) => *n, _ => unreachable!(), }; - ReplacementSubDAG { + let rewritten = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(*target_n), + offset: 0, + partition_by: partition_by.clone(), + child: Rc::clone(source), + })) + .ok()?; + Some(ReplacementSubDAG { strategy: "TopKLimitReuseStrategy", - replacement: Replacement::Rewrite(Rc::new(QueryExpr::Limit { - n: *target_n, - offset: 0, - child: Rc::clone(source), - })), + replacement: Replacement::SubDAG(rewritten), provenance: ReplacementProvenance::LogicalRewrite, rationale: format!( "derives top-{target_n} from the compatible shared top-{source_n} result; both rank the identical input with the same ordering" ), - } + }) }) .collect() } @@ -106,38 +112,39 @@ impl ReplacementStrategy for TopKLimitReuseStrategy { #[cfg(test)] mod tests { use super::*; - use asap_types::pre_asap::{Schema, Source}; + use crate::test_support::scan; + use asap_types::ir::operator::operator_properties::GroupKeys; + use asap_types::ir::schema::Schema; - fn scan_named(metric: &str) -> Rc { - Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![], - schema: Schema::with_time_index(vec![], 0, vec![]), - }) + fn scan_named(metric: &str) -> Rc { + scan(metric, Schema::with_time_index(vec![], 0, vec![])) + } + + fn limit(n: usize, offset: usize, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(n), + offset, + partition_by: GroupKeys::none(), + child, + })) + .unwrap() } #[test] fn smaller_limit_reuses_larger_compatible_limit() { let child = scan_named("m"); - let small = Rc::new(QueryExpr::Limit { - n: 5, - offset: 0, - child: Rc::clone(&child), - }); - let large = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child, - }); + let small = limit(5, 0, Rc::clone(&child)); + let large = limit(10, 0, child); let strategy = TopKLimitReuseStrategy::new(&[Rc::clone(&small), Rc::clone(&large)]); let replacements = strategy.replacements(&TargetSubDAG::new(&small)); assert_eq!(replacements.len(), 1); - let Replacement::Rewrite(rewrite) = &replacements[0].replacement else { + let Replacement::SubDAG(rewrite) = &replacements[0].replacement else { panic!() }; - let QueryExpr::Limit { n: 5, child, .. } = rewrite.as_ref() else { + let Some(NonASAPOp::Limit { + n: Some(5), child, .. + }) = rewrite.non_asap() + else { panic!() }; assert!(Rc::ptr_eq(child, &large)); @@ -147,21 +154,9 @@ mod tests { fn offset_or_different_input_is_not_reused() { let a = scan_named("a"); let b = scan_named("b"); - let small = Rc::new(QueryExpr::Limit { - n: 5, - offset: 0, - child: a, - }); - let large = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child: b, - }); - let offset = Rc::new(QueryExpr::Limit { - n: 20, - offset: 1, - child: scan_named("a"), - }); + let small = limit(5, 0, a); + let large = limit(10, 0, b); + let offset = limit(20, 1, scan_named("a")); let strategy = TopKLimitReuseStrategy::new(&[Rc::clone(&small), large, offset]); assert!(!strategy.matches(&TargetSubDAG::new(&small))); } diff --git a/crates/logical-optimizer/src/test_support.rs b/crates/logical-optimizer/src/test_support.rs new file mode 100644 index 000000000..368b5234b --- /dev/null +++ b/crates/logical-optimizer/src/test_support.rs @@ -0,0 +1,213 @@ +// Shared fixture helpers; not every test module uses every helper. +#![allow(dead_code)] + +use std::rc::Rc; + +use asap_types::ir::OperatorNode; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, + Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, +}; + +pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Rc { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .pop() + .unwrap() +} + +// ── Shared pre-ASAP fixture builders ───────────────────────────────────── +// +// Every builder returns an `Rc` whose schema is derived by +// `OperatorNode::new_shared`, so a fixture is exactly what a front end +// would hand the planner. Added by the test migration; only add here, never +// rename or remove (several test modules share these). + +use std::time::Duration; + +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::{GroupKeys, Reduction, Source}; +use asap_types::ir::properties::timing::{ + apply_materialization_timings, MaterializationAssignment, TimingMemo, +}; +use asap_types::ir::schema::{ColumnId, DataType, Field, Schema}; +use asap_types::ir::{NonASAPOp, Predicate, ScalarExpr, TimeRangeKind}; + +/// A `TimeSeries("m")` scan over `[ts(0), value(1), labels...]`, time index 0, +/// no unique key. +pub(crate) fn metric_scan(labels: &[&str]) -> Rc { + metric_scan_with_keys(labels, vec![]) +} + +/// [`metric_scan`] with explicit `unique_keys` (a `[[0]]` key makes CSE +/// willing to hoist the scan). +pub(crate) fn metric_scan_with_keys( + labels: &[&str], + unique_keys: Vec>, +) -> Rc { + let mut columns = vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ]; + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); + scan("m", Schema::with_time_index(columns, 0, unique_keys)) +} + +/// A predicate-free `TimeSeries(metric)` scan with the given schema. +pub(crate) fn scan(metric: &str, schema: Schema) -> Rc { + scan_from( + Source::TimeSeries { + metric: metric.into(), + }, + schema, + ) +} + +pub(crate) fn scan_from(source: Source, schema: Schema) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source, + predicates: vec![], + schema, + })) + .unwrap() +} + +/// A general aggregate node. +pub(crate) fn aggregate( + reduction: Reduction, + measures: Vec, + output_names: Vec, + having: Option, + child: Rc, +) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction, + measures, + output_names, + filters: vec![], + having, + child, + })) + .unwrap() +} + +/// `intent by (by)` — a single-measure, `HAVING`-free grouped aggregate. +pub(crate) fn agg( + by: Vec, + intent: AggIntent, + child: Rc, +) -> Rc { + aggregate(Reduction::by(by), vec![intent], vec![], None, child) +} + +/// `intent without (excluded)`. +pub(crate) fn without_agg( + excluded: Vec, + intent: AggIntent, + child: Rc, +) -> Rc { + aggregate( + Reduction::Reduce(GroupKeys::without(excluded)), + vec![intent], + vec![], + None, + child, + ) +} + +/// A per-entity (per-series) single-measure aggregate. +pub(crate) fn agg_per_entity(intent: AggIntent, child: Rc) -> Rc { + aggregate(Reduction::PerEntity, vec![intent], vec![], None, child) +} + +pub(crate) fn filter(pred: ScalarExpr, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(pred), + child, + })) + .unwrap() +} + +pub(crate) fn dedup(cols: Vec, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols, + child, + })) + .unwrap() +} + +/// An explicit range selector `child[range]`. +pub(crate) fn time_range(range: Duration, child: Rc) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + range, + kind: TimeRangeKind::Range, + child, + })) + .unwrap() +} + +/// `root` timed under the default (every summary at query time) +/// assignment — the shape export and the post-ASAP validators consume. +pub(crate) fn timed(root: &Rc) -> Rc { + apply_materialization_timings( + root, + &MaterializationAssignment::all_query_time(), + &mut TimingMemo::new(), + ) + .expect("default materialization timings apply") +} + +/// `root` timed with every summary maintained at ingestion time. +pub(crate) fn maintained(root: &Rc) -> Rc { + apply_materialization_timings( + root, + &MaterializationAssignment::all_ingestion_time(), + &mut TimingMemo::new(), + ) + .expect("maintained materialization timings apply") +} + +/// Time `root` under the default materialization assignment (which runs every +/// data-state / population-contract check) and export it as a physical ASAP DAG. +pub(crate) fn time_and_export( + root: &Rc, +) -> Result< + asap_types::ir::export::PhysicalASAPDAG, + asap_types::ir::properties::execution::ExecutionDataStateError, +> { + let timed = apply_materialization_timings( + root, + &MaterializationAssignment::all_query_time(), + &mut TimingMemo::new(), + )?; + asap_types::ir::export::compile_physical_asap_dag(&timed) +} diff --git a/crates/logical-optimizer/tests/logical_candidates.rs b/crates/logical-optimizer/tests/logical_candidates.rs new file mode 100644 index 000000000..db2f0bc9f --- /dev/null +++ b/crates/logical-optimizer/tests/logical_candidates.rs @@ -0,0 +1,278 @@ +//! Frontend-to-Pass-1 acceptance: candidate discovery precedes empirical selection. +use asap_logical_optimizer::{ + pass1::logical_candidates::enumerate_local_logical_candidates, + pass1::logical_candidates::local_realizations_for_intent, + pass1::logical_candidates::LogicalCandidateError, Realization, +}; +use asap_types::ir::operator::operator_properties::{Reduction, Source}; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::schema::{DataType, ExactKind, Field, Schema, SketchAlgorithm}; +use asap_types::ir::{NonASAPOp, Operator, OperatorNode, QueryRoot, ScalarExpr}; +use asap_types::types::AccuracyTarget; +use std::rc::Rc; + +fn approximate() -> AccuracyTarget { + AccuracyTarget::EpsilonDelta { + epsilon: 0.05, + delta: 0.01, + } +} +fn algorithms(choices: &[Realization]) -> Vec { + choices + .iter() + .filter_map(|choice| match choice { + Realization::Sketch(kind) => Some(kind.algorithm().clone()), + _ => None, + }) + .collect() +} +fn aggregate(intent: AggIntent) -> Rc { + let child = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "flows".into(), + }, + predicates: vec![], + schema: Schema::lifted(vec![Field::plain("src_ip", DataType::Utf8, false)], None), + })) + .unwrap(); + OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Aggregate { + child, + reduction: Reduction::by(vec![]), + measures: vec![intent], + output_names: vec![], + filters: vec![], + having: None, + })) + .unwrap() +} + +/// Example 2 preserves specialized distinct summaries and the universal alternative. +#[test] +fn cardinality_keeps_exact_specialized_and_universal_alternatives() { + let choices = local_realizations_for_intent(&AggIntent::Cardinality { + cols: vec![0], + accuracy: approximate(), + }) + .unwrap(); + assert!(matches!(choices[0], Realization::PassThrough)); + assert_eq!( + algorithms(&choices), + vec![ + SketchAlgorithm::Hll, + SketchAlgorithm::Theta, + SketchAlgorithm::Kmv, + SketchAlgorithm::UnivMon + ] + ); + let tuple = local_realizations_for_intent(&AggIntent::Cardinality { + cols: vec![0, 1], + accuracy: approximate(), + }) + .unwrap(); + assert!(!algorithms(&tuple).contains(&SketchAlgorithm::UnivMon)); +} + +/// Frequency moments retain exact execution and a universal sketch without certification. +#[test] +fn frequency_statistics_keep_universal_choices() { + for intent in [ + AggIntent::FrequencyL2 { + col: Some(0), + accuracy: approximate(), + }, + AggIntent::FrequencyEntropy { + col: Some(0), + accuracy: approximate(), + }, + ] { + let choices = local_realizations_for_intent(&intent).unwrap(); + assert!(matches!(choices[0], Realization::PassThrough)); + assert_eq!(algorithms(&choices), vec![SketchAlgorithm::UnivMon]); + } +} + +/// An exact request cannot acquire an approximate sketch merely because one is available. +#[test] +fn exact_quantile_stays_exact_and_approximate_keeps_both_families() { + let choices = local_realizations_for_intent(&AggIntent::Quantile { + col: Some(0), + q: 0.99, + accuracy: approximate(), + }) + .unwrap(); + assert_eq!( + algorithms(&choices), + vec![SketchAlgorithm::Kll, SketchAlgorithm::DDSketch] + ); + let exact = local_realizations_for_intent(&AggIntent::Quantile { + col: Some(0), + q: 0.99, + accuracy: AccuracyTarget::Exact, + }) + .unwrap(); + assert_eq!(exact, vec![Realization::PassThrough]); +} + +/// Scalar roots expose their producer targets; repeated references retain one target identity. +#[test] +fn scalar_root_producers_are_discovered_once() { + let producer = aggregate(AggIntent::Cardinality { + cols: vec![0], + accuracy: approximate(), + }); + let roots = vec![ + ( + "scalar", + QueryRoot::Scalar(ScalarExpr::ScalarSubquery(producer.clone())), + ), + ("relation", QueryRoot::Operator(producer.clone())), + ]; + let candidates = enumerate_local_logical_candidates(roots).unwrap(); + assert_eq!(candidates.roots.len(), 2); + for (_, root) in &candidates.roots { + asap_types::ir::export::compile_logical_asap_query(root) + .unwrap() + .validate() + .unwrap(); + } + assert_eq!(candidates.targets.len(), 1); + assert!(Rc::ptr_eq(&candidates.targets[0].target, &producer)); + assert!(producer.timing.is_none()); + assert!(producer.guarantee.is_none()); + asap_types::ir::export::compile_logical_asap_dag(&producer) + .unwrap() + .validate() + .unwrap(); +} + +/// Example 1 rate lowering reaches the exact accumulator choice without a cost model. +#[test] +fn promql_lowering_reaches_local_candidates_without_execution_timing() { + use asap_types::workload::{ + AccuracyRequirement, BatchEntry, PlanningWorkload, Query, QueryLanguage, QueryRequirements, + QueryWorkload, + }; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("sum by (job) (rate(http_requests_total[1m]))".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(approximate()), + ..Default::default() + }, + predictability: Default::default(), + invocations: 1, + execute_at: None, + time_selection: Default::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(asap_types::workload::DataWorkload { + data_ingestion_interval: asap_types::workload::Evidence { + value: Some(asap_types::workload::DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let roots = asap_frontend_promql::lower_promql_query_workload(&workload, 0).unwrap(); + let candidates = + enumerate_local_logical_candidates(roots.into_iter().enumerate().collect()).unwrap(); + assert!(candidates + .targets + .iter() + .any(|target| target.alternatives.iter().any(|choice| matches!( + choice, + Realization::ExactAggregate { + kind: ExactKind::Rate, + .. + } + )))); + assert!(candidates + .targets + .iter() + .all(|target| target.target.timing.is_none())); +} + +/// Physical annotations and invalid probability requirements fail at the stage boundary. +#[test] +fn assigned_timing_and_invalid_accuracy_are_rejected() { + let mut producer = (*aggregate(AggIntent::Count { + accuracy: approximate(), + })) + .clone(); + producer.timing = Some(asap_types::ir::properties::ExecutionTiming::QueryTime); + assert!(matches!( + enumerate_local_logical_candidates(vec![(0, QueryRoot::Operator(Rc::new(producer)))]), + Err(LogicalCandidateError::AssignedTiming) + )); + for target in [ + AccuracyTarget::Epsilon(f64::NAN), + AccuracyTarget::EpsilonDelta { + epsilon: 0.1, + delta: 0.0, + }, + ] { + assert!(matches!( + local_realizations_for_intent(&AggIntent::Count { accuracy: target }), + Err(LogicalCandidateError::InvalidAccuracy) + )); + } +} + +/// Local TopK keeps both declared heap substrates without choosing an implementation. +#[test] +fn topk_keeps_both_specialized_heap_choices() { + let choices = local_realizations_for_intent(&AggIntent::TopK { + k: 10, + accuracy: approximate(), + }) + .unwrap(); + assert!(matches!(choices[0], Realization::PassThrough)); + assert_eq!( + algorithms(&choices), + vec![ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap + ] + ); +} + +/// A workload candidate replaces only chosen targets, declares whole-source +/// coverage on the summary, and keeps unchosen plans identical. +#[test] +fn composed_candidate_replaces_chosen_target_with_summary_evaluation() { + use asap_logical_optimizer::pass1::logical_candidates::compose_logical_candidate; + use asap_types::ir::ASAPOp; + let producer = aggregate(AggIntent::Cardinality { + cols: vec![0], + accuracy: approximate(), + }); + let inventory = + enumerate_local_logical_candidates(vec![(0, QueryRoot::Operator(producer.clone()))]) + .unwrap(); + let exact = compose_logical_candidate(&inventory, &[0]).unwrap(); + assert!(matches!(&exact[0].1, QueryRoot::Operator(node) if Rc::ptr_eq(node, &producer))); + + let hll = compose_logical_candidate(&inventory, &[1]).unwrap(); + let QueryRoot::Operator(estimate) = &hll[0].1 else { + panic!("operator root expected") + }; + let Some(ASAPOp::SummaryEstimate { summary_input, .. }) = estimate.asap() else { + panic!("summary evaluation expected") + }; + let coverage = summary_input.coverage.as_ref().unwrap(); + assert_eq!( + coverage.source, + Source::Table { + table_ref: "flows".into() + } + ); + assert!(coverage.regions[0].time_ms.is_none() && coverage.regions[0].population.is_empty()); + asap_types::ir::export::compile_logical_asap_workload(&[hll[0].1.clone()]) + .unwrap() + .validate() + .unwrap(); + assert!(compose_logical_candidate(&inventory, &[99]).is_err()); +} diff --git a/crates/logical-optimizer/tests/stage1_cost_independence.rs b/crates/logical-optimizer/tests/stage1_cost_independence.rs new file mode 100644 index 000000000..24669ed95 --- /dev/null +++ b/crates/logical-optimizer/tests/stage1_cost_independence.rs @@ -0,0 +1,112 @@ +//! Stage 1 (logical candidate generation) stays independent of the cost +//! model: only Stage 3 prices plans (#572, decision Q36(a)). + +use std::path::{Path, PathBuf}; + +/// Module paths Stage 1 production code must not name. +const FORBIDDEN_MODULES: &[&str] = &["cost_model", "recurrence"]; + +/// Crates Stage 1 must not depend on: later stages, the facade and the executor. +const FORBIDDEN_CRATES: &[&str] = &[ + "asap-physical-optimizer", + "asap-plan-selection", + "asap-planner", + "asap-executor", +]; + +/// Every `.rs` file under `dir`. +fn rust_files(dir: &Path) -> Vec { + let mut files = Vec::new(); + for entry in std::fs::read_dir(dir).unwrap() { + let path = entry.unwrap().path(); + if path.is_dir() { + files.extend(rust_files(&path)); + } else if path.extension().is_some_and(|ext| ext == "rs") { + files.push(path); + } + } + files.sort(); + files +} + +/// The source lines outside `#[cfg(test)]` items and comments, numbered. +fn production_lines(source: &str) -> Vec<(usize, &str)> { + let mut lines = Vec::new(); + let mut skip_next_item = false; + let mut depth = 0i64; + for (index, line) in source.lines().enumerate() { + let trimmed = line.trim(); + if depth > 0 { + depth += brace_balance(line); + continue; + } + if trimmed == "#[cfg(test)]" { + skip_next_item = true; + continue; + } + if skip_next_item { + if trimmed.starts_with("#[") || trimmed.is_empty() { + continue; + } + skip_next_item = false; + depth = brace_balance(line); + continue; + } + if !trimmed.starts_with("//") { + lines.push((index + 1, line)); + } + } + lines +} + +fn brace_balance(line: &str) -> i64 { + line.chars() + .map(|c| match c { + '{' => 1, + '}' => -1, + _ => 0, + }) + .sum() +} + +/// Stage 1 production code names neither `cost_model` nor `recurrence`. +#[test] +fn stage1_does_not_import_cost_model_or_recurrence() { + let src = Path::new(env!("CARGO_MANIFEST_DIR")).join("src"); + let mut offenders = Vec::new(); + for file in rust_files(&src) { + let source = std::fs::read_to_string(&file).unwrap(); + for (number, line) in production_lines(&source) { + if FORBIDDEN_MODULES + .iter() + .any(|module| line.contains(&format!("{module}::"))) + { + let file = file.strip_prefix(&src).unwrap().display(); + offenders.push(format!("{file}:{number}: {}", line.trim())); + } + } + } + assert!( + offenders.is_empty(), + "Stage 1 must not depend on the cost model:\n{}", + offenders.join("\n") + ); +} + +/// The manifest names no later stage, facade or executor crate, so Cargo +/// rejects any import of them. +#[test] +fn stage1_manifest_has_no_path_back_to_later_stages() { + let manifest = + std::fs::read_to_string(Path::new(env!("CARGO_MANIFEST_DIR")).join("Cargo.toml")).unwrap(); + let offenders: Vec<&str> = manifest + .lines() + .filter(|line| !line.trim_start().starts_with('#')) + .filter(|line| FORBIDDEN_CRATES.iter().any(|name| line.contains(name))) + .collect(); + assert!( + offenders.is_empty(), + "asap-logical-optimizer must not depend on a later stage:\n{}", + offenders.join("\n") + ); +} diff --git a/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs b/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs index 3ac1170e8..6befd206d 100644 --- a/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs +++ b/crates/metricsql-parser-vendored/src/optimizer/const_evaluator.rs @@ -12,7 +12,7 @@ use crate::functions::{BuiltinFunction, TransformFunction}; use crate::parser::{parse_number, ParseError, ParseResult}; #[allow(rustdoc::private_intra_doc_links)] -/// Partially evaluate `Expr`s so constant subtrees are evaluated at plan time. +/// Partially evaluate `Expr`s so constant sub-DAGs are evaluated at plan time. /// /// Note it does not handle algebraic rewrites such as `(a or false)` /// --> `a`, which is handled by [`Simplifier`] diff --git a/crates/physical-optimizer/Cargo.toml b/crates/physical-optimizer/Cargo.toml new file mode 100644 index 000000000..24b0ca7db --- /dev/null +++ b/crates/physical-optimizer/Cargo.toml @@ -0,0 +1,17 @@ +[package] +name = "asap-physical-optimizer" +version = "0.1.0" +edition = "2021" + +# #509 Stage 2: physical candidates for each Stage 1 logical candidate. +# Production code depends only on asap-types: a Stage 1 candidate reaches it as +# OperatorNode roots. Stage 1 and the PromQL front end are test-only +# dev-dependencies. Never depends on Stage 3, the facade or the executor. +# tests/stage2_dependencies.rs checks this manifest. +[dependencies] +asap-types = { path = "../types" } +thiserror = "2" + +[dev-dependencies] +asap-logical-optimizer = { path = "../logical-optimizer" } +asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/physical-optimizer/src/implementation/mod.rs b/crates/physical-optimizer/src/implementation/mod.rs new file mode 100644 index 000000000..776d75a93 --- /dev/null +++ b/crates/physical-optimizer/src/implementation/mod.rs @@ -0,0 +1,3 @@ +//! Physical operator implementation of one logical candidate. + +pub mod physical_candidates; diff --git a/crates/physical-optimizer/src/implementation/physical_candidates.rs b/crates/physical-optimizer/src/implementation/physical_candidates.rs new file mode 100644 index 000000000..25507b78c --- /dev/null +++ b/crates/physical-optimizer/src/implementation/physical_candidates.rs @@ -0,0 +1,281 @@ +//! #509 Stage 2 (MVP): physical operator implementation of one logical +//! candidate. No materialization choice is made: every node runs at query +//! time, so each logical candidate yields exactly one physical candidate. +//! +//! The runtime has one implementation per logical operator except exact +//! top-k, which it cannot run as `Aggregate{[TopK]}`. Stage 2 records that +//! choice in the DAG by rewriting it to a per-group sort followed by a +//! per-group limit, the shape the PromQL frontend uses for generic `topk`. +//! A summary needs no rewrite: `SummaryAgg` → `SummaryEstimate` already is +//! build → estimate. +use std::collections::HashMap; +use std::rc::Rc; + +use asap_types::ir::export::{compile_physical_asap_workload_with_node_ids, PhysicalASAPDAG}; +use asap_types::ir::operator::{AggIntent, Reduction}; +use asap_types::ir::properties::ExecutionDataStateError; +use asap_types::ir::scalar::column_resolution::resolve_column_ref; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::{ + apply_materialization_timings, ASAPOp, MaterializationAssignment, NonASAPOp, Operator, + OperatorNode, ScalarExpr, SchemaDerivationError, SortKey, TimingMemo, +}; +use thiserror::Error; + +/// One Stage 2 candidate, derived from exactly one Stage 1 candidate. +/// `roots` are the timed operator roots (one per query, in workload order) +/// that `dag` exports. +#[derive(Debug, Clone)] +pub struct PhysicalCandidate { + pub id: String, + pub from_logical: String, + pub label: String, + pub roots: Vec>, + pub dag: PhysicalASAPDAG, +} + +#[derive(Debug, Error)] +pub enum Stage2Error { + #[error(transparent)] + Structure(#[from] SchemaDerivationError), + #[error(transparent)] + Timing(#[from] ExecutionDataStateError), + #[error("exact top-k has no sortable value column: {0}")] + NoValueColumn(String), + #[error("maintained populations run at ingestion time, which Stage 2 does not plan yet")] + IngestionTimeOnly, +} + +/// Implement every operator of one logical candidate and time it at query +/// time. `id` and `label` are left empty for the caller to name. Sharing +/// between `roots` is preserved: one memo serves the whole workload. +pub fn stage2_physical( + from_logical: &str, + roots: &[Rc], +) -> Result { + let mut memo = HashMap::new(); + let implemented = roots + .iter() + .map(|root| implement(root, &mut memo)) + .collect::, _>>()?; + reject_maintained_populations(&implemented)?; + // The default assignment computes every summary state at query time. + let assignment = MaterializationAssignment::default(); + let mut timing = TimingMemo::new(); + let timed = implemented + .iter() + .map(|root| apply_materialization_timings(root, &assignment, &mut timing)) + .collect::, _>>()?; + let dag = compile_physical_asap_workload_with_node_ids(&timed)?.dag; + Ok(PhysicalCandidate { + id: String::new(), + from_logical: from_logical.to_string(), + label: String::new(), + roots: timed, + dag, + }) +} + +/// A maintained population always runs at ingestion time, which Stage 2 +/// does not plan yet. +fn reject_maintained_populations(roots: &[Rc]) -> Result<(), Stage2Error> { + let maintained = roots.iter().flat_map(OperatorNode::reachable).any(|node| { + matches!( + node.operator, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) + ) + }); + if maintained { + return Err(Stage2Error::IngestionTimeOnly); + } + Ok(()) +} + +type Memo = HashMap<*const OperatorNode, Rc>; + +fn implement(node: &Rc, memo: &mut Memo) -> Result, Stage2Error> { + if let Some(done) = memo.get(&Rc::as_ptr(node)) { + return Ok(done.clone()); + } + for child in node.children() { + implement(child, memo)?; + } + let implemented = match node.non_asap() { + Some(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(keys), + measures, + filters, + having: None, + child, + .. + }) if filters.iter().all(Option::is_none) => match measures.as_slice() { + [AggIntent::TopK { k, .. }] => { + let child = memo[&Rc::as_ptr(child)].clone(); + let value = resolve_column_ref(&ColumnRef::SampleValue, &child.schema) + .map_err(|e| Stage2Error::NoValueColumn(e.to_string()))?; + let sorted = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Sort { + keys: vec![SortKey { + expr: ScalarExpr::Column(value), + ascending: false, + nulls_first: false, + }], + partition_by: keys.clone(), + child, + }))?; + OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Limit { + n: Some(*k), + offset: 0, + partition_by: keys.clone(), + child: sorted, + }))? + } + _ => rebuilt(node, memo)?, + }, + _ => rebuilt(node, memo)?, + }; + memo.insert(Rc::as_ptr(node), implemented.clone()); + Ok(implemented) +} + +/// `node` over its implemented children; the same `Rc` when none changed. +fn rebuilt(node: &Rc, memo: &Memo) -> Result, Stage2Error> { + let changed = node + .children() + .iter() + .any(|child| !Rc::ptr_eq(child, &memo[&Rc::as_ptr(child)])); + Ok(if changed { + Rc::new(node.map_children(|child| memo[&Rc::as_ptr(child)].clone())?) + } else { + node.clone() + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::test_support::lower_promql; + use asap_types::ir::export::PhysicalASAPOperatorPayload as Payload; + use asap_types::ir::export::{NonASAPOpKind, PhysicalASAPNodeId}; + use asap_types::ir::properties::ExecutionTiming; + use asap_types::ir::QueryRoot; + use asap_types::types::AccuracyTarget; + + /// Exact `topk by (job)` becomes Limit(Sort) partitioned by `job`, and a + /// child shared by two roots stays one node (one `Rc`, one DAG node). + #[test] + fn exact_topk_becomes_sort_then_limit_and_keeps_shared_child() { + let exact = AccuracyTarget::Exact; + let topk = lower_promql("topk by (job) (10, sum_over_time(m[1m]))", exact.clone()); + let sum = lower_promql("sum by (job) (sum_over_time(m[1m]))", exact); + // Share Q2's per-series child with Q1, as Stage 1 sharing would. + let NonASAPOp::Aggregate { child: shared, .. } = topk.expect_non_asap() else { + panic!("topk aggregate") + }; + let sum = Rc::new(sum.map_children(|_| shared.clone()).unwrap()); + let candidate = stage2_physical("L1", &[topk, sum]).unwrap(); + + let NonASAPOp::Limit { + n: Some(10), + partition_by, + child: sort, + .. + } = candidate.roots[0].expect_non_asap() + else { + panic!("limit root") + }; + assert!(!partition_by.keys().is_empty()); + let NonASAPOp::Sort { + child: below_sort, + partition_by: sort_partition, + .. + } = sort.expect_non_asap() + else { + panic!("sort below limit") + }; + assert_eq!(sort_partition, partition_by); + let NonASAPOp::Aggregate { + child: below_sum, .. + } = candidate.roots[1].expect_non_asap() + else { + panic!("sum root") + }; + assert!(Rc::ptr_eq(below_sort, below_sum)); + + let dag = &candidate.dag; + let kind = |id: PhysicalASAPNodeId| match &dag + .nodes + .iter() + .find(|n| n.id == id) + .unwrap() + .payload + { + Payload::Relational { + operator: NonASAPOpKind::Sort { .. }, + } => "sort", + Payload::Relational { + operator: NonASAPOpKind::Limit { .. }, + } => "limit", + _ => "other", + }; + let sort_id = dag + .edges + .iter() + .find(|e| kind(e.producer) == "sort" && e.consumer == dag.roots[0]) + .expect("sort -> limit edge") + .producer; + let producer_of = |consumer| { + dag.edges + .iter() + .find(|e| e.consumer == consumer) + .unwrap() + .producer + }; + assert_eq!(producer_of(sort_id), producer_of(dag.roots[1])); + } + + /// With summaries chosen, every node, the summary build included, runs at + /// query time. + #[test] + fn everything_runs_at_query_time() { + let root = lower_promql( + "topk by (job) (10, sum_over_time(m[1m]))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }, + ); + let inventory = + asap_logical_optimizer::pass1::logical_candidates::enumerate_local_logical_candidates( + vec![(0, QueryRoot::Operator(root))], + ) + .unwrap(); + // A summary (the last alternative) for every target. + let choice: Vec<_> = inventory + .targets + .iter() + .map(|t| t.alternatives.len() - 1) + .collect(); + let roots: Vec<_> = + asap_logical_optimizer::pass1::logical_candidates::compose_logical_candidate( + &inventory, &choice, + ) + .unwrap() + .into_iter() + .map(|(_, root)| match root { + QueryRoot::Operator(node) => node, + QueryRoot::Scalar(_) => panic!("operator root"), + }) + .collect(); + let candidate = stage2_physical("L1", &roots).unwrap(); + assert!(candidate + .dag + .nodes + .iter() + .any(|n| matches!(n.payload, Payload::SummaryAgg { .. }))); + assert!(candidate + .dag + .nodes + .iter() + .all(|n| n.output_state.timing == ExecutionTiming::QueryTime)); + } +} diff --git a/crates/physical-optimizer/src/lib.rs b/crates/physical-optimizer/src/lib.rs new file mode 100644 index 000000000..77fb5396a --- /dev/null +++ b/crates/physical-optimizer/src/lib.rs @@ -0,0 +1,15 @@ +//! `asap-physical-optimizer` — #509 Stage 2: physical candidates. +//! +//! It turns each Stage 1 logical candidate into physical candidates: how each +//! operator is implemented, and (once Stage 2 materialization exists) which +//! sub-DAGs are materialized and when. Cargo enforces the stage order: this +//! crate depends only on `asap-types`, never on Stage 3, the facade or the +//! executor. +//! +//! - [`implementation`] — physical operator implementation. Every node runs at +//! query time until Stage 2 materialization (#509) exists. + +pub mod implementation; + +#[cfg(test)] +mod test_support; diff --git a/crates/asap-aware-mapping/src/test_support.rs b/crates/physical-optimizer/src/test_support.rs similarity index 86% rename from crates/asap-aware-mapping/src/test_support.rs rename to crates/physical-optimizer/src/test_support.rs index 612cf7c0d..f6721a3d7 100644 --- a/crates/asap-aware-mapping/src/test_support.rs +++ b/crates/physical-optimizer/src/test_support.rs @@ -1,11 +1,16 @@ -use asap_types::pre_asap::QueryExpr; +// Fixture helpers for this crate's tests: the subset of +// `asap-logical-optimizer`'s `test_support` that Stage 2 tests use. + +use std::rc::Rc; + +use asap_types::ir::OperatorNode; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, }; -pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> QueryExpr { +pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Rc { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, diff --git a/crates/physical-optimizer/tests/stage2_dependencies.rs b/crates/physical-optimizer/tests/stage2_dependencies.rs new file mode 100644 index 000000000..70e991d2d --- /dev/null +++ b/crates/physical-optimizer/tests/stage2_dependencies.rs @@ -0,0 +1,25 @@ +//! Stage 2 (physical candidates) depends on no later stage: Cargo enforces +//! the one-way #509 stage flow (#572). + +use std::path::Path; + +/// Crates Stage 2 must not depend on: Stage 3, the facade and the executor. +const FORBIDDEN_CRATES: &[&str] = &["asap-plan-selection", "asap-planner", "asap-executor"]; + +/// The manifest names no later stage, facade or executor crate, so Cargo +/// rejects any import of them. +#[test] +fn stage2_manifest_has_no_path_back_to_later_stages() { + let manifest = + std::fs::read_to_string(Path::new(env!("CARGO_MANIFEST_DIR")).join("Cargo.toml")).unwrap(); + let offenders: Vec<&str> = manifest + .lines() + .filter(|line| !line.trim_start().starts_with('#')) + .filter(|line| FORBIDDEN_CRATES.iter().any(|name| line.contains(name))) + .collect(); + assert!( + offenders.is_empty(), + "asap-physical-optimizer must not depend on a later stage:\n{}", + offenders.join("\n") + ); +} diff --git a/crates/plan-selection/Cargo.toml b/crates/plan-selection/Cargo.toml new file mode 100644 index 000000000..c4e01e1d5 --- /dev/null +++ b/crates/plan-selection/Cargo.toml @@ -0,0 +1,18 @@ +[package] +name = "asap-plan-selection" +version = "0.1.0" +edition = "2021" + +# #509 Stage 3: plan selection, the only stage that uses the cost model. +# Depends on asap-types, Stage 1 and Stage 2; never on the facade or the +# executor. tests/stage3_dependencies.rs checks this manifest. +[dependencies] +asap-types = { path = "../types" } +asap-logical-optimizer = { path = "../logical-optimizer" } +asap-physical-optimizer = { path = "../physical-optimizer" } +thiserror = "2" +serde = { version = "1", features = ["derive"] } +serde_json = "1" + +[dev-dependencies] +asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/plan-selection/src/candidate_selection.rs b/crates/plan-selection/src/candidate_selection.rs new file mode 100644 index 000000000..5a86a6903 --- /dev/null +++ b/crates/plan-selection/src/candidate_selection.rs @@ -0,0 +1,2860 @@ +//! Legacy whole-workload selection over the Stage 1 search space +//! ([`CandidateLogicalASAPDAGs`]): per-target cost ranking, recurrence +//! profiles and global selection. These are free functions over Stage 1 +//! types, so that Stage 1 does not depend on the cost model. Assembling the +//! selected DAG is Stage 1's [`GlobalSelection`]; [`CostedGlobalSelection`] +//! adds the cost comparison behind each chosen exact composition. +//! +//! The stage pipeline does not call this module. It is deleted under #580. + +use std::collections::{HashMap, HashSet, VecDeque}; +use std::rc::Rc; + +use asap_types::ir::schema::{FieldDataType, GroupingStrategy, SketchAlgorithm}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; + +use crate::cost::cost_model::{ + raw_recompute_cost_rate, CostModel, CseCandidate, ExactCompositionCostInputs, + ExactCompositionCostRequest, ShareDecision, +}; +use crate::cost::recurrence::{ + CostRate, Horizon, RecurrenceError, RecurrenceProfile, RootRecurrence, UpdateRate, +}; +use asap_logical_optimizer::pass1::exact_composition::OperationPlacement; +use asap_logical_optimizer::pass1::replacement::{ + bindable_intent, cse_candidate_pair, direct_child_counts, is_logical_rewrite, realize_child, + CandidateLogicalASAPDAGs, GlobalSelection, PreparedComposition, Replacement, + ReplacementProvenance, ReplacementSubDAG, TargetSubDAG, TargetSubDAGCandidates, + TargetSubDAGSelection, +}; + +/// Physical feasibility evidence for `candidate`. A pure logical +/// rewrite needs no new operator. Unknown support is checked during +/// physical/deployment compilation; explicit rejection prevents selection. +pub fn runtime_support_evidence( + candidate: &ReplacementSubDAG, + cost_model: &dyn CostModel, +) -> Option { + match &candidate.replacement { + Replacement::ExactComposition(composition) => { + cost_model.value_operation_support_evidence(&composition.op, composition.placement) + } + // Any summary decision, including one rooted in a relational + // operator above its evaluations, asks the deployment for support. + Replacement::SubDAG(node) if !is_logical_rewrite(node) => { + cost_model.summary_support_evidence(node) + } + Replacement::SubDAG(_) => Some(true), + } +} + +/// The `sorted_by(cost_model)` step: every group, each with its own +/// candidates ranked best-first under `cost_model` where this module +/// knows how (see the module docs' "Cost-based final selection" +/// section) — groups themselves stay in discovery order, since targets +/// are independent decision points, not alternatives competing with +/// each other. +/// +/// Ranking itself is decided entirely by [`rank_group`] before +/// [`RankedTargetSubDAGCandidates::costs`] is ever computed — pairing each candidate with +/// [`CostModel::grouping_state_cost`] for grouping alternatives, or +/// [`CostModel::estimate_cost`] otherwise, is an additive annotation +/// for a caller that wants to *display* a cost (e.g. a +/// DAG-visualization view), not a second ranking signal, so plugging in +/// a `CostModel` whose `estimate_cost` disagrees with its own +/// `rank_candidates`/`cse_share_decision` (a deployment bug, not +/// something this method tries to protect against) would show a +/// `RankedTargetSubDAGCandidates` whose `costs` aren't monotonically non-decreasing — +/// `cost_sorted`'s own ordering guarantee is unaffected either way. +pub fn cost_sorted<'a, Id>( + space: &'a CandidateLogicalASAPDAGs, + cost_model: &dyn CostModel, +) -> Vec> { + space + .order() + .iter() + .map(|ptr| { + let group = &space.groups()[ptr]; + let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); + let mut candidates = rank_group(group, cost_model); + // Availability is candidate-specific and cannot be expressed + // by `rank_candidates`' exhaustive permutation contract. + // Keep unavailable alternatives for explanation, but place + // them after every selectable candidate. + candidates + .sort_by_key(|candidate| cost_model.candidate_cost(candidate, &target).is_none()); + let costs = candidates + .iter() + .map(|c| { + cost_model + .grouping_state_cost(c, &target) + .map_or_else(|| cost_model.estimate_cost(c, &target), |cost| cost.0) + }) + .collect(); + RankedTargetSubDAGCandidates { + target: &group.target, + consumer_count: group.consumer_count, + candidates, + costs, + } + }) + .collect() +} + +/// Recurrence-aware counterpart to [`cost_sorted`]. CSE +/// share/recompute pairs are ordered with the target's recurrence +/// profile; all other candidate shapes retain their existing ranking. +pub fn cost_sorted_with_recurrence<'a, Id>( + space: &'a CandidateLogicalASAPDAGs, + cost_model: &dyn CostModel, + profiles: &RecurrenceProfileMap, + horizon: Option, +) -> Result>, RecurrenceError> { + space + .order() + .iter() + .map(|ptr| { + let group = &space.groups()[ptr]; + let mut candidates = rank_group(group, cost_model); + if cse_candidate_pair(group).is_some() { + if let Some(decision) = decide_group_with_recurrence( + group, + group.consumer_count, + profiles.for_target(&group.target), + horizon, + cost_model, + )? { + candidates.sort_by_key(|candidate| match candidate.provenance { + ReplacementProvenance::CseShare if decision == ShareDecision::Share => 0, + ReplacementProvenance::CseRecompute + if decision == ShareDecision::RecomputeIndependently => + { + 0 + } + ReplacementProvenance::CseShare | ReplacementProvenance::CseRecompute => 2, + _ => 1, + }); + } + } + let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); + let costs = candidates + .iter() + .map(|candidate| { + cost_model + .grouping_state_cost(candidate, &target) + .map_or_else( + || cost_model.estimate_cost(candidate, &target), + |cost| cost.0, + ) + }) + .collect(); + Ok(RankedTargetSubDAGCandidates { + target: &group.target, + consumer_count: group.consumer_count, + candidates, + costs, + }) + }) + .collect() +} + +// ── Recurrence-aware cost context (issue #287) ────────────────────────── + +/// One [`RecurrenceProfile`] per discovered [`TargetSubDAGCandidates`] target, built by +/// [`recurrence_profiles`] — the "carry `RepeatingEntry.demand` +/// and relevant `DataWorkload` into ASAP-aware search/cost context" +/// half of issue #287. Looked up by `Rc` pointer identity, the same +/// currency [`CandidateLogicalASAPDAGs::candidates_for_target`]/[`GlobalSelection::for_target`] already +/// use. +/// Holds an owned `Rc` clone alongside each profile (not just +/// its raw pointer) so this map keeps every node it describes alive for as +/// long as the map itself lives — a `RecurrenceProfileMap` is safe to outlive +/// the `CandidateLogicalASAPDAGs` it was built from. Without this, a raw `*const OperatorNode` key +/// could, after the originating `CandidateLogicalASAPDAGs` (the only other owner of those +/// `Rc`s) is dropped, collide with an unrelated, later allocation that +/// happens to reuse the same freed address — silently returning a stale +/// profile for the wrong node (issue #287 review, bug 4). +#[derive(Debug, Clone)] +pub struct RecurrenceProfileMap { + profiles: HashMap<*const OperatorNode, (Rc, RecurrenceProfile)>, +} + +impl RecurrenceProfileMap { + /// The [`RecurrenceProfile`] for `target`, or + /// [`RecurrenceProfile::EMPTY`] when `target` wasn't a discovered site + /// in the [`CandidateLogicalASAPDAGs`] this map was built from (or carried no + /// recurring/one-shot/update-rate metadata at all) — always a valid, + /// "no metadata" answer, never a panic. + pub fn for_target(&self, target: &Rc) -> RecurrenceProfile { + self.profiles + .get(&Rc::as_ptr(target)) + .map(|(_, profile)| *profile) + .unwrap_or(RecurrenceProfile::EMPTY) + } +} + +/// Build one [`RecurrenceProfile`] per discovered site, by walking every +/// root's whole reachable sub-DAG (the same relational-skeleton +/// traversal `discover_targets` itself used to discover those sites) +/// and folding each root's own recurrence tag +/// (a normalized repeating rate or a one-time invocation count) into every +/// site reachable from it. +/// +/// `root_recurrence` is positional: `root_recurrence[i]` describes +/// `space.roots[i]` — the same order [`search_workload`]/ +/// [`search_workload_with`] were originally called with (post-CSE +/// dedup preserves both root count and order — see +/// `asap_types::ir::cse::share_common_sub_dags`'s own +/// `.map(...).collect()` body). This keeps `Id` fully opaque (no `Eq`/ +/// `Hash`/`Clone` bound needed on it at all — issue #287's "keep +/// caller/query identifiers opaque" requirement) at the cost of the +/// caller keeping the two slices in step; `root_recurrence.len()` must +/// equal `space.roots.len()`. +/// +/// A shared sub-DAG reachable from more than one root aggregates every +/// reaching root's contribution — repeating roots' rates are summed and +/// one-shot roots +/// increment [`RecurrenceProfile::one_shot_consumers`] — so a summary +/// consumed by queries with different intervals gets one profile +/// reflecting all of them, per issue #287's "support a shared sub-DAG +/// consumed by queries with different intervals". +/// +/// `update_rate` is applied uniformly to every discovered site *that +/// this walk actually reached from some root* (see the "unreachable +/// sites" note below): today's +/// [`asap_types::workload::DataWorkload`] is a single +/// workload-level value (applies to every query in a `QueryWorkload`), +/// not per-target, so there is no finer-grained source to attach +/// instead. `None` when no `DataWorkload` evidence was available — +/// preserves "missing metadata" behavior for the update-rate term alone +/// even when repeating/one-shot consumer information is present. +/// +/// A parent that structurally references the same child more than once +/// (e.g. `BinaryOp{lhs: X, rhs: X}`) credits that child with one +/// contribution per reference, not one contribution per distinct node — +/// matching how [`TargetSubDAGCandidates::consumer_count`] counts that occurrence. +/// Multiplicity is propagated through the full descendant path: if the +/// repeated parent is independently evaluated twice, its child is also +/// evaluated twice. This supplies recurrence-aware selection with the +/// effective structural execution rate rather than mere reachability. +/// +/// **Unreachable sites**: [`CandidateLogicalASAPDAGs`] can contain a site no root's own +/// structural DAG actually reaches — e.g. one only ever produced by a +/// [`Replacement::Rewrite`] candidate a [`ReplacementStrategy`] invented +/// (this walk only follows [`TargetSubDAGCandidates::target`]'s own structural +/// children, the same scope `discover_targets` uses for the original +/// roots, never a candidate's rewritten value). Such a site gets +/// [`RecurrenceProfile::EMPTY`] — in particular, `update_rate` is +/// **not** stamped onto it — so it falls back to the ordinary +/// structural decision instead of being charged an ingest-driven +/// maintenance cost against a real evaluation/one-shot signal of +/// exactly zero, which previously made `RecomputeIndependently` win +/// there unconditionally, regardless of the site's actual +/// `consumer_count` (issue #287 review, bug 2). +/// +/// Returns [`RecurrenceError::InvalidEvaluationRate`] if any repeating +/// rate is non-finite or negative, +/// [`RecurrenceError::InvalidUpdateRate`] if `update_rate` is non-finite +/// or negative, or [`RecurrenceError::RootCountMismatch`] if +/// `root_recurrence.len() != space.roots.len()`. +pub fn recurrence_profiles( + space: &CandidateLogicalASAPDAGs, + root_recurrence: &[RootRecurrence], + update_rate: Option, +) -> Result { + if root_recurrence.len() != space.roots.len() { + return Err( + crate::cost::recurrence::RecurrenceError::RootCountMismatch { + expected: space.roots.len(), + got: root_recurrence.len(), + }, + ); + } + if let Some(rate) = update_rate { + crate::cost::recurrence::validate_update_rate(rate)?; + } + for recurrence in root_recurrence { + if let RootRecurrence::Repeating(rate) = recurrence { + if !rate.0.is_finite() || rate.0 < 0.0 { + return Err(crate::cost::recurrence::RecurrenceError::InvalidEvaluationRate(*rate)); + } + } + } + + let mut rates: HashMap<*const OperatorNode, f64> = HashMap::new(); + let mut one_shot_counts: HashMap<*const OperatorNode, usize> = HashMap::new(); + // Sites actually reached by at least one root's own recurrence tag + // during the walk below — see this method's own "Unreachable + // sites" doc. + let mut reached: HashSet<*const OperatorNode> = HashSet::new(); + + for ((_, root), recurrence) in space.roots.iter().zip(root_recurrence) { + let recurrence = *recurrence; + let root_ptr = Rc::as_ptr(root); + // Carry path multiplicity transitively. If a shared ancestor is + // referenced twice, every descendant below an independently + // recomputed occurrence is evaluated twice as well; stopping + // expansion after the first pointer visit undercounts exactly + // the effective-consumer rate recurrence-aware costing needs. + let mut queue: VecDeque<(*const OperatorNode, usize)> = VecDeque::new(); + queue.push_back((root_ptr, 1)); + + while let Some((ptr, path_count)) = queue.pop_front() { + contribute( + ptr, + path_count, + recurrence, + &mut rates, + &mut one_shot_counts, + &mut reached, + ); + // Every reachable node was itself discovered as its own + // `TargetSubDAGCandidates` (`discover_targets` walks the identical + // relational-skeleton scope) — its own `target` is the + // canonical `Rc` to read children off. + if let Some(group) = space.groups().get(&ptr) { + for (child, edge_count) in direct_child_counts(&group.target) { + queue.push_back(( + child, + path_count + .checked_mul(edge_count) + .expect("query DAG path multiplicity overflowed usize"), + )); + } + } + } + } + + let mut profiles = HashMap::with_capacity(space.order().len()); + for ptr in space.order() { + let rate = rates.get(ptr).copied().unwrap_or(0.0); + let evaluation_rate = (rate > 0.0).then_some(crate::cost::recurrence::EvaluationRate(rate)); + let one_shot_consumers = one_shot_counts.get(ptr).copied().unwrap_or(0); + // Bug 2 fix (see "Unreachable sites" above): only a reached + // site carries the caller-supplied `update_rate`. + let site_update_rate = if reached.contains(ptr) { + update_rate + } else { + None + }; + let node = Rc::clone(&space.groups()[ptr].target); + profiles.insert( + *ptr, + ( + node, + RecurrenceProfile { + evaluation_rate, + one_shot_consumers, + update_rate: site_update_rate, + }, + ), + ); + } + + Ok(RecurrenceProfileMap { profiles }) +} + +/// Record `times` occurrences of `recurrence` against `ptr` — `times > 1` +/// when a single parent structurally references `ptr` more than once (see +/// [`recurrence_profiles`]'s own doc on edge multiplicity). +/// A no-op for `times == 0` (an `Rc` returned as a `direct_child_counts` +/// child always has `edge_count >= 1` in practice, but this keeps the +/// helper correct regardless). +fn contribute( + ptr: *const OperatorNode, + times: usize, + recurrence: RootRecurrence, + rates: &mut HashMap<*const OperatorNode, f64>, + one_shot_counts: &mut HashMap<*const OperatorNode, usize>, + reached: &mut HashSet<*const OperatorNode>, +) { + if times == 0 { + return; + } + reached.insert(ptr); + match recurrence { + RootRecurrence::Repeating(rate) => { + *rates.entry(ptr).or_insert(0.0) += rate.0 * times as f64; + } + RootRecurrence::OneShotCount(count) => { + *one_shot_counts.entry(ptr).or_insert(0) += count.saturating_mul(times); + } + RootRecurrence::Unknown => {} + } +} + +/// One [`TargetSubDAGCandidates`]'s candidates, ranked best-first by +/// [`cost_sorted`]. +#[derive(Debug)] +pub struct RankedTargetSubDAGCandidates<'a> { + pub target: &'a Rc, + pub consumer_count: usize, + pub candidates: Vec<&'a ReplacementSubDAG>, + /// `costs[i]` is `candidates[i]`'s own grouping-state cost when available, + /// and its [`CostModel::estimate_cost`] otherwise + /// estimate — aligned index-for-index with `candidates`, one number per + /// candidate, for a caller that wants an actual `f64` next to each + /// candidate (e.g. "candidate A costs ≈ X, candidate B costs ≈ Y") and + /// not just `candidates`' own relative order. `f64::NAN` throughout + /// unless `cost_model` overrides `estimate_cost` — see that method's own + /// doc. + pub costs: Vec, +} + +/// Rank `group`'s candidates best-first under `cost_model`, per the module +/// docs' "Cost-based final selection" section. Falls back to discovery +/// order whenever there's nothing to rank (0 or 1 candidates) or this +/// module doesn't have a defined `CostModel` comparison for the shape it +/// sees — it never invents one. +fn rank_group<'a>( + group: &'a TargetSubDAGCandidates, + cost_model: &dyn CostModel, +) -> Vec<&'a ReplacementSubDAG> { + let mut ranked: Vec<&ReplacementSubDAG> = group.candidates.iter().collect(); + if ranked.len() <= 1 { + return ranked; + } + + // Shape 1: the exact `SharedSubDAGStrategy` share-vs-recompute pair — + // rank via `CostModel::cse_share_decision`, the same comparison + // the local CSE ranking path already uses. + if cse_candidate_pair(group).is_some() { + if let Some(prefer_target) = cse_preference(group, cost_model) { + ranked.sort_by_key(|c| match c.provenance { + ReplacementProvenance::CseShare if prefer_target => 0, + ReplacementProvenance::CseRecompute if !prefer_target => 0, + ReplacementProvenance::CseShare | ReplacementProvenance::CseRecompute => 2, + _ => 1, + }); + } + return ranked; + } + + // Shape 2: independent and Hydra grouping alternatives for the same + // sketch algorithms. When deployment statistics provide a subpopulation + // estimate, compare N independent states with the shared grid directly. + let target = TargetSubDAG::with_consumer_count(&group.target, group.consumer_count); + let has_hydra = ranked.iter().any(|candidate| { + let Replacement::SubDAG(node) = &candidate.replacement else { + return false; + }; + summary_grouping(node).is_some_and(|grouping| { + matches!(grouping, GroupingStrategy::SharedMultiSubpopulation { .. }) + }) + }); + let grouping_costs: Option> = if has_hydra { + ranked + .iter() + .map(|candidate| { + cost_model + .grouping_state_cost(candidate, &target) + .map(|cost| cost.0) + }) + .collect() + } else { + None + }; + if let Some(costs) = grouping_costs { + let by_ptr: HashMap<*const ReplacementSubDAG, f64> = ranked + .iter() + .zip(costs) + .map(|(candidate, cost)| (*candidate as *const ReplacementSubDAG, cost)) + .collect(); + ranked.sort_by(|a, b| { + by_ptr[&(*a as *const ReplacementSubDAG)] + .total_cmp(&by_ptr[&(*b as *const ReplacementSubDAG)]) + }); + return ranked; + } + + // Shape 3: `ASAPStrategies`'s sketch-family candidates (every + // candidate is a `Summary` that realizes a `SketchAlgorithm`) — rank via + // `CostModel::rank_candidates`, the same hook `realizations_for_intent` + // itself consults. + if let Some(intent) = bindable_intent(&group.target) { + let kinds: Option> = ranked + .iter() + .map(|c| match &c.replacement { + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, + }) + .collect(); + if let Some(kinds) = kinds { + let order = + crate::cost::cost_model::validated_candidate_ranking(cost_model, intent, &kinds); + ranked.sort_by_key(|c| { + let kind = match &c.replacement { + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, + }; + kind.and_then(|k| order.iter().position(|o| *o == k)) + .unwrap_or(usize::MAX) + }); + return ranked; + } + } + + // A target may be handled by more than one strategy (for example, a + // shared aggregate has both bound-summary and share/recompute rewrite + // candidates). No shape-specific hook spans those different candidate + // types, so compare the numeric estimates the CostModel exposes for that + // purpose. `total_cmp` gives deterministic placement to a model's NaN + // placeholders without dropping any candidate. + ranked.sort_by(|a, b| { + match ( + cost_model.candidate_cost(a, &target), + cost_model.candidate_cost(b, &target), + ) { + (Some(a), Some(b)) => a.0.total_cmp(&b.0), + (Some(_), None) => std::cmp::Ordering::Less, + (None, Some(_)) => std::cmp::Ordering::Greater, + (None, None) => cost_model + .estimate_cost(a, &target) + .total_cmp(&cost_model.estimate_cost(b, &target)), + } + }); + ranked +} + +/// For a group whose candidates are all [`Replacement::Rewrite`] (the +/// [`SharedSubDAGStrategy`] shape): does [`CostModel::cse_share_decision`] +/// prefer the candidate that shares `group.target`'s own `Rc` (`true`), or +/// the one that recomputes independently (`false`)? `None` when there's no +/// real comparison to make — fewer than 2 consumers (mirrors +/// [`SharedSubDAGStrategy::matches`]'s own gate), or `group.target` can't +/// actually be bound at all (no candidate and no logical fallback — never +/// expected in practice for a target that's already part of a legitimate +/// workload DAG, but this degrades to "keep discovery order" rather than +/// panicking). +fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> Option { + if group.consumer_count < 2 { + return None; + } + let bound = realize_one(&group.target)?; + let candidate = CseCandidate { + sub_dag: &group.target, + bound_summary: &bound, + consumer_count: group.consumer_count, + }; + Some(match cost_model.cse_share_decision(&candidate) { + ShareDecision::Share => true, + ShareDecision::RecomputeIndependently => false, + }) +} + +/// [`cse_preference`] only needs one representative bound [`OperatorNode`] +/// for `target` (to build a [`CseCandidate`] for +/// [`CostModel::cse_share_decision`]), not the full ranked candidate list +/// [`ASAPStrategies::replacements`] returns — so this just reuses +/// [`realize_child`], the same rank-and-take-first helper +/// `construct_summary_agg`'s own recursion and +/// [`crate::cost::cost_model::DefaultCostModel::estimate_cost`] already use, +/// wrapped to swallow the (here, uninteresting) error into `None`. +fn realize_one(target: &Rc) -> Option> { + realize_child(target).ok() +} + +/// The `SketchAlgorithm` a bound [`Replacement::SubDAG`] candidate ultimately +/// realizes, if any (`None` for an `ExactAggregate`/pass-through +/// sub-DAG — nothing to rank against another `SketchAlgorithm`). +/// +/// Mirrors this module's own `#[cfg(test)]`-only `summary_family_algorithm` +/// helper (in the test module below), which does the identical +/// `SummaryEstimate`-unwrap-then-match for that module's own tests; that +/// copy is test-only, so this needs its own for real (non-test) ranking +/// code — the same "duplicate a small, self-contained traversal rather than +/// restructure a test helper" call this file's own top doc already makes +/// for `discover_targets`. +pub(crate) fn sketch_kind_of(node: &OperatorNode) -> Option { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + sketch_kind_of(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + }) => Some(kind.algorithm().clone()), + _ => None, + } +} + +/// The grouping strategy used by a bound summary candidate, unwrapping its +/// evaluation node when necessary. +fn summary_grouping(node: &OperatorNode) -> Option<&GroupingStrategy> { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + summary_grouping(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { grouping, .. }) => Some(grouping), + _ => None, + } +} + +// ── global_selection ───────────────────────────────────────────────────── + +/// Why [`global_selection`] committed an exact composition at a +/// site: which child candidate it composes with, and the +/// cost-units-per-second comparison against the raw fallback that it won. +#[derive(Debug)] +pub struct CompositionDecision<'a> { + /// The exact child/operation pair validated by the search accuracy model. + pub plan: Rc, + /// The child target the composed operator consumes. + pub child_target: &'a Rc, + /// For a read-time operation: the child's own candidate committed alongside + /// (the summary evaluation the operator folds). `None` for an update-path + /// transform, whose input is raw update data — its cost is charged to + /// the maintained summary *above* it instead. + pub child_candidate: Option<&'a ReplacementSubDAG>, + /// The composed plan's recurring rate — `read_operation_plan_cost_rate` + /// or `maintenance_operation_plan_cost_rate`. + pub cost_rate: CostRate, + /// `raw_recompute_cost_rate` — the kept-sub-DAG baseline it beat. + pub baseline_rate: CostRate, + /// The statistics (and their provenance) both rates were computed from. + pub inputs: ExactCompositionCostInputs, +} + +/// [`global_selection`]'s result: the [`GlobalSelection`] it committed to, +/// plus the cost comparison behind each exact composition it chose. The +/// selection is a Stage 1 type and carries no cost data, so the comparison +/// is kept here. +#[derive(Debug)] +pub struct CostedGlobalSelection<'a> { + selection: GlobalSelection<'a>, + compositions: HashMap<*const OperatorNode, CompositionDecision<'a>>, +} + +impl<'a> CostedGlobalSelection<'a> { + /// The decision behind `target`'s chosen exact composition, if it chose one. + pub fn composition(&self, target: &Rc) -> Option<&CompositionDecision<'a>> { + self.compositions.get(&Rc::as_ptr(target)) + } +} + +impl<'a> std::ops::Deref for CostedGlobalSelection<'a> { + type Target = GlobalSelection<'a>; + + fn deref(&self) -> &Self::Target { + &self.selection + } +} + +/// The maintained `SummaryAgg` a bound summary candidate builds (under +/// its `SummaryEstimate` evaluation, if any) — the summary an `ValueOperationAtIngestionTime` +/// beneath it feeds, for `maintenance_operation_plan_cost_rate`. +fn maintained_summary(node: &Rc) -> Option<&Rc> { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + maintained_summary(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { .. }) => Some(node), + _ => None, + } +} + +fn is_composition_candidate(candidate: &ReplacementSubDAG) -> bool { + matches!(candidate.replacement, Replacement::ExactComposition(_)) +} + +/// Everything [`global_selection`] threads between sites for +/// exact compositions (issue #171): child candidates already committed by +/// an earlier parent, and the maintained summary above each site. +#[derive(Default)] +struct CompositionContext { + /// child target ptr → the child's candidate an ancestor's composition + /// already committed to (a later parent must compose with the *same* + /// one, and the child's own selection is forced to it). + committed_child: HashMap<*const OperatorNode, *const ReplacementSubDAG>, + /// site ptr → the maintained `SummaryAgg` directly above it, when its + /// parent chose a bound summary — what an `ValueOperationAtIngestionTime` here feeds. + maintaining_parent: HashMap<*const OperatorNode, Rc>, +} + +/// One eligible composed alternative at a site, before the cheapest wins. +struct CompositionOption<'a> { + candidate: &'a ReplacementSubDAG, + decision: CompositionDecision<'a>, +} + +/// Every [`Replacement::ExactComposition`] candidate of `group` whose +/// composed-plan rate is *known* and beats the raw-recompute baseline — +/// costed against each compatible child candidate already in `CandidateLogicalASAPDAGs` +/// (or the one an earlier parent committed). Unknown statistics yield no +/// option at all: the conservative kept-sub-DAG path stays. +fn composition_options<'a>( + group: &'a TargetSubDAGCandidates, + groups: &'a HashMap<*const OperatorNode, TargetSubDAGCandidates>, + effective: usize, + cost_model: &dyn CostModel, + context: &CompositionContext, + plans: &[PreparedComposition], +) -> Vec> { + let mut options = Vec::new(); + for candidate in &group.candidates { + let Replacement::ExactComposition(composition) = &candidate.replacement else { + continue; + }; + if runtime_support_evidence(candidate, cost_model) != Some(true) { + continue; + } + let child_ptr = Rc::as_ptr(&composition.child_target); + let Some(child_group) = groups.get(&child_ptr) else { + continue; + }; + let already_committed = context.committed_child.get(&child_ptr).copied(); + let cost = |summary: &OperatorNode, shared: bool| { + let request = ExactCompositionCostRequest { + target: &group.target, + composition, + summary, + effective_consumer_count: effective, + }; + let mut inputs = cost_model.exact_composition_cost_inputs(&request); + if shared { + // Shared state is counted once: an earlier parent already + // pays this child's maintenance, so the marginal cost here + // is zero — a *known* zero, unlike an unknown input. + if let Some(maintenance) = inputs.summary_maintenance_cost_per_update.as_mut() { + *maintenance = 0.0; + } + } + let rate = inputs.composed_plan_cost_rate(composition.placement)?; + let baseline = raw_recompute_cost_rate(&inputs)?; + (rate < baseline).then_some((rate, baseline, inputs)) + }; + match composition.placement { + OperationPlacement::Read => { + let child_candidates: Vec<&'a ReplacementSubDAG> = match already_committed { + // SAFETY-free: the pointer was taken from `groups`'s own + // candidate storage, which outlives this borrow. + Some(ptr) => child_group + .candidates + .iter() + .filter(|c| std::ptr::eq(*c, ptr)) + .collect(), + None => child_group.candidates.iter().collect(), + }; + for child_candidate in child_candidates { + if !is_automatically_selectable(child_candidate, cost_model) { + continue; + } + let Replacement::SubDAG(summary) = &child_candidate.replacement else { + continue; + }; + if is_logical_rewrite(summary) || !composition.accepts_child(summary) { + continue; + } + let Some(prepared) = plans.iter().find(|p| { + p.target == Rc::as_ptr(&group.target) + && p.operation.same_as(composition) + && Rc::ptr_eq(&p.child, summary) + }) else { + continue; + }; + let Some((rate, baseline, inputs)) = cost(summary, already_committed.is_some()) + else { + continue; + }; + options.push(CompositionOption { + candidate, + decision: CompositionDecision { + plan: Rc::clone(&prepared.plan), + child_target: &composition.child_target, + child_candidate: Some(child_candidate), + cost_rate: rate, + baseline_rate: baseline, + inputs, + }, + }); + } + } + OperationPlacement::Maintenance => { + let Some(prepared) = plans.iter().find(|p| { + p.target == Rc::as_ptr(&group.target) && p.operation.same_as(composition) + }) else { + continue; + }; + // An maintenance-time operation only pays off beneath a + // maintained summary; with nothing above it, its output is + // never read and the raw fallback is the same computation. + let Some(parent) = context.maintaining_parent.get(&Rc::as_ptr(&group.target)) + else { + continue; + }; + let Some((rate, baseline, inputs)) = cost(parent, false) else { + continue; + }; + options.push(CompositionOption { + candidate, + decision: CompositionDecision { + plan: Rc::clone(&prepared.plan), + child_target: &composition.child_target, + child_candidate: None, + cost_rate: rate, + baseline_rate: baseline, + inputs, + }, + }); + } + } + } + options +} + +/// The whole-plan (cross-group) selection step the module docs' +/// "Whole-plan (cross-group) selection" section describes: one +/// [`TargetSubDAGSelection`] per discovered site, each ranked against an +/// `effective_consumer_count` that accounts for every ancestor +/// [`SharedSubDAGStrategy`] decision on the path to it — unlike +/// [`cost_sorted`], whose per-group ranking only ever sees a +/// group's own raw [`TargetSubDAGCandidates::consumer_count`]. +/// Uncertified DDSketch ratios remain in [`CandidateLogicalASAPDAGs`] for downstream +/// inspection but are not chosen automatically by this selector. +pub fn global_selection<'a, Id>( + space: &'a CandidateLogicalASAPDAGs, + cost_model: &dyn CostModel, +) -> CostedGlobalSelection<'a> { + global_selection_impl(space, cost_model, None, None) + .expect("structural global selection cannot produce a recurrence error") +} + +/// Recurrence-aware counterpart to [`global_selection`]. The same +/// whole-plan traversal and effective structural consumer counts are +/// retained, while every CSE share/recompute choice is made from the +/// corresponding recurrence profile. +pub fn global_selection_with_recurrence<'a, Id>( + space: &'a CandidateLogicalASAPDAGs, + cost_model: &dyn CostModel, + profiles: &RecurrenceProfileMap, + horizon: Option, +) -> Result, RecurrenceError> { + global_selection_impl(space, cost_model, Some(profiles), horizon) +} + +fn global_selection_impl<'a, Id>( + space: &'a CandidateLogicalASAPDAGs, + cost_model: &dyn CostModel, + profiles: Option<&RecurrenceProfileMap>, + horizon: Option, +) -> Result, RecurrenceError> { + let dag = reference_dag(space); + let topo = topological_order(space.order(), &dag); + + let mut effective_uses = dag.external_root_uses.clone(); + let mut chosen_share: HashMap<*const OperatorNode, ShareDecision> = HashMap::new(); + let mut groups: HashMap<*const OperatorNode, TargetSubDAGSelection<'a>> = HashMap::new(); + let mut compositions: HashMap<*const OperatorNode, CompositionDecision<'a>> = HashMap::new(); + let mut context = CompositionContext::default(); + + for ptr in &topo { + let group = &space.groups()[ptr]; + + let effective = effective_uses.get(ptr).copied().unwrap_or(0); + effective_uses.insert(*ptr, effective); + + // ── Exact compositions (issue #171) ───────────────────────── + // A child an earlier parent's composition committed to is + // forced to exactly that candidate — the parent/child pair is + // one decision. Otherwise, a composition here wins only when + // its cost-units-per-second rate is *known* and beats the raw + // recompute baseline; missing statistics keep the conservative + // path below. + let mut composition_decision = None; + let forced = context + .committed_child + .get(ptr) + .and_then(|&cptr| group.candidates.iter().find(|c| std::ptr::eq(*c, cptr))); + let composed = if forced.is_some() { + None + } else { + composition_options( + group, + space.groups(), + effective, + cost_model, + &context, + space.composition_plans(), + ) + .into_iter() + .min_by(|a, b| a.decision.cost_rate.0.total_cmp(&b.decision.cost_rate.0)) + }; + if let Some(option) = &composed { + if let Some(child_candidate) = option.decision.child_candidate { + context.committed_child.insert( + Rc::as_ptr(option.decision.child_target), + child_candidate as *const ReplacementSubDAG, + ); + } + if let Replacement::ExactComposition(composition) = &option.candidate.replacement { + if composition.placement == OperationPlacement::Maintenance { + // A chain of functions feeds the same summary. + if let Some(parent) = context.maintaining_parent.get(ptr).cloned() { + context + .maintaining_parent + .insert(Rc::as_ptr(&composition.child_target), parent); + } + } + } + } + + let complete_plan_choice = (!forced.is_some() + && composed.is_none() + && cost_model.candidate_cost_covers_complete_plan()) + .then(|| { + let effective_target = TargetSubDAG::with_consumer_count(&group.target, effective); + let bound = group + .candidates + .iter() + .filter(|candidate| { + !is_cse_candidate(candidate) + && !is_composition_candidate(candidate) + && is_automatically_selectable(candidate, cost_model) + }) + .filter_map(|candidate| { + cost_model + .candidate_cost(candidate, &effective_target) + .map(|cost| (candidate, cost)) + }) + .min_by(|(_, left), (_, right)| left.0.total_cmp(&right.0)) + .map(|(candidate, _)| candidate); + bound.or_else(|| { + (cost_model.allow_uncosted_legacy_selection() && effective >= 2) + .then(|| { + decide_with_effective_count(group, effective, cost_model).and_then( + |decision| { + let candidate = pick_shared_sub_dag_candidate(group, decision)?; + chosen_share.insert(*ptr, decision); + Some(candidate) + }, + ) + }) + .flatten() + }) + }) + .flatten(); + + let chosen = if let Some(forced) = forced { + Some(forced) + } else if let Some(option) = composed { + composition_decision = Some(option.decision); + Some(option.candidate) + } else if cost_model.candidate_cost_covers_complete_plan() { + complete_plan_choice + } else if effective >= 2 && cse_candidate_pair(group).is_some() { + let decision = if let Some(profiles) = profiles { + decide_group_with_recurrence( + group, + effective, + profiles.for_target(&group.target), + horizon, + cost_model, + )? + } else { + decide_with_effective_count(group, effective, cost_model) + }; + match decision { + Some(decision) => { + let cse = pick_shared_sub_dag_candidate(group, decision); + let effective_target = + TargetSubDAG::with_consumer_count(&group.target, effective); + let logical = group + .candidates + .iter() + .filter(|candidate| { + !is_cse_candidate(candidate) + && !is_composition_candidate(candidate) + && is_automatically_selectable(candidate, cost_model) + }) + .filter_map(|candidate| { + cost_model + .candidate_cost(candidate, &effective_target) + .map(|cost| (candidate, cost)) + }) + .min_by(|(_, a), (_, b)| a.0.total_cmp(&b.0)) + .map(|(candidate, _)| candidate); + let cse = cse.filter(|candidate| { + cost_model + .candidate_cost(candidate, &effective_target) + .is_some() + || cost_model.allow_uncosted_legacy_selection() + }); + match (cse, logical) { + (Some(cse), Some(logical)) + if cost_model + .candidate_cost(cse, &effective_target) + .is_none_or(|cse_cost| { + cost_model + .candidate_cost(logical, &effective_target) + .is_some_and(|logical_cost| logical_cost.0 < cse_cost.0) + }) => + { + Some(logical) + } + (cse, _) => { + if cse.is_some() { + chosen_share.insert(*ptr, decision); + } + cse + } + } + } + // `realize_child` couldn't produce even a logical fallback — + // not expected in practice for a target that's already + // part of a legitimate workload DAG (mirrors + // `cse_preference`'s own doc on this same degrade). + // Falling back to ordinary local ranking is still a + // valid answer, just not a cross-group-aware one; this + // group also contributes no Share collapse to its own + // children (see `multiplier`'s `_ => effective` arm). + None => rank_group(group, cost_model).into_iter().find(|candidate| { + !is_composition_candidate(candidate) + && is_automatically_selectable(candidate, cost_model) + && (cost_model + .candidate_cost( + candidate, + &TargetSubDAG::with_consumer_count(&group.target, effective), + ) + .is_some() + || cost_model.allow_uncosted_legacy_selection()) + }), + } + } else { + let effective_target = TargetSubDAG::with_consumer_count(&group.target, effective); + rank_group(group, cost_model) + .into_iter() + .find(|candidate| { + !is_cse_candidate(candidate) + && !is_composition_candidate(candidate) + && is_automatically_selectable(candidate, cost_model) + && (cost_model + .candidate_cost(candidate, &effective_target) + .is_some() + || cost_model.allow_uncosted_legacy_selection()) + }) + .or_else(|| { + cse_candidate_pair(group) + .map(|(share, _)| share) + .filter(|candidate| { + cost_model + .candidate_cost(candidate, &effective_target) + .is_some() + || cost_model.allow_uncosted_legacy_selection() + }) + }) + }; + + // Record the maintained summary this site's bound candidate + // builds, for a child that may compose an `ValueOperationAtIngestionTime` + // beneath it. + if let (Some(Replacement::SubDAG(node)), Some(NonASAPOp::Aggregate { child, .. })) = + (chosen.map(|c| &c.replacement), group.target.non_asap()) + { + if let Some(summary) = maintained_summary(node) { + context + .maintaining_parent + .insert(Rc::as_ptr(child), Rc::clone(summary)); + } + } + + let outgoing_multiplier = multiplier(*ptr, &effective_uses, &chosen_share); + match chosen { + Some(ReplacementSubDAG { + replacement: Replacement::SubDAG(source), + provenance: ReplacementProvenance::AccuracyReconciliation, + .. + }) => { + // Accuracy reconciliation reads another discovered memo + // group, rather than inlining that group's children. Let + // the source group receive the uses and propagate them + // through its own selected realization when its turn + // arrives in topological order. + *effective_uses.entry(Rc::as_ptr(source)).or_insert(0) += outgoing_multiplier; + } + _ => { + let selected_rewrite = match chosen.map(|candidate| &candidate.replacement) { + Some(Replacement::SubDAG(rewrite)) if is_logical_rewrite(rewrite) => rewrite, + Some(Replacement::SubDAG(_) | Replacement::ExactComposition(_)) | None => { + &group.target + } + }; + for (child, edge_count) in direct_child_counts(selected_rewrite) { + *effective_uses.entry(child).or_insert(0) += edge_count * outgoing_multiplier; + } + } + } + + groups.insert( + *ptr, + TargetSubDAGSelection { + target: &group.target, + consumer_count: group.consumer_count, + effective_consumer_count: effective, + chosen, + }, + ); + if let Some(decision) = composition_decision { + compositions.insert(*ptr, decision); + } + } + + let composition_plans = compositions + .iter() + .map(|(ptr, decision)| (*ptr, Rc::clone(&decision.plan))) + .collect(); + Ok(CostedGlobalSelection { + selection: GlobalSelection::new(space.order().to_vec(), groups, composition_plans), + compositions, + }) +} + +fn is_cse_candidate(candidate: &ReplacementSubDAG) -> bool { + matches!( + candidate.provenance, + ReplacementProvenance::CseShare | ReplacementProvenance::CseRecompute + ) +} + +fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn CostModel) -> bool { + candidate.provenance != ReplacementProvenance::RootPhysicalRealization + && !candidate.has_missing_accuracy_evidence() + && runtime_support_evidence(candidate, cost_model) != Some(false) +} + +/// How much one direct reference to `parent_ptr` actually costs, once +/// `parent_ptr`'s own chosen candidate (if it has a Share/Recompute pair at +/// all) is taken into account: +/// +/// - `1`, if `parent_ptr` chose [`ShareDecision::Share`] — one shared +/// execution backs every reference to it, so referencing it costs no more +/// than referencing it once. +/// - `parent_ptr`'s own `effective_consumer_count` otherwise — either it +/// chose [`ShareDecision::RecomputeIndependently`] (each of its own uses +/// gets its own independent execution, so referencing it costs as much as +/// its *own* full multiplicity), or it has no Share/Recompute decision at +/// all (not a [`SharedSubDAGStrategy`] shape — nothing here collapses +/// its multiplicity to one, so whatever multiplicity *its* ancestors +/// established simply passes through). +/// +/// Composing this recurrence transitively up the whole ancestor chain (not +/// just the immediate parent) is exactly what makes +/// [`global_selection`]'s `effective_consumer_count` differ from +/// [`TargetSubDAGCandidates::consumer_count`] whenever a `RecomputeIndependently` +/// ancestor sits anywhere on the path from a root to a site — see the +/// module docs' "Whole-plan (cross-group) selection" section. +fn multiplier( + parent_ptr: *const OperatorNode, + effective_uses: &HashMap<*const OperatorNode, usize>, + chosen_share: &HashMap<*const OperatorNode, ShareDecision>, +) -> usize { + let effective = *effective_uses.get(&parent_ptr).expect( + "topological_order guarantees a parent is processed (and its effective_consumer_count \ + recorded) before any of its children", + ); + match chosen_share.get(&parent_ptr) { + Some(ShareDecision::Share) => 1, + _ => effective, + } +} + +/// [`CostModel::cse_share_decision`] for `group`, against an explicit +/// `effective_consumer_count` instead of `group.consumer_count` — the +/// cross-group-aware counterpart to [`cse_preference`], which uses the raw +/// structural count. `None` only when [`realize_child`] can't produce even a +/// logical fallback for `group.target` (see that function's own doc). +fn decide_with_effective_count( + group: &TargetSubDAGCandidates, + effective_consumer_count: usize, + cost_model: &dyn CostModel, +) -> Option { + let bound = realize_child(&group.target).ok()?; + let candidate = CseCandidate { + sub_dag: &group.target, + bound_summary: &bound, + consumer_count: effective_consumer_count, + }; + Some(cost_model.cse_share_decision(&candidate)) +} + +fn decide_group_with_recurrence( + group: &TargetSubDAGCandidates, + effective_consumer_count: usize, + recurrence: RecurrenceProfile, + horizon: Option, + cost_model: &dyn CostModel, +) -> Result, RecurrenceError> { + let Some(bound) = realize_child(&group.target).ok() else { + return Ok(None); + }; + let candidate = CseCandidate { + sub_dag: &group.target, + bound_summary: &bound, + consumer_count: effective_consumer_count, + }; + Ok(Some( + cost_model + .cse_share_decision_with_recurrence(&candidate, &recurrence, horizon)? + .decision, + )) +} + +/// The [`SharedSubDAGStrategy`] candidate matching `decision`: the one +/// that shares `group.target`'s own `Rc` for [`ShareDecision::Share`], the +/// freshly-allocated one for [`ShareDecision::RecomputeIndependently`] — +/// the same `Rc`-identity distinction [`is_duplicate_rewrite`]'s own doc +/// explains is the *only* signal this IR carries for that choice. +fn pick_shared_sub_dag_candidate( + group: &TargetSubDAGCandidates, + decision: ShareDecision, +) -> Option<&ReplacementSubDAG> { + let (share, recompute) = cse_candidate_pair(group)?; + Some(match decision { + ShareDecision::Share => share, + ShareDecision::RecomputeIndependently => recompute, + }) +} + +// ── reference DAG + topological order ───────────────────────────────── + +/// The parent/child structure [`global_selection`]'s DP walks — +/// built separately from `discover_targets`'s own `order`/`nodes`/`counts` +/// maps (which only track *aggregate* reference counts, not per-parent +/// breakdown or direction). Selection needs per-parent edge counts to +/// distinguish shared producers from repeated uses within one consumer. +struct ReferenceDAG { + /// child ptr -> `(parent ptr, edge count from that one parent)`, for + /// every direct operator-child edge in the relational-skeleton scope + /// [`walk_children`] itself uses (an edge count above 1 happens when + /// one parent references the same child from two different fields, + /// e.g. a `Join`'s `left`/`right` both being the same `Rc`). + parents_of: HashMap<*const OperatorNode, Vec<(*const OperatorNode, usize)>>, + /// parent ptr -> every distinct child ptr it directly references — the + /// reverse of `parents_of`, for [`topological_order`]'s Kahn's-algorithm + /// traversal. + children_of: HashMap<*const OperatorNode, Vec<*const OperatorNode>>, + /// How many of the workload's own `roots` point directly at each node — + /// a node's "external" use. Nothing inside the DAG decides this (it + /// isn't a reference from another discovered site), so it's never + /// subject to any ancestor's Share/Recompute choice — it's the base + /// case [`global_selection`]'s recurrence starts from. + external_root_uses: HashMap<*const OperatorNode, usize>, +} + +/// Build an ordering DAG containing every edge that could be selected: +/// the original target's edges plus every rewrite candidate's edges. An +/// accuracy-reconciliation rewrite points at another discovered memo group, +/// so it contributes an edge to that group itself; other rewrites contribute +/// their relational children as before. The +/// DAG is deliberately only used for topological ordering; effective-use +/// counts are propagated through the one candidate actually selected. +fn reference_dag(space: &CandidateLogicalASAPDAGs) -> ReferenceDAG { + let mut dag = ReferenceDAG { + parents_of: HashMap::new(), + children_of: HashMap::new(), + external_root_uses: HashMap::new(), + }; + for (_, root) in &space.roots { + *dag.external_root_uses.entry(Rc::as_ptr(root)).or_insert(0) += 1; + } + for ptr in space.order() { + let group = &space.groups()[ptr]; + record_possible_edges(*ptr, &group.target, &mut dag); + for candidate in &group.candidates { + if let Replacement::SubDAG(rewrite) = &candidate.replacement { + if !is_logical_rewrite(rewrite) { + continue; + } + if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { + add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut dag); + } else { + record_possible_edges(*ptr, rewrite, &mut dag); + } + } + } + } + dag +} + +/// Record one `parent_ptr -> child` edge (both directions — see +/// [`ReferenceDAG`]'s fields), retaining the greatest multiplicity seen +/// when the target and alternative rewrites expose the same edge. +fn add_edge( + parent_ptr: *const OperatorNode, + child_ptr: *const OperatorNode, + edge_count: usize, + dag: &mut ReferenceDAG, +) { + let siblings = dag.parents_of.entry(child_ptr).or_default(); + match siblings.iter_mut().find(|(p, _)| *p == parent_ptr) { + Some((_, count)) => *count = (*count).max(edge_count), + None => siblings.push((parent_ptr, edge_count)), + } + let kids = dag.children_of.entry(parent_ptr).or_default(); + if !kids.contains(&child_ptr) { + kids.push(child_ptr); + } +} + +fn record_possible_edges( + parent_ptr: *const OperatorNode, + node: &OperatorNode, + dag: &mut ReferenceDAG, +) { + for (child_ptr, edge_count) in direct_child_counts(node) { + add_edge(parent_ptr, child_ptr, edge_count, dag); + } +} + +/// A topological order over `order` (parent before every child) via Kahn's +/// algorithm on `dag`'s reverse adjacency — needed because +/// `discover_targets`'s own `order` is only a valid *discovery* order +/// (first-seen-first), not a valid topological one: a node reached via two +/// different root paths can have a parent that's discovered *after* it (see +/// this function's own test for a worked diamond example), which is exactly +/// backwards for [`global_selection`]'s recurrence. +fn topological_order( + order: &[*const OperatorNode], + dag: &ReferenceDAG, +) -> Vec<*const OperatorNode> { + let mut in_degree: HashMap<*const OperatorNode, usize> = HashMap::new(); + for ptr in order { + let degree = dag.parents_of.get(ptr).map(Vec::len).unwrap_or(0); + in_degree.insert(*ptr, degree); + } + + let mut queue: VecDeque<*const OperatorNode> = order + .iter() + .copied() + .filter(|ptr| in_degree[ptr] == 0) + .collect(); + + let mut topo = Vec::with_capacity(order.len()); + while let Some(ptr) = queue.pop_front() { + topo.push(ptr); + if let Some(children) = dag.children_of.get(&ptr) { + for child in children { + if let Some(degree) = in_degree.get_mut(child) { + *degree -= 1; + if *degree == 0 { + queue.push_back(*child); + } + } + } + } + } + + assert_eq!( + topo.len(), + order.len(), + "topological_order: the discovered-site reference dag has a cycle — every \ + OperatorNode is built from Rc children, which can't form one, so this indicates a bug \ + in reference_dag rather than a real cyclic workload", + ); + topo +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cost::cost_model::{Cost, DefaultCostModel}; + use crate::test_support::{agg, lower_promql, metric_scan}; + use asap_logical_optimizer::accuracy::{ + AccuracyModel, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, + }; + use asap_logical_optimizer::pass1::replacement::{ + default_strategies, search_workload, search_workload_with, search_workload_with_targets, + ASAPStrategies, ReplacementStrategy, + }; + use asap_logical_optimizer::pass2::reconciliation::AccuracyReconciliationStrategy; + use asap_types::ir::operator::agg_intent::{default_cardinality, default_quantile, AggIntent}; + use asap_types::ir::operator::operator_properties::Reduction; + use asap_types::ir::properties::{ + AccuracyError, CompositionOperator, ErrorMetric, ResultGuarantee, + }; + use asap_types::ir::schema::ColumnId; + use asap_types::ir::schema::SketchStatistic; + use asap_types::ir::{Predicate, ScalarExpr}; + use asap_types::types::AccuracyTarget; + + fn equi_pred(left: ColumnId, right: ColumnId) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(left)), + op: asap_types::ir::scalar::CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(right)), + semantics: asap_types::ir::ExprSemantics::Sql, + }) + } + + fn quantile_eps_intent(q: f64, e: f64) -> AggIntent { + AggIntent::Quantile { + col: None, + q, + accuracy: AccuracyTarget::Epsilon(e), + } + } + + #[test] + fn relational_join_is_exact_only_when_both_inputs_are_exact() { + // `relational_join_guarantee` folded into assembly's generic + // "keep the operator, assemble its children" branch: an assembled + // inner equi-`Join` is exact exactly when both assembled inputs are. + let join = |left_intent: AggIntent, right_intent: AggIntent| { + let left = agg(vec![2], left_intent, metric_scan(&["job"])); + let right = agg( + vec![2], + right_intent, + crate::test_support::scan("n", metric_scan(&["job"]).schema.clone()), + ); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: asap_types::ir::operator::operator_properties::JoinKind::Inner, + pred: equi_pred(0, 2), + left, + right, + })) + .unwrap() + }; + let is_exact = |node: &OperatorNode| { + node.guarantee + .as_ref() + .is_some_and(ResultGuarantee::is_exact) + }; + for (root, both_exact_expected) in [ + ( + join(AggIntent::Sum { col: None }, AggIntent::Sum { col: None }), + true, + ), + ( + join(AggIntent::Sum { col: None }, quantile_eps_intent(0.5, 0.05)), + false, + ), + ] { + let space = search_workload(vec![(0usize, Rc::clone(&root))]); + let assembled = global_selection(&space, &DefaultCostModel) + .assemble_selected_query(&space.roots[0].1) + .unwrap() + .unwrap(); + let Some(NonASAPOp::Join { left, right, .. }) = assembled.non_asap() else { + panic!("the join is kept and its inputs assembled: {assembled:?}"); + }; + assert_eq!( + is_exact(&assembled), + is_exact(left) && is_exact(right), + "join guarantee must be exact iff both inputs are exact" + ); + if both_exact_expected { + assert!(is_exact(&assembled), "exact inputs give an exact join"); + } + } + } + + // ── cost-based ranking ─────────────────────────────────────────────── + + #[test] + fn cost_sorted_orders_shared_sub_dag_candidates_by_cse_share_decision() { + // Many consumers of a cheap-to-recompute, cheap-to-maintain exact + // accumulator: cse_share_decision should prefer Share (see + // cost_model.rs's own `cse_share_decision_shares_when_recompute_dominates_maintenance`). + let mut roots = Vec::new(); + let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + for i in 0..20 { + roots.push((i, Rc::new((*shared).clone()))); + } + let space = search_workload(roots); + let group = space.candidates_for_target(&space.roots[0].1).unwrap(); + assert_eq!(group.consumer_count, 20); + + let ranked = cost_sorted(&space, &DefaultCostModel); + let ranked_group = ranked + .iter() + .find(|g| Rc::ptr_eq(g.target, &space.roots[0].1)) + .unwrap(); + assert!(matches!( + &ranked_group.candidates[0].replacement, + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, &group.target) + )); + let rewrites: Vec<&ReplacementSubDAG> = ranked_group + .candidates + .iter() + .filter(|c| matches!(&c.replacement, Replacement::SubDAG(n) if !n.contains_asap())) + .copied() + .collect(); + assert_eq!(rewrites.len(), 2); + let first_shares_target = match &rewrites[0].replacement { + Replacement::SubDAG(rc) => Rc::ptr_eq(rc, &group.target), + Replacement::ExactComposition(_) => false, + }; + assert!( + first_shares_target, + "with 20 cheap consumers, Share should rank first: {rewrites:?}" + ); + } + + #[test] + fn cost_sorted_orders_sketch_candidates_by_rank_candidates() { + struct PreferDDSketch; + impl CostModel for PreferDDSketch { + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + let mut v = candidates.to_vec(); + if let Some(pos) = v.iter().position(|k| *k == SketchAlgorithm::DDSketch) { + let dd = v.remove(pos); + v.insert(0, dd); + } + v + } + } + + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let space = search_workload(vec![("q", root)]); + let ranked = cost_sorted(&space, &PreferDDSketch); + let agg_group = ranked + .iter() + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) + .unwrap(); + assert_eq!(agg_group.candidates.len(), 2); + let first_kind = match &agg_group.candidates[0].replacement { + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, + }; + assert_eq!(first_kind, Some(SketchAlgorithm::DDSketch)); + } + + #[test] + fn grouping_cost_cannot_resurrect_unprovable_hydra_candidates() { + struct EstimatedSubpopulations(usize); + + impl CostModel for EstimatedSubpopulations { + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + + fn estimated_subpopulation_count(&self, _target: &OperatorNode) -> Option { + Some(self.0) + } + } + + fn first_grouping(estimated_count: usize) -> GroupingStrategy { + let model = EstimatedSubpopulations(estimated_count); + let intent = AggIntent::Count { + accuracy: AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + }; + let root = agg(vec![2, 3], intent, metric_scan(&["tenant_id", "endpoint"])); + let space = search_workload(vec![("tenant_endpoint_count", root)]); + let ranked = cost_sorted(&space, &model); + let aggregate = ranked + .iter() + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) + .expect("aggregate group"); + let Replacement::SubDAG(node) = &aggregate.candidates[0].replacement else { + panic!("grouping candidate must be a summary") + }; + summary_grouping(node) + .expect("bound summary grouping") + .clone() + } + + assert_eq!( + first_grouping(10_000), + GroupingStrategy::PerSubpopulationInstance + ); + assert_eq!( + first_grouping(10), + GroupingStrategy::PerSubpopulationInstance + ); + } + + /// [`RankedTargetSubDAGCandidates::costs`] is a per-candidate annotation, aligned + /// index-for-index with `candidates` — each entry must equal what + /// calling [`CostModel::estimate_cost`] directly on that same candidate + /// and target produces, not some other (or stale) number. + #[test] + fn cost_sorted_pairs_each_candidate_with_its_own_estimate_cost() { + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let space = search_workload(vec![("q", root)]); + let ranked = cost_sorted(&space, &DefaultCostModel); + let agg_group = ranked + .iter() + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) + .unwrap(); + assert_eq!( + agg_group.costs.len(), + agg_group.candidates.len(), + "costs must be aligned 1:1 with candidates" + ); + assert!(!agg_group.costs.is_empty()); + + let target = TargetSubDAG::with_consumer_count(agg_group.target, agg_group.consumer_count); + for (candidate, &cost) in agg_group.candidates.iter().zip(&agg_group.costs) { + assert_eq!( + cost, + DefaultCostModel.estimate_cost(candidate, &target), + "RankedTargetSubDAGCandidates::costs must match calling CostModel::estimate_cost directly \ + for the same candidate/target" + ); + } + } + + // ── global_selection (issue #271) ─────────────────────────────────── + + /// A `CostModel` with a constant, `sub-DAG`-independent recompute cost + /// and shared-maintenance cost, chosen (40 recompute-per-use, 100 + /// maintenance) so that a `SharedSubDAGStrategy` group's + /// `cse_share_decision` flips exactly between a consumer count of 2 + /// (recompute total 80, below maintenance: `RecomputeIndependently`) + /// and a consumer count of 3 (recompute total 120, above + /// maintenance: `Share`) — the precise threshold + /// `effective_consumer_count_corrects_a_nested_groups_share_decision` + /// needs to cross. + struct ConstantCseCost; + impl CostModel for ConstantCseCost { + fn allow_uncosted_legacy_selection(&self) -> bool { + true + } + + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn cse_recompute_cost(&self, _candidate: &CseCandidate) -> Cost { + Cost(40.0) + } + fn cse_shared_maintenance_cost(&self, _candidate: &CseCandidate) -> Cost { + Cost(100.0) + } + } + + /// A costed logical choice must not panic when an explicitly allowed CSE + /// choice has no numeric cost. + #[test] + fn costed_logical_candidate_beats_uncosted_legacy_cse_choice() { + struct MixedCost; + impl CostModel for MixedCost { + fn allow_uncosted_legacy_selection(&self) -> bool { + true + } + + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + + fn candidate_cost( + &self, + candidate: &ReplacementSubDAG, + _target: &TargetSubDAG<'_>, + ) -> Option { + (!is_cse_candidate(candidate)).then_some(Cost(1.0)) + } + } + + let aggregate = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let space = search_workload(vec![("left", Rc::clone(&aggregate)), ("right", aggregate)]); + let root = &space.roots[0].1; + assert!(cse_candidate_pair(space.candidates_for_target(root).unwrap()).is_some()); + let selected = global_selection(&space, &MixedCost); + let chosen = selected.for_target(root).unwrap().chosen.unwrap(); + assert!(!is_cse_candidate(chosen)); + } + + #[test] + fn global_selection_matches_cost_sorted_for_a_non_interacting_workload() { + // No nested sharing at all — global_selection's effective_consumer_count + // must equal the group's own raw consumer_count, and its `chosen` + // candidate must be cost_sorted's top pick, for both the sketch + // group and its child Scan. + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let space = search_workload(vec![("q", root)]); + + let ranked = cost_sorted(&space, &DefaultCostModel); + let selected = global_selection(&space, &DefaultCostModel); + assert_eq!(ranked.len(), selected.target_selections().count()); + + for ranked_group in &ranked { + let selected_group = selected.for_target(ranked_group.target).unwrap(); + assert_eq!( + selected_group.effective_consumer_count, ranked_group.consumer_count, + "no ancestor is ever RecomputeIndependently here, so effective must equal raw" + ); + assert_eq!( + selected_group.chosen.map(|c| &c.rationale), + ranked_group.candidates.first().map(|c| &c.rationale), + "with no cross-group interaction, global_selection's pick must match \ + cost_sorted's top-ranked candidate" + ); + } + } + + #[test] + fn global_selection_leaves_an_unmatched_group_as_none() { + // A bare Scan: no registered strategy has an opinion on it, so it + // gets a group with an empty candidate list (see TargetSubDAGCandidates's own + // doc) — global_selection must not invent a candidate for it. + let root = metric_scan(&["job"]); + let space = search_workload(vec![("q", root)]); + let selected = global_selection(&space, &DefaultCostModel); + let scan_group = selected + .target_selections() + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Scan { .. }))) + .unwrap(); + assert!(scan_group.chosen.is_none()); + assert_eq!(scan_group.effective_consumer_count, 1); + } + + #[test] + fn global_selection_falls_back_to_local_ranking_for_sketch_family_groups() { + // ASAPStrategies groups have no cross-group-aware cost hook + // (rank_candidates takes no consumer_count) — global_selection must + // still return cost_sorted's own top pick for them (documented in + // the module docs' "Whole-plan (cross-group) selection" section), + // not silently drop the candidate or fall back to discovery order. + struct PreferDDSketch; + impl CostModel for PreferDDSketch { + fn allow_uncosted_legacy_selection(&self) -> bool { + true + } + + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + let mut v = candidates.to_vec(); + if let Some(pos) = v.iter().position(|k| *k == SketchAlgorithm::DDSketch) { + let dd = v.remove(pos); + v.insert(0, dd); + } + v + } + } + + let root = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let space = search_workload(vec![("q", root)]); + let selected = global_selection(&space, &PreferDDSketch); + let agg_group = selected + .target_selections() + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) + .unwrap(); + let kind = match &agg_group.chosen.unwrap().replacement { + Replacement::SubDAG(node) => sketch_kind_of(node), + Replacement::ExactComposition(_) => None, + }; + assert_eq!(kind, Some(SketchAlgorithm::DDSketch)); + + struct Uncosted; + impl CostModel for Uncosted { + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + } + assert!(global_selection(&space, &Uncosted) + .for_target(&space.roots[0].1) + .unwrap() + .chosen + .is_none()); + } + + #[test] + fn mixed_rewrite_group_keeps_and_selects_its_explicit_cse_pair() { + let target = metric_scan(&["job"]); + let group = TargetSubDAGCandidates { + target: Rc::clone(&target), + consumer_count: 2, + rejected: Vec::new(), + candidates: vec![ + ReplacementSubDAG { + strategy: "TestStrategy", + replacement: Replacement::SubDAG(Rc::clone(&target)), + provenance: ReplacementProvenance::CseShare, + rationale: "share".into(), + }, + ReplacementSubDAG { + strategy: "TestStrategy", + replacement: Replacement::SubDAG(Rc::new(target.as_ref().clone())), + provenance: ReplacementProvenance::CseRecompute, + rationale: "recompute".into(), + }, + ReplacementSubDAG { + strategy: "TestStrategy", + replacement: Replacement::SubDAG( + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlVectorFromScalar(ScalarExpr::EvalTimestamp), + )) + .unwrap(), + ), + provenance: ReplacementProvenance::LogicalRewrite, + rationale: "different rewrite strategy".into(), + }, + ], + }; + + assert!(cse_candidate_pair(&group).is_some()); + let ranked = rank_group(&group, &ConstantCseCost); + assert_eq!( + ranked + .iter() + .map(|c| c.rationale.as_str()) + .collect::>(), + vec!["recompute", "different rewrite strategy", "share"], + "the preferred CSE choice must be ranked without losing the unrelated rewrite" + ); + let chosen = pick_shared_sub_dag_candidate( + &group, + decide_with_effective_count(&group, 2, &ConstantCseCost).unwrap(), + ) + .unwrap(); + assert_eq!(chosen.provenance, ReplacementProvenance::CseRecompute); + } + + #[test] + fn effective_consumer_count_corrects_a_nested_groups_share_decision() { + // The interaction issue #271 describes: an outer shared sub-DAG `a` + // (referenced by 2 roots, so consumer_count == 2) wraps an inner + // shared sub-DAG `c` (referenced once through `a`'s own child edge, + // plus once more directly by a third, separate root — so `c`'s own + // *raw* structural consumer_count is also 2, independent of `a`). + // + // root1 ─┐ + // ├─▶ a = Filter(child = c) ─▶ c = Dedup(job) + // root2 ─┘ + // root3 ───────────────────────────▶ c (same shared Rc) + // + // `a` and `c` are both non-`Aggregate` nodes (`Filter`/`Dedup`) so + // neither is bindable — each group is a *clean* two-candidate + // SharedSubDAGStrategy share-vs-recompute pair, with no + // ASAPStrategies `Summary` candidate mixed in to complicate + // ranking (see `shared_aggregate_across_two_roots_gets_both_strategies_candidates` + // for what a *mixed*-shape group looks like — deliberately avoided + // here to isolate the SharedSubDAGStrategy-only interaction). + // + // Under ConstantCseCost, consumer_count == 2 loses to maintenance + // (2 * 40 = 80 < 100 ⇒ RecomputeIndependently); consumer_count == 3 wins + // (3 * 40 = 120 > 100 ⇒ Share). `cost_sorted` only ever sees `c`'s raw + // count (2) and picks RecomputeIndependently for it — the WRONG + // answer once `a` itself is accounted for: `a`'s own decision is + // also RecomputeIndependently (same 80-vs-100 threshold), so `a` + // actually runs twice, and each run recomputes `c` once more — + // `c`'s *true* effective count is 2 (via `a`) + 1 (via root3) = 3, + // which flips its own decision to Share. Only global_selection, + // which folds `a`'s decision into `c`'s effective_consumer_count + // before deciding `c`, gets this right. + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; + + let c = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap() + }; + let a = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: c(), + })) + .unwrap() + }; + + let space = search_workload(vec![("root1", a()), ("root2", a()), ("root3", c())]); + + // Fixture sanity: root1/root2 merged onto one shared `a`, and `c` + // (root1/root2's shared child, and root3 itself) merged onto one + // shared `c` with raw consumer_count 2, and both groups are clean + // (non-mixed) two-candidate SharedSubDAGStrategy pairs. + assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); + let a_rc = &space.roots[0].1; + let Some(NonASAPOp::Filter { child: c_via_a, .. }) = a_rc.non_asap() else { + panic!("expected root1/root2 to still be a Filter"); + }; + assert!(Rc::ptr_eq(c_via_a, &space.roots[2].1)); + let a_group = space.candidates_for_target(a_rc).unwrap(); + let c_group = space.candidates_for_target(c_via_a).unwrap(); + assert_eq!( + a_group.consumer_count, 2, + "fixture sanity: a has 2 consumers" + ); + assert_eq!( + c_group.consumer_count, 2, + "fixture sanity: c has 2 raw consumers (via a's child edge, and via root3)" + ); + assert_eq!( + a_group.candidates.len(), + 2, + "fixture sanity: a is a clean Rewrite pair" + ); + assert_eq!( + c_group.candidates.len(), + 2, + "fixture sanity: c is a clean Rewrite pair" + ); + + // The naive/local answer: cost_sorted ranks c using its raw count + // (2) alone and prefers RecomputeIndependently. + let ranked = cost_sorted(&space, &ConstantCseCost); + let c_ranked = ranked + .iter() + .find(|g| Rc::ptr_eq(g.target, c_via_a)) + .unwrap(); + let c_top_shares = matches!( + &c_ranked.candidates[0].replacement, + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, c_via_a) + ); + assert!( + !c_top_shares, + "cost_sorted, blind to a's own decision, must (wrongly) prefer \ + RecomputeIndependently for c using its raw consumer_count of 2" + ); + + // The corrected, cross-group-aware answer: global_selection folds + // a's own RecomputeIndependently choice into c's effective count + // (2 from a + 1 from root3 = 3) and flips to Share. + let selected = global_selection(&space, &ConstantCseCost); + let a_selected = selected.for_target(a_rc).unwrap(); + let c_selected = selected.for_target(c_via_a).unwrap(); + + assert_eq!( + a_selected.effective_consumer_count, 2, + "a has no interacting ancestor" + ); + let a_shares = matches!( + &a_selected.chosen.unwrap().replacement, + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, a_rc) + ); + assert!( + !a_shares, + "fixture sanity: a itself must also choose RecomputeIndependently" + ); + + assert_eq!( + c_selected.effective_consumer_count, 3, + "c's effective count must be 2 (a, itself recomputed twice) + 1 (root3)" + ); + let c_shares = matches!( + &c_selected.chosen.unwrap().replacement, + Replacement::SubDAG(rc) if Rc::ptr_eq(rc, c_via_a) + ); + assert!( + c_shares, + "global_selection must flip c to Share once a's own recomputation is accounted for" + ); + } + + #[test] + fn complete_plan_costs_reject_unbound_cse_arms() { + struct CompletePlanCost; + impl CostModel for CompletePlanCost { + fn candidate_cost_covers_complete_plan(&self) -> bool { + true + } + + fn candidate_cost( + &self, + candidate: &ReplacementSubDAG, + _target: &TargetSubDAG<'_>, + ) -> Option { + assert!(!is_cse_candidate(candidate)); + None + } + + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + + fn cse_share_decision(&self, _candidate: &CseCandidate) -> ShareDecision { + ShareDecision::RecomputeIndependently + } + } + + let shared = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap(); + let space = search_workload(vec![ + ("left", Rc::clone(&shared)), + ("right", Rc::clone(&shared)), + ]); + let planned = &space.roots[0].1; + + let selected = global_selection(&space, &CompletePlanCost); + assert!(selected.for_target(planned).unwrap().chosen.is_none()); + } + + #[test] + fn effective_repetition_materializes_a_cse_choice_for_a_single_edge_child() { + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; + + let c = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap() + }; + let a = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: c(), + })) + .unwrap() + }; + let space = search_workload(vec![("root1", a()), ("root2", a())]); + let a_rc = &space.roots[0].1; + let Some(NonASAPOp::Filter { child: c_rc, .. }) = a_rc.non_asap() else { + panic!("expected Filter root"); + }; + + assert_eq!(space.candidates_for_target(c_rc).unwrap().consumer_count, 1); + assert!(cse_candidate_pair(space.candidates_for_target(c_rc).unwrap()).is_some()); + + let selected = global_selection(&space, &ConstantCseCost); + let child = selected.for_target(c_rc).unwrap(); + assert_eq!(child.effective_consumer_count, 2); + assert!(child.chosen.is_some()); + } + + #[test] + fn shared_ancestor_keeps_a_single_use_cse_descendant_selected() { + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; + + struct AlwaysShare; + impl CostModel for AlwaysShare { + fn allow_uncosted_legacy_selection(&self) -> bool { + true + } + + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + + fn cse_share_decision(&self, _candidate: &CseCandidate) -> ShareDecision { + ShareDecision::Share + } + } + + let child = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["job"]), + })) + .unwrap() + }; + let parent = || { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: child(), + })) + .unwrap() + }; + let space = search_workload(vec![("root1", parent()), ("root2", parent())]); + let parent_rc = &space.roots[0].1; + let Some(NonASAPOp::Filter { + child: child_rc, .. + }) = parent_rc.non_asap() + else { + panic!("expected Filter root"); + }; + + let selected = global_selection(&space, &AlwaysShare); + assert_eq!( + selected + .for_target(parent_rc) + .unwrap() + .effective_consumer_count, + 2 + ); + let child_selection = selected.for_target(child_rc).unwrap(); + assert_eq!(child_selection.effective_consumer_count, 1); + assert_eq!( + child_selection.chosen.map(|candidate| candidate.provenance), + Some(ReplacementProvenance::CseShare), + "a descendant collapsed to one execution still needs a selected plan" + ); + } + + #[test] + fn global_selection_propagates_uses_through_the_selected_rewrite() { + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; + + struct ReplaceFilterChild; + impl ReplacementStrategy for ReplaceFilterChild { + fn matches(&self, target: &TargetSubDAG<'_>) -> bool { + matches!(target.root.non_asap(), Some(NonASAPOp::Filter { .. })) + } + + fn replacements(&self, _target: &TargetSubDAG<'_>) -> Vec { + vec![ReplacementSubDAG { + strategy: "ReplaceFilterChild", + replacement: Replacement::SubDAG( + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::Dedup { + cols: vec![0], + child: metric_scan(&["replacement"]), + }, + )) + .unwrap(), + ), + provenance: ReplacementProvenance::LogicalRewrite, + rationale: "replace the Filter and its input".into(), + }] + } + } + + let original_child = metric_scan(&["original"]); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: Rc::clone(&original_child), + })) + .unwrap(); + let strategies: Vec> = vec![Box::new(ReplaceFilterChild)]; + let space = search_workload_with(vec![("q", root)], &strategies); + let root = &space.roots[0].1; + let selected = global_selection(&space, &DefaultCostModel); + let Replacement::SubDAG(rewrite) = &selected + .for_target(root) + .unwrap() + .chosen + .unwrap() + .replacement + else { + panic!("expected logical rewrite"); + }; + let Some(NonASAPOp::Dedup { + child: replacement_child, + .. + }) = rewrite.non_asap() + else { + panic!("expected Dedup rewrite"); + }; + let Some(NonASAPOp::Filter { + child: original_child, + .. + }) = root.non_asap() + else { + panic!("expected Filter root"); + }; + + assert_eq!( + selected + .for_target(original_child) + .unwrap() + .effective_consumer_count, + 0 + ); + assert_eq!( + selected + .for_target(replacement_child) + .unwrap() + .effective_consumer_count, + 1 + ); + } + + // A cheap but physically infeasible candidate must not be selected. + #[test] + fn explicit_summary_infeasibility_prevents_selection() { + struct Unsupported; + impl CostModel for Unsupported { + fn rank_candidates( + &self, + _: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn candidate_cost(&self, _: &ReplacementSubDAG, _: &TargetSubDAG<'_>) -> Option { + Some(Cost(1.0)) + } + fn summary_support_evidence(&self, _: &OperatorNode) -> Option { + Some(false) + } + } + let root = lower_promql("sum_over_time(a[1m])", AccuracyTarget::Exact); + let space = search_workload(vec![("q", root)]); + let selected = global_selection(&space, &Unsupported); + assert!(selected + .for_target(&space.roots[0].1) + .unwrap() + .chosen + .is_none()); + } + + // Composable temporal/grouped Sum must be executable as one producer. + #[test] + fn grouped_temporal_sum_has_one_summary_producer_candidate() { + let root = lower_promql("sum by(job)(sum_over_time(a[1m]))", AccuracyTarget::Exact); + let candidates = ASAPStrategies::default().replacements(&TargetSubDAG::new(&root)); + assert!(candidates + .iter() + .any(|candidate| matches!(&candidate.replacement, + Replacement::SubDAG(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())))); + struct PreferComposed; + impl CostModel for PreferComposed { + fn rank_candidates( + &self, + _: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn candidate_cost( + &self, + candidate: &ReplacementSubDAG, + _: &TargetSubDAG<'_>, + ) -> Option { + Some(Cost( + if matches!(&candidate.replacement, + Replacement::SubDAG(node) if matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())) + { + 1.0 + } else { + 100.0 + }, + )) + } + } + let space = search_workload(vec![("q", root.clone())]); + let selected = global_selection(&space, &PreferComposed); + let node = selected + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + assert!(matches!(&node.operator, + Operator::ASAP(ASAPOp::SummaryAgg { reduction: Reduction::Reduce(_), child, .. }) + if !child.contains_asap())); + } + + // Mixed candidate ranking must honor explicit costs, not legacy estimates. + #[test] + fn mixed_candidate_ranking_uses_explicit_candidate_costs() { + struct ExplicitCosts; + impl CostModel for ExplicitCosts { + fn rank_candidates( + &self, + _: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + fn candidate_cost( + &self, + candidate: &ReplacementSubDAG, + _: &TargetSubDAG<'_>, + ) -> Option { + Some(Cost( + if candidate.provenance == ReplacementProvenance::LogicalRewrite { + 1.0 + } else { + 100.0 + }, + )) + } + } + let root = lower_promql("sum by(job)(sum_over_time(a[1m]))", AccuracyTarget::Exact); + let space = search_workload(vec![("q", root)]); + let selection = global_selection(&space, &ExplicitCosts); + let selected = selection + .for_target(&space.roots[0].1) + .unwrap() + .chosen + .unwrap(); + assert_eq!(selected.provenance, ReplacementProvenance::LogicalRewrite); + } + + #[test] + fn global_selection_compares_a_logical_rewrite_with_the_cse_choice() { + struct PreferLogicalRewrite; + + impl CostModel for PreferLogicalRewrite { + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + + fn estimate_cost( + &self, + candidate: &ReplacementSubDAG, + _target: &TargetSubDAG<'_>, + ) -> f64 { + match candidate.provenance { + ReplacementProvenance::LogicalRewrite => 0.0, + _ => 100.0, + } + } + } + + let a = agg(vec![2], AggIntent::Avg { col: None }, metric_scan(&["job"])); + let b = agg(vec![2], AggIntent::Avg { col: None }, metric_scan(&["job"])); + let space = search_workload(vec![("a", a), ("b", b)]); + let root = &space.roots[0].1; + let selected = global_selection(&space, &PreferLogicalRewrite); + + assert_eq!( + selected + .for_target(root) + .and_then(|group| group.chosen) + .map(|candidate| candidate.provenance), + Some(ReplacementProvenance::LogicalRewrite) + ); + } + + #[test] + fn topological_order_puts_a_later_discovered_parent_before_its_child() { + // Mirrors nested_shared_sub-DAG_below_an_unshared_parent_is_still_discovered's + // diamond fixture: discover_targets's own `order` visits root_b (a + // parent of `shared`) *after* `shared` itself, because `shared` was + // already fully walked via root_a first. A naive "process + // discover_targets's own order" DP would see root_b's child edge + // after already processing `shared` — topological_order must not + // make that mistake. + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::Predicate; + + let shared = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); + let root_a = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(1))), + child: shared.clone(), + })) + .unwrap(); + let root_b = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Int64(2))), + child: shared, + })) + .unwrap(); + let roots = vec![("a", root_a), ("b", root_b)]; + + let space = search_workload(roots); + let order = space.order(); + let dag = reference_dag(&space); + + // Discovery-order sanity: root_b comes after the shared child in + // discover_targets's own order (the exact non-topological case this + // test exists to cover). + let Some(NonASAPOp::Filter { + child: shared_via_a, + .. + }) = space.roots[0].1.non_asap() + else { + panic!("expected a Filter root"); + }; + let shared_ptr = Rc::as_ptr(shared_via_a); + let root_b_ptr = Rc::as_ptr(&space.roots[1].1); + let shared_discovery_pos = order.iter().position(|p| *p == shared_ptr).unwrap(); + let root_b_discovery_pos = order.iter().position(|p| *p == root_b_ptr).unwrap(); + assert!( + root_b_discovery_pos > shared_discovery_pos, + "fixture sanity: discover_targets's own order must NOT already be topological here" + ); + + let topo = topological_order(order, &dag); + let shared_topo_pos = topo.iter().position(|p| *p == shared_ptr).unwrap(); + let root_b_topo_pos = topo.iter().position(|p| *p == root_b_ptr).unwrap(); + assert!( + root_b_topo_pos < shared_topo_pos, + "topological_order must place root_b (a parent of the shared node) before it, \ + unlike discover_targets's own discovery order" + ); + } + + // ── Selection over Stage 1 candidates (split from `replacement` tests) ─ + + // Every exposed query result has a evaluation; internal accumulator frontiers stay states. + #[test] + fn selected_query_roots_do_not_leak_exact_accumulator_state() { + for query in [ + "sum by(job)(rate(m[1m]))", + "sum by(job)(m)", + "sum_over_time(m[1m])", + ] { + let root = lower_promql(query, AccuracyTarget::Exact); + let space = search_workload(vec![(0usize, root)]); + let selected = global_selection(&space, &DefaultCostModel) + .assemble_selected_query(&space.roots[0].1) + .unwrap() + .unwrap(); + assert!( + selected + .schema + .fields + .iter() + .all(|field| matches!(field.dtype, FieldDataType::Plain(_))), + "{query}: query root leaks state: {:?}", + selected.schema + ); + } + } + + /// Candidates kept with missing accuracy evidence are never chosen automatically. + #[test] + fn selection_never_chooses_a_candidate_missing_accuracy_evidence() { + let never_unproven = |space: &CandidateLogicalASAPDAGs<&str>| { + let selected = global_selection(space, &DefaultCostModel); + assert!(!selected + .for_target(&space.roots[0].1) + .unwrap() + .chosen + .is_some_and(ReplacementSubDAG::has_missing_accuracy_evidence)); + assert!(selected + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .is_some()); + }; + let count = AggIntent::Count { + accuracy: AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }, + }; + never_unproven(&search_workload(vec![( + "q", + agg(vec![2], count, metric_scan(&["job"])), + )])); + never_unproven(&search_workload_with_targets( + vec![( + "q", + agg(vec![2], default_cardinality(), metric_scan(&["job"])), + Some(AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + }), + )], + &default_strategies(), + &DefaultAccuracyModel, + )); + let inner = agg( + vec![2], + AggIntent::Count { + accuracy: AccuracyTarget::Epsilon(0.01), + }, + metric_scan(&["job"]), + ); + let topk = agg( + vec![], + AggIntent::TopK { + k: 10, + accuracy: AccuracyTarget::Epsilon(0.01), + }, + inner, + ); + never_unproven(&search_workload_with_targets( + vec![("q", topk, Some(AccuracyTarget::Epsilon(0.01)))], + &default_strategies(), + &DefaultAccuracyModel, + )); + } + + /// A root target that rejects every summary leaves nothing to select. + #[test] + fn nothing_is_selected_when_the_root_target_rejects_every_summary() { + let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); + let space = search_workload_with_targets( + vec![("q", q, Some(AccuracyTarget::Epsilon(0.001)))], + &default_strategies(), + &DefaultAccuracyModel, + ); + let root = &space.roots[0].1; + assert!(global_selection(&space, &DefaultCostModel) + .for_target(root) + .unwrap() + .chosen + .is_none()); + } + + /// Admits rank-over-rank composition, which `DefaultAccuracyModel` has no + /// rule for, so a nested quantile summary can be selected. + struct RankAdditiveModel; + + impl AccuracyModel for RankAdditiveModel { + fn local_guarantee( + &self, + family: &FieldDataType, + query: &SketchStatistic, + ) -> Option { + DefaultAccuracyModel.local_guarantee(family, query) + } + + fn propagate( + &self, + op: &CompositionOperator, + inputs: &[ResultGuarantee], + local: Option<&ResultGuarantee>, + stats: &PropagationStats, + ) -> Result { + let rank = |g: &ResultGuarantee| g.is_exact() || g.metric == ErrorMetric::Rank; + if let (CompositionOperator::ApproximateAggregate, true, Some(local)) = + (op, inputs.iter().all(rank), local) + { + let relabel = |g: &ResultGuarantee| ResultGuarantee { + metric: ErrorMetric::AbsoluteValue, + ..g.clone() + }; + let inputs: Vec<_> = inputs.iter().map(relabel).collect(); + let mut out = + DefaultAccuracyModel.propagate(op, &inputs, Some(&relabel(local)), stats)?; + out.metric = local.metric; + return Ok(out); + } + DefaultAccuracyModel.propagate(op, inputs, local, stats) + } + + fn satisfies(&self, guarantee: &ResultGuarantee, target: &AccuracyTarget) -> bool { + DefaultAccuracyModel.satisfies(guarantee, target) + } + } + + #[test] + fn global_selection_can_choose_nested_summaries() { + // The same nested summary remains available through workload search + // and global cost ranking. + let inner = agg( + vec![2], + quantile_eps_intent(0.5, 0.1), + metric_scan(&["job"]), + ); + let outer = agg(vec![], quantile_eps_intent(0.99, 0.1), inner); + let strategies: Vec> = vec![Box::new( + ASAPStrategies::new_with_planning_inputs(&RankAdditiveModel, &EqualSplitAllocator), + )]; + let space = search_workload_with(vec![("q", Rc::clone(&outer))], &strategies); + let root = &space.roots[0].1; + let group = space.candidates_for_target(root).unwrap(); + assert!(!group.rejected.is_empty()); + assert!(group.candidates.iter().all(|c| match &c.replacement { + // A summary candidate (old `Replacement::Summary`) contains an + // ASAP node; a logical rewrite (old `Replacement::Rewrite`) does not. + Replacement::SubDAG(node) if node.contains_asap() => { + node.guarantee.as_ref().is_some_and(|g| { + DefaultAccuracyModel.satisfies(g, &AccuracyTarget::Epsilon(0.1)) + }) + } + Replacement::SubDAG(_) => false, + Replacement::ExactComposition(_) => false, + })); + let ranked = cost_sorted(&space, &DefaultCostModel); + let root_ranked = ranked.iter().find(|g| Rc::ptr_eq(g.target, root)).unwrap(); + assert_eq!(root_ranked.candidates.len(), group.candidates.len()); + + let selection = global_selection(&space, &DefaultCostModel); + let chosen = selection + .for_target(root) + .unwrap() + .chosen + .expect("a nested summary candidate wins"); + let Replacement::SubDAG(node) = &chosen.replacement else { + panic!() + }; + assert!(matches!( + node.operator, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + )); + } + + // ── Accuracy reconciliation (moved from `accuracy::reconciliation` tests) ─ + + mod accuracy_reconciliation { + use super::*; + use asap_types::ir::operator::operator_properties::Source; + use asap_types::ir::schema::{DataType, Field, Schema}; + + /// `[ts(0), value(1), job(2)]`, uniquely keyed by `ts` so CSE can hoist it. + fn metric_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("job", DataType::Utf8, true), + ], + 0, + vec![vec![0]], + ), + })) + .unwrap() + } + + fn quantile( + q: f64, + accuracy: AccuracyTarget, + child: &Rc, + ) -> Rc { + agg( + vec![2], + AggIntent::Quantile { + col: None, + q, + accuracy, + }, + Rc::clone(child), + ) + } + + // ── cost: reading the sibling must not be priced like recomputing + // `target` independently per consumer ──────────────────────────────── + + #[test] + fn estimate_cost_does_not_scale_with_the_readers_own_consumer_count() { + // Regression guard for the review-reported sign inversion: pricing + // this candidate like `CseRecompute` ("rebuild `target`, once per + // consumer") made it artificially *more* expensive exactly as more + // of `target`'s own consumers stood to benefit from reading the + // already-necessary tighter sibling instead — the literal opposite + // of the intended incentive. The real cost is "one more read against + // `rc`'s own build," which must not scale with `target`'s own + // `consumer_count`. + let scan = metric_scan(); + let tight = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let loose = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); + let strategy = + AccuracyReconciliationStrategy::new(&[Rc::clone(&tight), Rc::clone(&loose)]); + let candidate = strategy + .replacements(&TargetSubDAG::new(&loose)) + .into_iter() + .next() + .expect("loose has a reconciliation candidate reading the tight sibling"); + + let cost_model = DefaultCostModel; + let single_consumer = TargetSubDAG::with_consumer_count(&loose, 1); + let many_consumers = TargetSubDAG::with_consumer_count(&loose, 5); + + let cost_single = cost_model.estimate_cost(&candidate, &single_consumer); + let cost_many = cost_model.estimate_cost(&candidate, &many_consumers); + + assert!( + cost_single.is_finite(), + "expected a real cost, not the NaN placeholder: {cost_single}" + ); + assert_eq!( + cost_single, cost_many, + "AccuracyReconciliation's estimate_cost must price 'read the sibling', not scale \ + with the reader's own consumer_count the way CseRecompute's 'rebuild independently \ + per consumer' formula does (single-consumer: {cost_single}, 5 consumers: \ + {cost_many})" + ); + } + + // ── cost_sorted / global_selection: single-consumer and shared-consumer ─ + + #[test] + fn cost_sorted_and_global_selection_handle_a_single_consumer_looser_target() { + let scan = metric_scan(); + let tight = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let loose = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); + let space = search_workload(vec![("tight", tight), ("loose", loose)]); + let loose_root = &space.roots[1].1; + + let cost_model = DefaultCostModel; + let ranked = cost_sorted(&space, &cost_model); + let loose_ranked = ranked + .iter() + .find(|group| Rc::ptr_eq(group.target, loose_root)) + .expect("loose has its own ranked group"); + assert!( + loose_ranked.costs.iter().all(|cost| cost.is_finite()), + "no candidate should cost NaN under DefaultCostModel: {:?}", + loose_ranked.costs + ); + assert!( + loose_ranked + .candidates + .iter() + .any(|c| c.strategy == "AccuracyReconciliationStrategy"), + "the reconciliation candidate must still be present, ranked, not filtered" + ); + + let selected = global_selection(&space, &cost_model); + let chosen = selected + .for_target(loose_root) + .and_then(|group| group.chosen); + assert!( + chosen.is_some(), + "global_selection must commit to some candidate for a single-consumer looser target" + ); + // With no recompute term at all (it never rebuilds `target`), this + // candidate strictly undercuts every ASAPStrategies + // candidate (which each pay a recompute term on top of their own + // maintenance term) under DefaultCostModel's numbers — the sane + // direction: reading an already-necessary sibling should be able to + // win on its own merit, not just fail to lose as badly as before. + assert_eq!( + chosen.map(|c| c.provenance), + Some(asap_logical_optimizer::pass1::replacement::ReplacementProvenance::AccuracyReconciliation) + ); + } + + #[test] + fn cost_sorted_and_global_selection_handle_a_shared_looser_target() { + // The loose accuracy target itself has 2 direct consumers (two + // independently-built but structurally identical loose queries + // merge onto one Rc via ordinary CSE), *and* a separate, + // single-consumer tight sibling exists over the same input — the + // scenario the issue itself targets: `SharedSubDAGStrategy`'s own + // CseShare/CseRecompute pair is on the table for the loose target's + // own 2 consumers at the same time as this strategy's "read the + // tight sibling instead" candidate. + let scan = metric_scan(); + let loose_a = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); + let loose_b = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); + let tight = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); + + let space = search_workload(vec![ + ("loose_a", Rc::new(loose_a)), + ("loose_b", Rc::new(loose_b)), + ("tight", Rc::new(tight)), + ]); + + // Fixture sanity: the two loose roots really did merge onto one Rc. + assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); + let loose_group = space + .candidates_for_target(&space.roots[0].1) + .expect("the merged loose target has a group"); + assert_eq!(loose_group.consumer_count, 2); + assert!( + loose_group + .candidates + .iter() + .any(|c| c.strategy == "AccuracyReconciliationStrategy"), + "the reconciliation candidate must still be proposed alongside the CSE share/recompute \ + pair, not crowded out: {:?}", + loose_group + .candidates + .iter() + .map(|c| (c.strategy, c.provenance)) + .collect::>() + ); + + let cost_model = DefaultCostModel; + let ranked = cost_sorted(&space, &cost_model); + let loose_ranked = ranked + .iter() + .find(|group| Rc::ptr_eq(group.target, &space.roots[0].1)) + .expect("loose has its own ranked group"); + assert!( + loose_ranked.costs.iter().all(|cost| cost.is_finite()), + "no candidate should cost NaN under DefaultCostModel, shared or not: {:?}", + loose_ranked.costs + ); + + let selected = global_selection(&space, &cost_model); + let chosen = selected + .for_target(&space.roots[0].1) + .and_then(|group| group.chosen); + assert!( + chosen.is_some(), + "global_selection must commit to some candidate for the shared looser target" + ); + // Under `DefaultCostModel`'s numbers, `CseShare` (flat maintenance, + // no recompute term) and this strategy's own candidate (also a + // flat, non-scaling read cost after the fix) land tied, and + // `global_selection` breaks ties in `CseShare`'s favor (it only ever + // overrides the CSE choice on a *strict* `<`, not `<=`) — a sane, + // deliberate tie-break, not the "reconciliation always loses to + // CseShare regardless of its own real merit" bug this test guards + // against (see `estimate_cost_does_not_scale_with_the_readers_own_consumer_count` + // for the direct regression check that the old `* consumer_count` + // scaling — which made this an unfair, ever-widening loss instead + // of a tie — is gone). + assert_eq!( + chosen.map(|c| c.provenance), + Some(asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseShare) + ); + } + + #[test] + fn global_selection_propagates_reconciled_consumers_to_the_tighter_group() { + struct PreferReconciliation; + + impl CostModel for PreferReconciliation { + fn rank_candidates( + &self, + _intent: &AggIntent, + candidates: &[SketchAlgorithm], + ) -> Vec { + candidates.to_vec() + } + + fn estimate_cost( + &self, + candidate: &ReplacementSubDAG, + _target: &TargetSubDAG<'_>, + ) -> f64 { + if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { + 0.0 + } else { + 100.0 + } + } + } + + let scan = metric_scan(); + let tight = quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); + let loose = quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan); + let space = search_workload(vec![ + ("tight", Rc::clone(&tight)), + ("loose", Rc::clone(&loose)), + ]); + + let selected = global_selection(&space, &PreferReconciliation); + let tight_root = &space.roots[0].1; + let loose_root = &space.roots[1].1; + assert_eq!( + selected + .for_target(loose_root) + .and_then(|group| group.chosen) + .map(|candidate| candidate.provenance), + Some(ReplacementProvenance::AccuracyReconciliation), + "fixture must select the cross-sibling rewrite" + ); + assert_eq!( + selected + .for_target(tight_root) + .expect("the tighter sibling is a discovered memo group") + .effective_consumer_count, + 2, + "the tighter build serves its original root and the reconciled looser root" + ); + } + } +} diff --git a/crates/asap-aware-mapping/src/analytical_cost.rs b/crates/plan-selection/src/cost/analytical_cost.rs similarity index 97% rename from crates/asap-aware-mapping/src/analytical_cost.rs rename to crates/plan-selection/src/cost/analytical_cost.rs index 3c19960a9..46c89f4eb 100644 --- a/crates/asap-aware-mapping/src/analytical_cost.rs +++ b/crates/plan-selection/src/cost/analytical_cost.rs @@ -7,14 +7,16 @@ use std::collections::{HashMap, HashSet}; -use asap_types::post_asap::{SketchAlgorithm, SketchParams}; -pub use asap_types::resources::{CacheCapacityEvidence, CacheEvidence, CacheProfile, ModeledCpu}; +use asap_types::ir::schema::{SketchAlgorithm, SketchParams}; +pub use asap_types::workload::resources::{ + CacheCapacityEvidence, CacheEvidence, CacheProfile, ModeledCpu, +}; use asap_types::workload::DataArrival; use serde::{Deserialize, Serialize}; -use crate::physical_operator_statistics::{ +use crate::cost::physical_operator_statistics::{ validate_comparison_scopes, ComparisonScope, EdgeStatistics, OperatorStatistics, - OperatorStatisticsProvider, PromqlEdgeStatistics, PromqlValueKind, SourceCoverage, + OperatorStatisticsProvider, PromqlEdgeStatistics, PromqlValueKind, ScanSelection, }; /// Version of the analytical formulas applied to evidenced physical plans. @@ -185,7 +187,7 @@ impl ResourceCalibration { #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] #[serde(from = "ResourceEstimateWire", into = "ResourceEstimateWire")] pub struct ResourceEstimate { - resources: asap_types::resources::PhysicalResources, + resources: asap_types::workload::resources::PhysicalResources, } // Keep the established three-field JSON format. Required fields make a missing @@ -216,7 +218,7 @@ impl From for ResourceEstimateWire { impl ResourceEstimate { pub const fn new(cpu_ops: f64, peak_memory_bytes: u64, scan_bytes: u64) -> Self { Self { - resources: asap_types::resources::PhysicalResources { + resources: asap_types::workload::resources::PhysicalResources { cpu: ModeledCpu { cpu_ops }, peak_memory_bytes: Some(peak_memory_bytes), retained_memory_bytes: None, @@ -243,7 +245,9 @@ impl ResourceEstimate { .expect("analytical scan bytes are required by construction") } - pub fn resources(&self) -> &asap_types::resources::PhysicalResources { + pub fn resources( + &self, + ) -> &asap_types::workload::resources::PhysicalResources { &self.resources } } @@ -383,9 +387,9 @@ pub struct PhysicalDAGNode { pub operator: PhysicalOperator, pub children: Vec, /// Exact comparison-scope coverage consumed by a scan. Non-scan nodes - /// leave this empty. Reusing `SourceCoverage` prevents a physical plan + /// leave this empty. Reusing `ScanSelection` prevents a physical plan /// from naming a source independently of its snapshot and predicates. - pub source_coverage: Option, + pub scan_selection: Option, /// Maximum transient edge buffer, distinct from logical `output_bytes`. pub output_buffer_bytes: u64, /// State that remains live after this node finishes (zero for ordinary @@ -573,9 +577,10 @@ pub fn estimate_physical_dag_with_cache( let node_statistics = &resolved_statistics[id]; match node.operator { PhysicalOperator::Scan => { - let coverage = node.source_coverage.as_ref().ok_or_else(|| { - AnalyticalCostError::MissingScanSourceCoverage(node.id.clone()) - })?; + let coverage = node + .scan_selection + .as_ref() + .ok_or_else(|| AnalyticalCostError::MissingScanSelection(node.id.clone()))?; if !scope.sources.contains(coverage) { return Err(AnalyticalCostError::ScanOutsideComparisonScope( node.id.clone(), @@ -585,9 +590,9 @@ pub fn estimate_physical_dag_with_cache( consumed_sources.push(coverage); } } - _ if node.source_coverage.is_some() => { + _ if node.scan_selection.is_some() => { return Err(AnalyticalCostError::InvalidPhysicalDAG( - "only scan nodes may declare source coverage", + "only scan nodes may declare scan selection", )); } _ => {} @@ -963,7 +968,7 @@ fn checked_cpu_product(rows: u64, operations_per_row: u64) -> Result Result { @@ -984,7 +989,7 @@ fn partitioned_order_estimate( fn validate_partitioning( input: EdgeStatistics, - partitioning: &crate::physical_operator_statistics::PartitionStatistics, + partitioning: &crate::cost::physical_operator_statistics::PartitionStatistics, partitioned: bool, ) -> Result<(), AnalyticalCostError> { let inconsistent = |reason| Err(AnalyticalCostError::InconsistentOperatorStatistics(reason)); @@ -1877,8 +1882,9 @@ pub(crate) fn validate_operator_semantics( } fn require_promql_unary( - edges: &crate::physical_operator_statistics::UnaryEdgeStatistics, -) -> Result { + edges: &crate::cost::physical_operator_statistics::UnaryEdgeStatistics, +) -> Result +{ let promql = edges.promql.ok_or(AnalyticalCostError::MissingOrStale( "promql_edge_statistics", ))?; @@ -1888,8 +1894,11 @@ fn require_promql_unary( } fn require_promql_binary( - edges: &crate::physical_operator_statistics::BinaryEdgeStatistics, -) -> Result { + edges: &crate::cost::physical_operator_statistics::BinaryEdgeStatistics, +) -> Result< + crate::cost::physical_operator_statistics::PromqlBinaryEdgeStatistics, + AnalyticalCostError, +> { let promql = edges.promql.ok_or(AnalyticalCostError::MissingOrStale( "promql_edge_statistics", ))?; @@ -1900,7 +1909,7 @@ fn require_promql_binary( } fn validate_promql_cardinality_preserving_shape( - promql: crate::physical_operator_statistics::PromqlUnaryEdgeStatistics, + promql: crate::cost::physical_operator_statistics::PromqlUnaryEdgeStatistics, ) -> Result<(), AnalyticalCostError> { validate_promql_edge(promql.input)?; validate_promql_edge(promql.output)?; @@ -1914,7 +1923,7 @@ fn validate_promql_cardinality_preserving_shape( } fn validate_promql_filter_shape( - promql: crate::physical_operator_statistics::PromqlUnaryEdgeStatistics, + promql: crate::cost::physical_operator_statistics::PromqlUnaryEdgeStatistics, ) -> Result<(), AnalyticalCostError> { validate_promql_edge(promql.input)?; validate_promql_edge(promql.output)?; @@ -1980,7 +1989,7 @@ fn validate_promql_edge_shape( fn validate_promql_bridge( operator: PhysicalOperator, - edges: &crate::physical_operator_statistics::UnaryEdgeStatistics, + edges: &crate::cost::physical_operator_statistics::UnaryEdgeStatistics, ) -> Result<(), AnalyticalCostError> { let promql = require_promql_unary(edges)?; let valid = match operator { @@ -2014,7 +2023,7 @@ fn validate_promql_binary( cardinality: PromqlVectorCardinality, build_side: Option, matching_key_bytes: u64, - edges: &crate::physical_operator_statistics::BinaryEdgeStatistics, + edges: &crate::cost::physical_operator_statistics::BinaryEdgeStatistics, ) -> Result<(), AnalyticalCostError> { let promql = require_promql_binary(edges)?; validate_instant_vector_rows(edges.output, promql.output)?; @@ -2105,7 +2114,7 @@ fn validate_promql_series_sample( grouping_key_count: u64, group_count: u64, key_bytes: u64, - edges: &crate::physical_operator_statistics::UnaryEdgeStatistics, + edges: &crate::cost::physical_operator_statistics::UnaryEdgeStatistics, ) -> Result<(), AnalyticalCostError> { let promql = require_promql_unary(edges)?; validate_instant_vector_rows(edges.output, promql.output)?; @@ -2163,8 +2172,6 @@ pub enum AnalyticalCostError { UnsupportedDataArrival(DataArrival), #[error("ingestion rate must be finite and non-negative, got {0}")] InvalidIngestionRate(f64), - #[error("summary lifecycle, maintenance mode, and evaluation schedule are inconsistent")] - IncompatibleLifecycleGuarantee, #[error("bootstrap row and byte evidence must either both be zero or both be non-zero")] InconsistentBootstrapEvidence, #[error("required summary operation cost {0} must be finite and positive, got {1}")] @@ -2195,13 +2202,13 @@ pub enum AnalyticalCostError { UnsupportedQueryOperator, #[error("inconsistent operator statistics: {0}")] InconsistentOperatorStatistics(&'static str), - #[error("summary operation {0} has no lifecycle-aware cost formula")] + #[error("summary operation {0} has no cost formula")] UnsupportedSummaryOperation(&'static str), #[error("required comparison-scope field {0} is missing")] MissingComparisonScope(&'static str), - #[error("scan node {0} does not declare source coverage")] - MissingScanSourceCoverage(String), - #[error("scan node {0} reads source coverage outside the comparison scope")] + #[error("scan node {0} does not declare scan selection")] + MissingScanSelection(String), + #[error("scan node {0} reads scan selection outside the comparison scope")] ScanOutsideComparisonScope(String), #[error("raw and candidate comparison scopes differ in {0}")] ComparisonScopeMismatch(&'static str), @@ -2231,10 +2238,10 @@ fn checked_bytes(parts: &[u64]) -> Result { #[cfg(test)] mod tests { use super::*; - use crate::physical_operator_statistics::{ + use crate::cost::physical_operator_statistics::{ validate_comparison_scopes, BinaryEdgeStatistics, ComparisonScope, EdgeStatistics, OperatorStatistics, PartitionStatistics, PromqlBinaryEdgeStatistics, PromqlEdgeStatistics, - PromqlUnaryEdgeStatistics, PromqlValueKind, SourceCoverage, UnaryEdgeStatistics, + PromqlUnaryEdgeStatistics, PromqlValueKind, ScanSelection, UnaryEdgeStatistics, }; /// Analytical estimates reuse the shared dimensions while preserving exact @@ -2245,7 +2252,7 @@ mod tests { assert_eq!(estimate.cpu_ops(), 12.5); assert_eq!(estimate.peak_memory_bytes(), u64::MAX); assert_eq!(estimate.scan_bytes(), 0); - let shared: &asap_types::resources::PhysicalResources = + let shared: &asap_types::workload::resources::PhysicalResources = estimate.resources(); assert_eq!(shared.cpu.cpu_ops, 12.5); assert_eq!(shared.peak_memory_bytes, Some(u64::MAX)); @@ -2475,7 +2482,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(comparison_scope().sources[0].clone()), + scan_selection: Some(comparison_scope().sources[0].clone()), output_buffer_bytes: 8, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2484,7 +2491,7 @@ mod tests { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 8, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2814,7 +2821,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2823,7 +2830,7 @@ mod tests { id: "left".into(), operator: filter_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 4, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2832,7 +2839,7 @@ mod tests { id: "right".into(), operator: filter_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 4, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2841,7 +2848,7 @@ mod tests { id: "root".into(), operator: PhysicalOperator::Concat, children: vec!["left".into(), "right".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 8, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2905,7 +2912,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::Once, @@ -2914,7 +2921,7 @@ mod tests { id: "state".into(), operator: aggregate_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 16, retained_bytes: 32, execution: ExecutionMultiplicity::Once, @@ -2926,7 +2933,7 @@ mod tests { offset: 0, }, children: vec!["state".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 16, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -2969,7 +2976,7 @@ mod tests { } fn comparison_scope() -> ComparisonScope { - use asap_types::pre_asap::query_expr::Source; + use asap_types::ir::operator::operator_properties::Source; use asap_types::workload::{ DurationMs, QueryRecurrence, QueryTimeScope, RepeatedDemand, RepetitionInterval, TimeSelection, TimestampMs, @@ -2987,7 +2994,7 @@ mod tests { lookback: Some(DurationMs(300_000)), as_of: Some(TimestampMs(1_000)), }, - sources: vec![SourceCoverage { + sources: vec![ScanSelection { source: Source::Table { table_ref: "metrics".into(), }, @@ -3008,9 +3015,7 @@ mod tests { #[test] fn comparison_rejects_different_snapshot_predicate_time_or_horizon() { - use std::rc::Rc; - - use asap_types::pre_asap::query_expr::{Predicate, QueryExpr}; + use asap_types::ir::{Predicate, ScalarExpr}; use asap_types::workload::{DurationMs, TimestampMs}; let raw = comparison_scope(); @@ -3026,7 +3031,7 @@ mod tests { candidate = raw.clone(); candidate.sources[0] .predicates - .push(Predicate(Rc::new(QueryExpr::promql_scalar(1.0)))); + .push(Predicate(ScalarExpr::literal_f64(1.0))); assert_eq!( validate_comparison_scopes(&raw, &candidate), Err(AnalyticalCostError::ComparisonScopeMismatch("sources")) @@ -3059,7 +3064,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3068,7 +3073,7 @@ mod tests { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 4, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3134,7 +3139,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3143,7 +3148,7 @@ mod tests { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 4, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3197,7 +3202,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3206,7 +3211,7 @@ mod tests { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3245,8 +3250,8 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(SourceCoverage { - source: asap_types::pre_asap::query_expr::Source::Table { + scan_selection: Some(ScanSelection { + source: asap_types::ir::operator::operator_properties::Source::Table { table_ref: "other_metrics".into(), }, source_snapshot_id: "catalog-version-42".into(), @@ -3283,7 +3288,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3302,9 +3307,7 @@ mod tests { assert_eq!( estimate_physical_dag(&nodes, "scan", &comparison_scope(), &provided), - Err(AnalyticalCostError::MissingScanSourceCoverage( - "scan".into() - )) + Err(AnalyticalCostError::MissingScanSelection("scan".into())) ); } @@ -3312,8 +3315,8 @@ mod tests { fn physical_dag_rejects_an_unconsumed_scope_source() { let mut scope = comparison_scope(); let coverage = scope.sources[0].clone(); - scope.sources.push(SourceCoverage { - source: asap_types::pre_asap::query_expr::Source::Table { + scope.sources.push(ScanSelection { + source: asap_types::ir::operator::operator_properties::Source::Table { table_ref: "auxiliary".into(), }, source_snapshot_id: "catalog-version-42".into(), @@ -3324,7 +3327,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3357,7 +3360,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3366,7 +3369,7 @@ mod tests { id: "aggregate".into(), operator: aggregate_operator(), children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 16, retained_bytes: 32, execution: ExecutionMultiplicity::Once, @@ -3411,7 +3414,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 10, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3642,7 +3645,7 @@ mod tests { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(comparison_scope().sources[0].clone()), + scan_selection: Some(comparison_scope().sources[0].clone()), output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -3677,10 +3680,10 @@ mod tests { // Legacy mapping imports are aliases of the shared schema, not a second type. #[test] fn shared_cache_types_work_through_legacy_mapping_imports() { - let central: asap_types::resources::CacheProfile = cache_profile(100, 50); + let central: asap_types::workload::resources::CacheProfile = cache_profile(100, 50); let legacy: CacheProfile = serde_json::from_value(serde_json::to_value(¢ral).unwrap()).unwrap(); - fn accepts_shared(_: &asap_types::resources::CacheProfile) {} + fn accepts_shared(_: &asap_types::workload::resources::CacheProfile) {} accepts_shared(&legacy); assert_eq!(central, legacy); assert_eq!( diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/plan-selection/src/cost/cost_model.rs similarity index 69% rename from crates/asap-aware-mapping/src/cost_model.rs rename to crates/plan-selection/src/cost/cost_model.rs index 0315e88f1..5f42fb2ef 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/plan-selection/src/cost/cost_model.rs @@ -6,68 +6,51 @@ //! needs knowledge this crate doesn't have and shouldn't acquire: the crate //! doc's layering invariant is that `asap-plan` depends only on [`asap_ir`], //! never on a runtime or a deployment model. What it *can* own is the -//! interface every deployment's cost model plugs into, so [`replacement`]'s -//! summary selection has exactly one extension point instead of forcing -//! each downstream (ASAPCollector + ASAPQuery-backend, ASAPFusion, …) to -//! fork `replacement::realizations_for_intent`. +//! interface every deployment's cost model plugs into, so selection over the +//! candidates [`replacement`] generates has exactly one extension point. +//! Candidate generation itself does not consult a cost model. //! -//! This trait is scoped to the approximate-**sketch** family specifically -//! ([`CostModel::rank_candidates`]/[`size_params`](CostModel::size_params) -//! take/return [`SketchAlgorithm`]/[`SketchParams`]) — `asap_sketch` also has +//! [`CostModel::rank_candidates`] is scoped to the approximate-**sketch** +//! family specifically (it takes [`SketchAlgorithm`]s) — `asap_sketch` also has //! sibling families for sampling-based, wavelet-transform, and fitted //! statistical-model summaries -//! ([`asap_types::post_asap::SamplingKind`]/…/[`asap_types::post_asap::StatModelKind`]), +//! ([`asap_types::ir::schema::SamplingKind`]/…/[`asap_types::ir::schema::StatModelKind`]), //! each with its own `(Kind, Params)` pair, deliberately *not* folded into -//! this trait: no core `AggIntent` picks one of those families today (only -//! [`CostModel::realize_extension`] can, for a deployment-specific -//! `AggIntent::Extension`), so there is no ranking/sizing decision for this -//! trait to own yet. Should a family other than `Sketch` ever need its own -//! `rank_candidates`/`size_params`, it gets its own trait methods rather -//! than overloading these ones across incompatible `Kind`/`Params` types. -//! -//! Every entry point that doesn't take an explicit `&dyn CostModel` -//! ([`SketchAlgorithmStrategy::default_cost_model`](crate::replacement::SketchAlgorithmStrategy::default_cost_model), -//! [`search_workload`](crate::replacement::search_workload)) runs against -//! [`DefaultCostModel`], so a deployment that never plugs in its own cost -//! model keeps today's static-preference-order behavior exactly, byte for -//! byte. +//! this trait: no core `AggIntent` picks one of those families today, so +//! there is no ranking decision for this trait to own yet. Should a family +//! other than `Sketch` ever need its own ranking, it gets its own trait +//! method rather than overloading this one across incompatible `Kind` types. //! //! ## CSE sharing (issue #237, #223 stage 4) //! //! [`CseCandidate`]/[`ShareDecision`]/[`CostModel::cse_share_decision`] below //! decide whether a CSE-detected shared sub-DAG -//! ([`asap_types::pre_asap::cse::share_common_sub_dags`], issue #223 stages +//! ([`asap_types::ir::cse::share_common_sub_dags`], issue #223 stages //! 1-2, PR #235) is actually worth sharing, via a real Volcano/Cascades-style //! cost comparison rather than a fixed rule. See //! `docs/design_docs/cse-cost-model-decision.md` for the full design discussion (why //! cost-based, why not a full plan-search engine, the layering constraint //! that forces detection to stay cost-agnostic). -//! [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted) -//! (via [`crate::replacement`]'s own `cse_preference`) and +//! [`cost_sorted`](crate::candidate_selection::cost_sorted) +//! (via [`asap_logical_optimizer::pass1::replacement`]'s own `cse_preference`) and //! [`DefaultCostModel::estimate_cost`] are this crate's own callers. use std::rc::Rc; -use asap_types::post_asap::{ - ExactOperation, FieldDataType, GroupingStrategy, HydraParams, ResultGuarantee, SketchAlgorithm, - SketchParams, SketchStatistic, SummaryExpr, SummaryMaintenanceLifecycleGuarantee, SummaryNode, - SummaryWindowFramework, +use asap_logical_optimizer::pass1::exact_composition::ExactOperation; +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::schema::{ + FieldDataType, GroupingStrategy, HydraParams, SketchAlgorithm, SketchParams, }; -use asap_types::pre_asap::agg_intent::AggIntent; -use asap_types::pre_asap::expr_ir::ColumnRef; -use asap_types::pre_asap::query_expr::QueryExpr; -use asap_types::types::AccuracyTarget; +use asap_types::ir::{ASAPOp, Operator, OperatorNode}; -use crate::exact_composition::{ExactComposition, OperationPlacement}; -use crate::recurrence::{ +use crate::cost::recurrence::{ self, CostRate, EvaluationRate, Horizon, RecurrenceCostExplanation, RecurrenceError, RecurrenceProfile, }; -use crate::replacement::{ - realize_child, Realization, Replacement, ReplacementProvenance, ReplacementSubDAG, TargetSubDAG, -}; -use crate::summary_maintenance_lifecycle::{ - SummaryMaintenanceCapabilities, SummaryMaintenanceLifecycleCostInputs, +use asap_logical_optimizer::pass1::exact_composition::{ExactComposition, OperationPlacement}; +use asap_logical_optimizer::pass1::replacement::{ + realize_child, Replacement, ReplacementProvenance, ReplacementSubDAG, TargetSubDAG, }; // ── Recurring-cost vocabulary for mixed exact/summary plans (issue #171) ── @@ -87,14 +70,14 @@ pub struct CostProvenance { } /// Which mixed-execution shapes the downstream runtime can actually -/// execute (issue #171). [`crate::exact_composition::ExactCompositionStrategy`] +/// execute (issue #171). [`asap_logical_optimizer::pass1::exact_composition::ExactCompositionStrategy`] /// proposes an `ValueOperationAtQueryTime` candidate only when /// `query_time` is set, and an `ValueOperationAtIngestionTime` candidate only /// when `ingestion_time` is — a runtime that cannot run an exact /// operator on the update path must never be handed one. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub struct ValueOperationCapabilities { - /// The runtime can apply an exact operator to summary readouts at + /// The runtime can apply an exact operator to summary evaluations at /// query evaluation time. pub query_time: bool, /// The runtime can apply an exact row transform on the update path, @@ -127,19 +110,19 @@ impl ValueOperationCapabilities { /// composes with. #[derive(Debug, Clone, Copy)] pub struct ExactCompositionCostRequest<'a> { - /// The pre-ASAP target the composed candidate replaces. - pub target: &'a QueryExpr, + /// The target the composed candidate replaces. + pub target: &'a OperatorNode, /// The composition itself — placement, operator, child target. pub composition: &'a ExactComposition, /// For [`OperationPlacement::Read`]: the child target's *selected* - /// summary readout candidate the exact operator consumes. For + /// summary evaluation candidate the exact operator consumes. For /// [`OperationPlacement::Maintenance`]: the maintained summary *above* the /// transform that consumes its output (the `SummaryAgg` this transform /// feeds). Either way, the summary whose maintenance/read cost the /// formula charges. - pub summary: &'a SummaryNode, + pub summary: &'a OperatorNode, /// How many times this site actually runs once ancestors' own choices - /// are accounted for (see `CandidateLogicalASAPDAGs::global_selection`). + /// are accounted for (see `candidate_selection::global_selection`). pub effective_consumer_count: usize, } @@ -147,11 +130,11 @@ pub struct ExactCompositionCostRequest<'a> { /// optional: **an unknown stays `None` — never a zero** — so a formula /// with a missing input yields no rate at all rather than a spuriously /// cheap one, and global selection then keeps the conservative -/// `KeepPreAsap` behavior. A deployment model that wants defaults supplies +/// keep-as-is behavior. A deployment model that wants defaults supplies /// them explicitly by overriding [`CostModel::exact_composition_cost_inputs`]. #[derive(Debug, Clone, PartialEq)] pub struct ExactCompositionCostInputs { - /// Exact operator cost per row it processes — per readout row for a + /// Exact operator cost per row it processes — per evaluation row for a /// read-time operation, per input row for an maintenance-time operation. pub exact_cost_per_row: Option, /// Rows the exact operator consumes per evaluation (read-time operation) or @@ -161,14 +144,14 @@ pub struct ExactCompositionCostInputs { pub expected_output_rows: Option, /// Cost of one update to the composed-with summary's maintained state. pub summary_maintenance_cost_per_update: Option, - /// Cost of one readout of that summary at evaluation time. + /// Cost of one evaluation of that summary at evaluation time. pub summary_read_cost: Option, /// Update (ingest) events per second reaching this site. pub update_rate: Option, /// Evaluations per second across every consumer of this site. pub evaluation_rate: Option, /// Cost of one full raw recompute of the target from pre-ASAP data — - /// the `KeepPreAsap` baseline's per-evaluation cost. + /// the kept-query baseline's per-evaluation cost. pub raw_recompute_cost: Option, /// Recurring formulas require `CostUnitsPerSecond`; totals yield no rate. pub unit: CostUnit, @@ -268,19 +251,19 @@ fn finite_rate(units_per_second: f64) -> Option { /// A CSE-detected, legality-gated shared sub-DAG with two or more consumers /// — the unit [`CostModel::cse_share_decision`] decides over. Built by -/// [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted) -/// (via [`crate::replacement`]'s own `cse_preference`) the first time it +/// [`cost_sorted`](crate::candidate_selection::cost_sorted) +/// (via [`asap_logical_optimizer::pass1::replacement`]'s own `cse_preference`) the first time it /// needs a representative bound node for a sub-DAG that -/// [`asap_types::pre_asap::cse::share_common_sub_dags`] already collapsed +/// [`asap_types::ir::cse::share_common_sub_dags`] already collapsed /// onto one `Rc` for two or more workload roots. See /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { - /// The shared pre-ASAP sub-DAG itself. - pub sub_dag: &'a QueryExpr, - /// The `SummaryNode` this sub-DAG bound to — gives the cost model the + /// The shared sub-DAG itself. + pub sub_dag: &'a Rc, + /// The node this sub-DAG bound to — gives the cost model the /// concrete `FieldDataType`/`(kind, params)` actually at stake, not - /// just the pre-ASAP shape. - pub bound_summary: &'a SummaryNode, + /// just the logical shape. + pub bound_summary: &'a OperatorNode, /// How many workload roots reference this exact shared sub-DAG, counted /// once up front over the whole workload (always >= 2 — a candidate is /// only ever constructed for an actually-shared sub-DAG). @@ -298,38 +281,6 @@ pub struct CseCandidate<'a> { #[derive(Debug, Clone, Copy, PartialEq, PartialOrd)] pub struct Cost(pub f64); -/// One physical summary state and the lifecycle selected for that exact DAG -/// node. Node identity is preserved so whole-DAG models can bind per-state -/// evidence without relying on traversal order. -pub struct CostedSummaryDeployment<'a> { - pub summary: &'a SummaryNode, - pub guarantee: &'a SummaryMaintenanceLifecycleGuarantee, - pub selected_cost: Cost, -} - -/// Complete candidate estimate returned to lifecycle and global plan search. -/// -/// A deployment-aware model may compare abstract summary-window primitives -/// using evidence supplied by downstream implementations. `window_frameworks` -/// is planner IR: it records the selected semantic realization contract. -/// `physical_plan_id` is separate provider-owned provenance for the concrete -/// implementation whose evidence won; it is not interpreted as planner IR. -#[derive(Debug, Clone, PartialEq)] -pub struct CompleteSummaryCandidateEstimate { - pub cost: Cost, - /// Stable provider identity of the complete implementation whose evidence - /// produced this estimate. - pub physical_plan_id: Option, - /// Window choice for each entry of the `deployments` slice passed to the - /// complete-cost hook. `None` explicitly means that deployment does not - /// use a summary-window framework. - pub window_frameworks: Vec>, - /// End-to-end guarantee supplied by the selected window realization. - /// `None` means that the complete model supplied no window-specific - /// guarantee; `Some` may be exact or approximate. - pub window_accuracy_guarantee: Option, -} - impl Cost { /// The cost of an operation that costs nothing at all. pub const ZERO: Cost = Cost(0.0); @@ -359,7 +310,7 @@ impl std::ops::Mul for Cost { /// [`CseCandidate`]. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ShareDecision { - /// Reuse one bound `SummaryNode` across every consumer. + /// Reuse one bound node across every consumer. Share, /// Bind each occurrence independently — the shared-maintenance cost /// isn't worth it for this candidate. @@ -367,23 +318,23 @@ pub enum ShareDecision { } /// Default [`CostModel::cse_recompute_cost`]: a structural-size proxy — the -/// number of *unique* nodes in `sub_dag`'s DAG -/// ([`asap_types::pre_asap::cse::dag_node_count`], the same module this +/// number of *unique* nodes in `sub-DAG`'s DAG +/// ([`asap_types::ir::cse::dag_node_count`], the same module this /// candidate's sharing was detected in). Deliberately **not** a raw -/// `serde_json` serialization length: after CSE, `sub_dag` generally has -/// internal sharing (a `CseCandidate` only exists because something got +/// `serde_json` serialization length: after CSE, `sub-DAG` is generally a +/// DAG, not a tree (a `CseCandidate` only exists because something got /// shared), and a naive full serialization re-serializes — over-counts — -/// any descendant `sub_dag` already shares internally, once per parent +/// any descendant `sub-DAG` already shares internally, once per parent /// that references it, instead of once for the whole DAG. `dag_node_count` /// dedupes by `Rc` pointer identity, so it charges each unique node's /// contribution exactly once regardless of how many places within -/// `sub_dag` reference it. Cheap to compute (one pass, no serialization), +/// `sub-DAG` reference it. Cheap to compute (one pass, no serialization), /// and still scales with real structural complexity — a genuinely tiny /// leaf costs little to recompute, a deep multi-join sub-DAG costs a lot. /// A deployment with real per-row/per-update cost knowledge should /// override [`CostModel::cse_recompute_cost`] instead of relying on this. -pub fn default_cse_recompute_cost(sub_dag: &QueryExpr) -> Cost { - Cost(asap_types::pre_asap::cse::dag_node_count(sub_dag) as f64) +pub fn default_cse_recompute_cost(sub_dag: &Rc) -> Cost { + Cost(asap_types::ir::cse::dag_node_count(sub_dag) as f64) } /// Default [`CostModel::cse_shared_maintenance_cost`]: a small @@ -411,15 +362,14 @@ pub fn default_cse_shared_maintenance_cost(family: &FieldDataType) -> Cost { Cost(weight * UNIT) } -/// Ranks the candidate sketch algorithms for one [`AggIntent`], best choice -/// first. +/// Selection-time preferences and costs over the candidates Stage 1 +/// generates. /// /// [`replacement::summary_candidates`] returns every algorithm that *can* answer an /// intent, in an arbitrary static preference order (issue #98's "one home" -/// for the candidate set). A `CostModel` re-orders that list under real, -/// deployment-specific cost knowledge this crate has no way to know about — -/// `replacement::realizations_for_intent` constructs every candidate in the -/// resulting order. +/// for the candidate set), and candidate generation keeps that order. A +/// `CostModel` re-orders the candidates when they are selected, under real, +/// deployment-specific cost knowledge this crate has no way to know about. pub trait CostModel { /// Whether [`Self::candidate_cost`] prices a complete physical /// alternative, including its raw baseline, rather than a local @@ -453,7 +403,7 @@ pub trait CostModel { } /// Rank `candidates` (as returned by - /// [`summary_candidates`](crate::replacement::summary_candidates)) for + /// [`summary_candidates`](asap_logical_optimizer::pass1::replacement::summary_candidates)) for /// `intent`, best choice first. /// /// Implementations MAY reorder freely, but MUST return exactly the input @@ -463,50 +413,17 @@ pub trait CostModel { /// [`ReplacementStrategy`]'s exhaustive, never-prune contract. This /// invariant is checked at every production call site. /// - /// [`ReplacementStrategy`]: crate::replacement::ReplacementStrategy + /// [`ReplacementStrategy`]: asap_logical_optimizer::pass1::replacement::ReplacementStrategy fn rank_candidates( &self, intent: &AggIntent, candidates: &[SketchAlgorithm], ) -> Vec; - /// Size [`SketchParams`] for `kind` (one of the candidates - /// [`rank_candidates`](Self::rank_candidates) put first) under the - /// resolved `(eps, delta)` accuracy budget. - /// - /// Splitting sizing out from candidate selection lets a deployment own - /// its own parameter-sizing math (e.g. an empirically-tuned table, or - /// discrete rungs required by a downstream catalog) without forking - /// `replacement::realizations_for_intent` — the same "one extension - /// point" rationale as `rank_candidates`, one level deeper. Default: - /// [`replacement::default_size_params`], `asap-plan`'s built-in formulas - /// (unchanged) — a deployment that only needs to reorder candidates, - /// not resize them, can leave this method unimplemented. - /// - /// # Contract - /// - /// The returned parameters MUST make `kind` satisfy the supplied - /// `(eps, delta)` accuracy budget. This is a semantic requirement, not a - /// requirement that parameter fields themselves be numerically monotonic: - /// catalog rungs and empirically tuned layouts are allowed, but returning - /// a configuration that misses the requested budget makes the resulting - /// plan invalid. Accuracy reconciliation relies on this same contract; - /// any implementation satisfying a tighter budget necessarily satisfies - /// a looser budget for the identical aggregate query. - fn size_params( - &self, - kind: SketchAlgorithm, - intent: &AggIntent, - eps: f64, - delta: f64, - ) -> SketchParams { - crate::replacement::default_size_params(kind, intent, eps, delta) - } - /// Estimated number of distinct subpopulations produced by `target`'s /// grouping keys. `None` means the deployment has no cardinality estimate; /// grouping alternatives remain legal but keep their discovery order. - fn estimated_subpopulation_count(&self, _target: &QueryExpr) -> Option { + fn estimated_subpopulation_count(&self, _target: &OperatorNode) -> Option { None } @@ -518,7 +435,7 @@ pub trait CostModel { candidate: &ReplacementSubDAG, target: &TargetSubDAG<'_>, ) -> Option { - let Replacement::Summary(node) = &candidate.replacement else { + let Replacement::SubDAG(node) = &candidate.replacement else { return None; }; let (kind, grouping) = sketch_state(node)?; @@ -534,40 +451,7 @@ pub trait CostModel { Some(Cost(units)) } - /// Realize an `AggIntent::Extension { ext_kind, payload }` — a - /// deployment-specific intent shape core has no realization opinion - /// for (issue #131). `replacement::realizations_for_intent` consults this - /// for every `Extension` node instead of hardcoding `PassThrough` - /// (issue #150). Default: `PassThrough` — preserves today's behavior - /// for every deployment that doesn't override this, exactly like - /// `size_params`'s default-delegates pattern above. - fn realize_extension(&self, _ext_kind: &str, _payload: &serde_json::Value) -> Realization { - Realization::PassThrough - } - - /// Build the `SummaryEstimate` readout for an `Extension` intent this - /// same `CostModel` realized as `Realization::Sketch` via - /// [`realize_extension`](Self::realize_extension). Only ever called - /// when `realize_extension` returned `Sketch` for the same - /// `(ext_kind, payload)` — `replacement::readout` has no other way to build a - /// `SketchStatistic` for a shape core doesn't know. A deployment that - /// overrides `realize_extension` to return `Sketch` for some - /// `ext_kind` MUST also override this for that same `ext_kind`, or - /// this default panics loudly (rather than silently misinterpreting - /// `payload`) the first time that intent is actually read out. - fn readout_extension( - &self, - ext_kind: &str, - _payload: &serde_json::Value, - _col: &ColumnRef, - ) -> SketchStatistic { - unimplemented!( - "CostModel::realize_extension returned Sketch for ext_kind={ext_kind:?} but \ - readout_extension wasn't overridden to match" - ) - } - - /// Estimate the one-time cost of recomputing `candidate.sub_dag` + /// Estimate the one-time cost of recomputing `candidate.sub-DAG` /// independently at a single use site. Default: /// [`default_cse_recompute_cost`] (a structural-size proxy). See /// `docs/design_docs/cse-cost-model-decision.md`. @@ -581,7 +465,7 @@ pub trait CostModel { /// weight table), applied to whichever field of /// `candidate.bound_summary`'s output schema actually carries summary /// state (falls back to the cheapest, `Plain`, weight if none does — - /// e.g. `bound_summary` is a passthrough `KeepPreAsap` node with nothing + /// e.g. `bound_summary` is a kept non-ASAP sub-DAG with nothing /// summary-shaped to maintain). See `docs/design_docs/cse-cost-model-decision.md`. fn cse_shared_maintenance_cost(&self, candidate: &CseCandidate) -> Cost { let family = candidate @@ -593,12 +477,12 @@ pub trait CostModel { .find(|dtype| !matches!(dtype, FieldDataType::Plain(_))) .cloned() .unwrap_or(FieldDataType::Plain( - asap_types::pre_asap::DataType::Float64, + asap_types::ir::schema::DataType::Float64, )); default_cse_shared_maintenance_cost(&family) } - /// Decide whether to reuse one shared `SummaryNode` across every + /// Decide whether to reuse one shared node across every /// consumer of `candidate`, or bind each occurrence independently — a /// Volcano/Cascades-style cost comparison (issue #237, #223 stage 4; see /// `docs/design_docs/cse-cost-model-decision.md`): share iff the estimated cost of @@ -622,7 +506,7 @@ pub trait CostModel { // ── Recurrence-aware costing (issue #287) ─────────────────────────── // - // See `crate::recurrence`'s module docs for the full cost model + // See `crate::cost::recurrence`'s module docs for the full cost model // (`maintained_cost_rate`/`recompute_cost_rate` formulas, units, // provenance of every new input). The three hooks below are the // per-update-event/per-read/per-recomputation cost primitives that @@ -633,7 +517,7 @@ pub trait CostModel { /// Cost of maintaining `candidate`'s bound summary for a single ingest /// update event. Units: cost units per update — the /// `maintenance_cost_per_update` term of `maintained_cost_rate` - /// (`crate::recurrence`), where it is multiplied by an `UpdateRate` in + /// (`crate::cost::recurrence`), where it is multiplied by an `UpdateRate` in /// **Hz** (`update_rate * maintenance_cost_per_update`). /// /// Default: a small nominal constant, `Cost(0.01)` — deliberately @@ -666,7 +550,7 @@ pub trait CostModel { Cost(1.0) } - /// Cost of recomputing `candidate.sub_dag` once, from the pre-ASAP/raw + /// Cost of recomputing `candidate.sub-DAG` once, from the pre-ASAP/raw /// path. Units: cost units per recomputation — the `raw_recompute_cost` /// term of `recompute_cost_rate`. Default: delegates to /// [`cse_recompute_cost`](Self::cse_recompute_cost) (the same @@ -703,7 +587,7 @@ pub trait CostModel { /// `Share`/`RecomputeIndependently` choice, weighted by how *often* /// `candidate`'s consumers actually run (`recurrence`) instead of only /// how many structurally exist (`candidate.consumer_count`). See - /// `crate::recurrence`'s module docs for the full design. + /// `crate::cost::recurrence`'s module docs for the full design. /// /// - `recurrence.is_empty()` (no [`RepeatingEntry`]/[`DataWorkload`]-derived /// metadata available): delegates to @@ -735,24 +619,24 @@ pub trait CostModel { /// [`ReplacementSubDAG`] candidate at `target` — a real `f64`, not just a /// relative rank, meant for a caller that wants to *display* "candidate A /// costs ≈ X, candidate B costs ≈ Y" (e.g. a DAG-visualization view built - /// on [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted)), + /// on [`cost_sorted`](crate::candidate_selection::cost_sorted)), /// not just order candidates against each other — that ordering job /// already belongs to [`rank_candidates`](Self::rank_candidates) (for a - /// [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) + /// [`ASAPStrategies`](asap_logical_optimizer::pass1::replacement::ASAPStrategies) /// group) and [`cse_share_decision`](Self::cse_share_decision) (for a - /// [`SharedSubDAGStrategy`](crate::replacement::SharedSubDAGStrategy) + /// [`SharedSubDAGStrategy`](asap_logical_optimizer::pass1::replacement::SharedSubDAGStrategy) /// group). /// /// One method covers both candidate shapes this crate ships: - /// `candidate.replacement`'s [`Replacement::Summary`] arm (a - /// `SketchAlgorithmStrategy` candidate — the bound `SummaryNode` is right - /// there, nothing to reconstruct) and its [`Replacement::Rewrite`] arm + /// `candidate.replacement`'s [`Replacement::SubDAG`] from a summary + /// realization (a `ASAPStrategies` candidate — the bound node is + /// right there, nothing to reconstruct) and the same arm from a rewrite /// (a `SharedSubDAGStrategy` share-vs-recompute candidate — no bound - /// `SummaryNode` of its own, since sharing is a decision about a target + /// summary of its own, since sharing is a decision about a target /// already bound some other way; a representative binding is recovered /// from `target` itself). `target` is threaded through explicitly /// (rather than only ever the target embedded in `candidate` — there - /// isn't one for a `Rewrite`) so both arms have the `consumer_count` + /// isn't one for a rewrite) so both arms have the `consumer_count` /// context a cost estimate needs to be meaningful. /// /// Default: **not a real cost model** — always returns `f64::NAN`. @@ -770,122 +654,16 @@ pub trait CostModel { f64::NAN } - /// Primitive build, update, read, retention, and retirement costs used to - /// compare physical summary-state lifecycles. Unknown values stay - /// unknown, preventing long-lived deployments from winning through - /// optimistic zeroes. - fn summary_maintenance_lifecycle_cost_inputs( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - SummaryMaintenanceLifecycleCostInputs::default() - } - - /// Horizon-aware form used by lifecycle planning. Models whose retention - /// objective is capacity rather than byte-seconds can normalize their - /// rate so the horizon integral equals one peak-capacity charge. - fn summary_maintenance_lifecycle_cost_inputs_for_horizon( - &self, - summary: &SummaryNode, - _horizon: Option, - ) -> SummaryMaintenanceLifecycleCostInputs { - self.summary_maintenance_lifecycle_cost_inputs(summary) - } - - /// Physical update/merge/delete support for one concrete summary. The - /// conservative default advertises no long-lived maintenance capability. - fn summary_maintenance_capabilities( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - SummaryMaintenanceCapabilities::default() - } - - /// Replace the sum of selected per-state lifecycle costs with a complete - /// root-DAG cost. The default preserves legacy models. Evidence-strict - /// models return `None` when any root operation is unavailable; callers - /// must not then reuse the partial per-state sum. - fn complete_summary_candidate_cost( - &self, - _root: &SummaryNode, - _target: Option<&QueryExpr>, - deployments: &[CostedSummaryDeployment<'_>], - _horizon: Option, - _expected_reads: Option, - _required_accuracy: &[AccuracyTarget], - ) -> Option { - Some(Cost( - deployments - .iter() - .map(|deployment| deployment.selected_cost.0) - .sum(), - )) - } - - /// Complete cost together with selected implementation provenance and - /// planner-visible window primitives. The default preserves cost models - /// that do not perform either decision. - fn complete_summary_candidate_estimate( - &self, - root: &SummaryNode, - target: Option<&QueryExpr>, - deployments: &[CostedSummaryDeployment<'_>], - horizon: Option, - expected_reads: Option, - required_accuracy: &[AccuracyTarget], - ) -> Option { - self.complete_summary_candidate_cost( - root, - target, - deployments, - horizon, - expected_reads, - required_accuracy, - ) - .map(|cost| CompleteSummaryCandidateEstimate { - cost, - physical_plan_id: None, - window_frameworks: vec![None; deployments.len()], - window_accuracy_guarantee: None, - }) - } - - /// Whether the complete-candidate hook is authoritative for lifecycle - /// costs. When true, lifecycle alternatives rejected only because their - /// legacy per-state cost is missing remain eligible for complete-DAG - /// evaluation. Semantic and runtime-capability rejections still apply. - fn complete_summary_candidate_estimate_covers_lifecycle_costs(&self) -> bool { - false - } - - /// Cost of evaluating `target` directly from its logical/raw inputs once. - /// When known, lifecycle-aware materialization compares this fallback with - /// the aggregate cost of the selected summary deployments. - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { - None - } - - /// Complete raw cost over the comparison context. The default preserves - /// per-read models; context-aware models override this when raw input - /// cardinality changes between evaluations. - fn raw_query_recompute_total_cost( - &self, - target: &QueryExpr, - expected_reads: f64, - ) -> Option { - self.raw_query_recompute_cost(target) - .map(|per_read| Cost(per_read.0 * expected_reads)) - } /// Physical feasibility evidence for a complete summary candidate. /// `None` defers admission to physical/deployment compilation; `Some(false)` /// excludes the candidate without changing its computation or parameters. - fn summary_support_evidence(&self, _summary: &SummaryNode) -> Option { + fn summary_support_evidence(&self, _summary: &OperatorNode) -> Option { None } /// Which mixed exact/summary execution shapes the downstream runtime /// advertises (issue #171). Gates candidate *generation* in - /// [`crate::exact_composition::ExactCompositionStrategy`]: a shape the + /// [`asap_logical_optimizer::pass1::exact_composition::ExactCompositionStrategy`]: a shape the /// runtime can't execute is never proposed, so it can't be selected /// either. /// @@ -932,7 +710,7 @@ pub trait CostModel { /// /// Default: every input unknown ([`ExactCompositionCostInputs::unknown`]) /// — unknown is never zero, and with no rate derivable - /// `CandidateLogicalASAPDAGs::global_selection` keeps the conservative `KeepPreAsap` + /// `candidate_selection::global_selection` keeps the conservative keep-as-is /// behavior for the site. A deployment that wants defaults must supply /// them here explicitly. fn exact_composition_cost_inputs( @@ -948,14 +726,16 @@ pub trait CostModel { } fn sketch_state( - node: &SummaryNode, -) -> Option<(&asap_types::post_asap::SketchKind, &GroupingStrategy)> { - match &node.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => sketch_state(summary_input), - SummaryExpr::SummaryAgg { + node: &OperatorNode, +) -> Option<(&asap_types::ir::schema::SketchKind, &GroupingStrategy)> { + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) => { + sketch_state(summary_input) + } + Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, grouping), .. - } => Some((kind, grouping)), + }) => Some((kind, grouping)), _ => None, } } @@ -1008,9 +788,9 @@ pub(crate) fn validated_candidate_ranking( } /// The default cost model: preserves [`summary_candidates`]'s built-in static -/// order and [`replacement::default_size_params`]'s built-in sizing unchanged. +/// order. /// -/// [`summary_candidates`]: crate::replacement::summary_candidates +/// [`summary_candidates`]: asap_logical_optimizer::pass1::replacement::summary_candidates pub struct DefaultCostModel; impl CostModel for DefaultCostModel { @@ -1027,15 +807,17 @@ impl CostModel for DefaultCostModel { /// `cse_share_decision`'s default body already composes — rather than a /// second formula: /// - /// - [`Replacement::Summary`]: `cse_recompute_cost` (the one-time + /// - A [`ReplacementProvenance::SummaryRealization`] candidate (a + /// `ASAPStrategies` binding): `cse_recompute_cost` (the one-time /// structural cost of building `target` at all) plus /// `cse_shared_maintenance_cost` of the candidate's own bound family /// (a pricier family — a sketch over an exact accumulator, say — /// costs more here, consistent with the per-family weighting /// [`default_cse_shared_maintenance_cost`] already orders candidates /// by). - /// - [`Replacement::Rewrite`]: recovers one representative bound - /// `SummaryNode` for `target` via `realize_child` (the same + /// - Any other [`Replacement::SubDAG`] (a logical rewrite or a CSE + /// share/recompute candidate): recovers one + /// representative bound node for `target` via `realize_child` (the same /// rank-and-take-first helper `replacement::realize_child` reuses for the /// identical need), then charges /// `cse_shared_maintenance_cost` for the candidate that shares @@ -1060,7 +842,9 @@ impl CostModel for DefaultCostModel { fn estimate_cost(&self, candidate: &ReplacementSubDAG, target: &TargetSubDAG<'_>) -> f64 { let consumer_count = target.consumer_count.max(1); match &candidate.replacement { - Replacement::Summary(node) => { + Replacement::SubDAG(node) + if candidate.provenance == ReplacementProvenance::SummaryRealization => + { let cse = CseCandidate { sub_dag: target.root, bound_summary: node, @@ -1068,10 +852,10 @@ impl CostModel for DefaultCostModel { }; (self.cse_recompute_cost(&cse) + self.cse_shared_maintenance_cost(&cse)).0 } - Replacement::Rewrite(rc) + Replacement::SubDAG(rc) if candidate.provenance == ReplacementProvenance::AccuracyReconciliation => { - let Ok(sibling_bound) = realize_child(rc, self) else { + let Ok(sibling_bound) = realize_child(rc) else { return f64::NAN; }; let cse = CseCandidate { @@ -1087,8 +871,8 @@ impl CostModel for DefaultCostModel { }; self.cse_shared_maintenance_cost(&cse).0 } - Replacement::Rewrite(rc) => { - let Ok(bound) = realize_child(target.root, self) else { + Replacement::SubDAG(rc) => { + let Ok(bound) = realize_child(target.root) else { return f64::NAN; }; let cse = CseCandidate { @@ -1103,7 +887,7 @@ impl CostModel for DefaultCostModel { } } // A composed candidate is costed in cost-units-per-second by - // `CandidateLogicalASAPDAGs::global_selection` against the child decision it + // `candidate_selection::global_selection` against the child decision it // is committed with — a different unit from this structural // estimate, and unknowable here without that child. `NaN` // keeps it from ever out-ranking a real estimate by accident. @@ -1115,8 +899,8 @@ impl CostModel for DefaultCostModel { #[cfg(test)] mod tests { use super::*; - use crate::replacement::summary_candidates; - use asap_types::pre_asap::agg_intent::default_cardinality; + use asap_logical_optimizer::pass1::replacement::summary_candidates; + use asap_types::ir::operator::agg_intent::default_cardinality; #[test] fn default_cost_model_preserves_static_order() { @@ -1192,66 +976,6 @@ mod tests { validated_candidate_ranking(&DuplicatesFirst, &intent, candidates); } - /// A deployment that only overrides `rank_candidates` keeps - /// `asap-plan`'s built-in sizing via the trait's default `size_params` - /// body — the split is opt-in per method, not all-or-nothing. - #[test] - fn size_params_default_body_matches_default_size_params() { - let intent = default_cardinality(); - assert_eq!( - AlwaysPreferLast.size_params(SketchAlgorithm::Hll, &intent, 0.01, 0.01), - crate::replacement::default_size_params(SketchAlgorithm::Hll, &intent, 0.01, 0.01), - ); - } - - /// A deployment CAN override `size_params` independently of - /// `rank_candidates` — e.g. to size against a catalog-constrained set - /// of discrete parameter rungs instead of `asap-plan`'s continuous - /// formulas. - struct DiscreteKllRungs; - - impl CostModel for DiscreteKllRungs { - fn rank_candidates( - &self, - _intent: &AggIntent, - candidates: &[SketchAlgorithm], - ) -> Vec { - candidates.to_vec() - } - - fn size_params( - &self, - kind: SketchAlgorithm, - intent: &AggIntent, - eps: f64, - delta: f64, - ) -> SketchParams { - match kind { - SketchAlgorithm::Kll => { - let k = if eps >= 0.01 { 200 } else { 2048 }; - SketchParams::Kll { k } - } - other => crate::replacement::default_size_params(other, intent, eps, delta), - } - } - } - - #[test] - fn custom_cost_model_can_override_sizing_independently_of_ranking() { - use asap_types::pre_asap::agg_intent::default_quantile; - - let intent = default_quantile(0.99); - assert_eq!( - DiscreteKllRungs.size_params(SketchAlgorithm::Kll, &intent, 0.001, 0.01), - SketchParams::Kll { k: 2048 }, - ); - // Untouched kinds still fall through to the default formula. - assert_eq!( - DiscreteKllRungs.size_params(SketchAlgorithm::Hll, &intent, 0.01, 0.01), - crate::replacement::default_size_params(SketchAlgorithm::Hll, &intent, 0.01, 0.01), - ); - } - // ── Recurring-cost formulas (issue #171) ───────────────────────────── fn known_inputs() -> ExactCompositionCostInputs { @@ -1282,7 +1006,7 @@ mod tests { // 2 * 100 assert_eq!(raw_recompute_cost_rate(&inputs).unwrap().0, 200.0); assert_eq!( - crate::recurrence::total_cost(CostRate(5.0), Horizon(10.0), Cost(3.0)), + crate::cost::recurrence::total_cost(CostRate(5.0), Horizon(10.0), Cost(3.0)), Cost(53.0) ); } @@ -1320,11 +1044,11 @@ mod tests { assert_eq!( DefaultCostModel.value_operation_support_evidence( &ExactOperation::Aggregate { - reduction: asap_types::pre_asap::query_expr::Reduction::by(vec![]), + reduction: asap_types::ir::operator::operator_properties::Reduction::by(vec![]), measures: vec![AggIntent::Max { col: None }], output_names: vec![], - having: None, filters: vec![], + having: None, }, OperationPlacement::Read, ), @@ -1346,14 +1070,15 @@ mod tests { // ── CSE sharing (issue #237, #223 stage 4) ────────────────────────── - use asap_types::post_asap::{ - ExactKind, ExactParams, Field, GroupingStrategy, Schema, SketchKind, SummaryExpr, + use asap_types::ir::operator::operator_properties::Source; + use asap_types::ir::schema::DataType; + use asap_types::ir::schema::{ + ExactKind, ExactParams, Field, GroupingStrategy, Schema, SketchKind, }; - use asap_types::pre_asap::query_expr::Source; - use asap_types::pre_asap::schema::DataType; + use asap_types::ir::{NonASAPOp, Predicate, ScalarExpr}; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -1364,37 +1089,39 @@ mod tests { 0, vec![], ), - } - } - - fn summary_node(family: FieldDataType) -> SummaryNode { - SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: std::rc::Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: Schema::lifted(vec![], None), - guarantee: None, + })) + .unwrap() + } + + /// A `SummaryAgg` directly over the kept `scan()` sub-DAG. + fn summary_node(family: FieldDataType) -> Rc { + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: scan(), + family: family.clone(), + input: asap_types::ir::schema::SummaryUpdate::column( + asap_types::ir::scalar::ColumnRef::Named("value".into()), + ), + reduction: asap_types::ir::operator::operator_properties::Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, }), - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column( - asap_types::pre_asap::expr_ir::ColumnRef::Named("value".into()), - ), - reduction: asap_types::pre_asap::query_expr::Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: None, - } + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(None), + ) } #[test] fn default_recompute_cost_is_positive_and_grows_with_structural_size() { let leaf = scan(); - let nested = QueryExpr::Dedup { - cols: vec![0], - child: std::rc::Rc::new(leaf.clone()), - }; + let nested = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: Rc::clone(&leaf), + })) + .unwrap(); assert!(default_cse_recompute_cost(&leaf) > Cost::ZERO); assert!(default_cse_recompute_cost(&nested) > default_cse_recompute_cost(&leaf)); } @@ -1407,27 +1134,27 @@ mod tests { /// identity-blind recursive walk) would count it. #[test] fn default_recompute_cost_does_not_double_count_an_internally_shared_descendant() { - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::{JoinKind, Predicate}; - - let true_pred = || { - Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( - true, - )))) - }; - let shared_leaf = std::rc::Rc::new(scan()); - let no_sharing = QueryExpr::Join { - kind: JoinKind::Inner, - pred: true_pred(), - left: std::rc::Rc::new(scan()), - right: std::rc::Rc::new(scan()), - }; - let with_sharing = QueryExpr::Join { - kind: JoinKind::Inner, - pred: true_pred(), - left: std::rc::Rc::clone(&shared_leaf), - right: std::rc::Rc::clone(&shared_leaf), - }; + use asap_types::ir::operator::operator_properties::JoinKind; + use asap_types::ir::scalar::ScalarValue; + + let true_pred = || Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))); + let shared_leaf = scan(); + let no_sharing = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left: scan(), + right: scan(), + })) + .unwrap(); + let with_sharing = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left: Rc::clone(&shared_leaf), + right: Rc::clone(&shared_leaf), + })) + .unwrap(); assert_eq!( default_cse_recompute_cost(&no_sharing), Cost(3.0), @@ -1482,8 +1209,8 @@ mod tests { let candidate = CseCandidate { sub_dag: &scan(), bound_summary: &summary_node(FieldDataType::StatModel( - asap_types::post_asap::StatModelKind::Parametric, - asap_types::post_asap::StatModelParams::Parametric { + asap_types::ir::schema::StatModelKind::Parametric, + asap_types::ir::schema::StatModelParams::Parametric { family: "gaussian_mixture".into(), }, )), @@ -1522,8 +1249,8 @@ mod tests { let candidate = CseCandidate { sub_dag: &scan(), bound_summary: &summary_node(FieldDataType::StatModel( - asap_types::post_asap::StatModelKind::Parametric, - asap_types::post_asap::StatModelParams::Parametric { + asap_types::ir::schema::StatModelKind::Parametric, + asap_types::ir::schema::StatModelParams::Parametric { family: "gaussian_mixture".into(), }, )), @@ -1555,20 +1282,20 @@ mod tests { } } - let root = Rc::new(scan()); + let root = scan(); let target = TargetSubDAG::new(&root); let candidate = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(FieldDataType::Plain( - asap_types::pre_asap::DataType::Float64, - )))), - provenance: crate::replacement::ReplacementProvenance::SummaryRealization, + replacement: Replacement::SubDAG(summary_node(FieldDataType::Plain( + asap_types::ir::schema::DataType::Float64, + ))), + provenance: asap_logical_optimizer::pass1::replacement::ReplacementProvenance::SummaryRealization, rationale: "whatever".into(), }; assert!(RankOnly.estimate_cost(&candidate, &target).is_nan()); } - /// `DefaultCostModel::estimate_cost` for a [`Replacement::Summary`] + /// `DefaultCostModel::estimate_cost` for a summary-rooted [`Replacement::SubDAG`] /// candidate reuses [`default_cse_shared_maintenance_cost`]'s own /// per-family ordering: a candidate bound to a cheap-to-maintain family /// (an exact accumulator) must cost less than one bound to an @@ -1578,26 +1305,27 @@ mod tests { /// above. #[test] fn estimate_cost_for_summary_orders_candidates_by_family_cheapest_to_priciest() { - let root = Rc::new(scan()); + let root = scan(); let target = TargetSubDAG::new(&root); let cheap = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node( - FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + replacement: Replacement::SubDAG(summary_node(FieldDataType::ExactAggregate( + ExactKind::Sum, + ExactParams::Sum, ))), - provenance: crate::replacement::ReplacementProvenance::SummaryRealization, + provenance: asap_logical_optimizer::pass1::replacement::ReplacementProvenance::SummaryRealization, rationale: "exact accumulator".into(), }; let pricey = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Summary(Rc::new(summary_node(FieldDataType::StatModel( - asap_types::post_asap::StatModelKind::Parametric, - asap_types::post_asap::StatModelParams::Parametric { + replacement: Replacement::SubDAG(summary_node(FieldDataType::StatModel( + asap_types::ir::schema::StatModelKind::Parametric, + asap_types::ir::schema::StatModelParams::Parametric { family: "gaussian_mixture".into(), }, - )))), - provenance: crate::replacement::ReplacementProvenance::SummaryRealization, + ))), + provenance: asap_logical_optimizer::pass1::replacement::ReplacementProvenance::SummaryRealization, rationale: "fitted statistical model".into(), }; @@ -1614,7 +1342,7 @@ mod tests { ); } - /// `DefaultCostModel::estimate_cost` for a [`Replacement::Rewrite`] pair + /// `DefaultCostModel::estimate_cost` for a relational [`Replacement::SubDAG`] pair /// (the `SharedSubDAGStrategy` share-vs-recompute shape) agrees with /// what `cse_share_decision` would already pick for the same target: with /// many consumers of a cheap-to-recompute leaf, the "share" candidate @@ -1625,19 +1353,20 @@ mod tests { /// directly. #[test] fn estimate_cost_for_rewrite_prefers_sharing_when_recompute_dominates_maintenance() { - let target_root = Rc::new(scan()); + let target_root = scan(); let target = TargetSubDAG::with_consumer_count(&target_root, 20); let share = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::clone(&target_root)), - provenance: crate::replacement::ReplacementProvenance::CseShare, + replacement: Replacement::SubDAG(Rc::clone(&target_root)), + provenance: asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseShare, rationale: "build once and share".into(), }; let recompute = ReplacementSubDAG { strategy: "TestStrategy", - replacement: Replacement::Rewrite(Rc::new((*target_root).clone())), - provenance: crate::replacement::ReplacementProvenance::CseRecompute, + replacement: Replacement::SubDAG(Rc::new((*target_root).clone())), + provenance: + asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseRecompute, rationale: "build independently".into(), }; diff --git a/crates/asap-aware-mapping/src/empirical_cost.rs b/crates/plan-selection/src/cost/empirical_cost.rs similarity index 77% rename from crates/asap-aware-mapping/src/empirical_cost.rs rename to crates/plan-selection/src/cost/empirical_cost.rs index b6d4f9a08..fc77bb85a 100644 --- a/crates/asap-aware-mapping/src/empirical_cost.rs +++ b/crates/plan-selection/src/cost/empirical_cost.rs @@ -2,23 +2,20 @@ //! configuration and environment; they are neither runtime feedback nor proofs //! of an accuracy guarantee. CPU quantities are nanoseconds, never CPU operations. -use asap_types::post_asap::{ - FieldDataType, GroupingStrategy, SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, -}; -use asap_types::pre_asap::AggIntent; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::schema::{SketchAlgorithm, SketchParams}; use serde::{Deserialize, Serialize}; -use crate::cost_model::{Cost, CostModel, DefaultCostModel}; -use crate::replacement::{ +use crate::cost::cost_model::{CostModel, DefaultCostModel}; +use asap_logical_optimizer::pass1::replacement::{ accuracy_budget, accuracy_target, default_size_params, ReplacementSubDAG, TargetSubDAG, }; -use crate::summary_maintenance_lifecycle::SummaryMaintenanceLifecycleCostInputs; pub const EVIDENCE_SCHEMA_VERSION: u32 = 1; pub const EVIDENCE_MODEL_VERSION: &str = "empirical-update-cpu-v1"; -pub use crate::empirical_resources::ResourceMeasurements; -pub use asap_types::resources::Measurement; +pub use crate::cost::empirical_resources::ResourceMeasurements; +pub use asap_types::workload::resources::Measurement; #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -208,40 +205,6 @@ impl EmpiricalEvidenceProvider { costs.sort_by(|a, b| a.1.total_cmp(&b.1)); costs.into_iter().map(|(algorithm, _)| algorithm).collect() } - - /// Costs for one independently instantiated sketch state, in CPU ns. - /// Unknown retention/retirement remain unavailable; CPU time must not be - /// mixed with an existing deployment's unitless or CPU-operation costs. - pub fn lifecycle_cost_inputs( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - let SummaryExpr::SummaryAgg { - family: FieldDataType::Sketch(kind, GroupingStrategy::PerSubpopulationInstance), - grouping: GroupingStrategy::PerSubpopulationInstance, - .. - } = &summary.expr - else { - return SummaryMaintenanceLifecycleCostInputs::default(); - }; - let Ok(row) = self.lookup(kind.algorithm(), kind.params()) else { - return SummaryMaintenanceLifecycleCostInputs::default(); - }; - SummaryMaintenanceLifecycleCostInputs { - build_cost: snapshot_build_cpu(row).map(Cost), - maintenance_cost_per_update: row - .metrics - .resources - .cpu - .update_cpu_ns - .as_ref() - .map(|m| Cost(m.value)), - // A point-frequency benchmark read does not price a total-count - // or quantile read. There is no query request in this hook. - summary_read_cost: None, - ..Default::default() - } - } } /// Standalone adapter for the existing planner boundary. Empirical data changes @@ -277,13 +240,6 @@ impl CostModel for EmpiricalCostModel { fn estimate_cost(&self, candidate: &ReplacementSubDAG, target: &TargetSubDAG<'_>) -> f64 { DefaultCostModel.estimate_cost(candidate, target) } - - fn summary_maintenance_lifecycle_cost_inputs( - &self, - summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - self.provider.lifecycle_cost_inputs(summary) - } } impl EvidenceArtifact { @@ -361,31 +317,6 @@ fn nonnegative(value: f64) -> bool { value.is_finite() && value >= 0.0 } -fn snapshot_build_cpu(row: &OfflineMeasurement) -> Option { - let cpu = row.metrics.resources.cpu.build_cpu_ns.as_ref()?.value - + row.metrics.resources.cpu.update_cpu_ns.as_ref()?.value - * row.distribution.sample_count as f64 - + snapshot_prepare_cpu(row)?; - nonnegative(cpu).then_some(cpu) -} - -/// The existing fixed-snapshot CMS/CountSketch contract needs no separate -/// preparation. Other families must measure that phase, including an explicit -/// zero when no preparation is necessary; absence is not free work. -pub(crate) fn snapshot_prepare_cpu(row: &OfflineMeasurement) -> Option { - match &row.metrics.resources.cpu.prepare_cpu_ns { - Some(measurement) => nonnegative(measurement.value).then_some(measurement.value), - None if matches!( - row.algorithm, - SketchAlgorithm::Cms | SketchAlgorithm::CountSketch - ) => - { - Some(0.0) - } - None => None, - } -} - fn validate_context( distribution: &DistributionDescriptor, environment: &EnvironmentDescriptor, @@ -461,7 +392,6 @@ fn valid_params(algorithm: &SketchAlgorithm, params: &SketchParams) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::replacement::{realizations_for_intent, Realization}; use asap_types::types::AccuracyTarget; /// The documented synthetic wire-format example remains importable and @@ -469,7 +399,7 @@ mod tests { #[test] fn checked_in_synthetic_example_is_valid() { let artifact: EvidenceArtifact = serde_json::from_str(include_str!( - "../tests/data/offline-evidence-synthetic.json" + "../../tests/data/offline-evidence-synthetic.json" )) .unwrap(); artifact.validate().unwrap(); @@ -478,7 +408,7 @@ mod tests { .implementation .contains("SYNTHETIC")); let schema: serde_json::Value = serde_json::from_str(include_str!( - "../../../docs/develop_docs/offline-sketch-evidence.schema.json" + "../../../../docs/develop_docs/offline-sketch-evidence.schema.json" )) .unwrap(); assert_eq!( @@ -491,30 +421,6 @@ mod tests { ); } - /// Lifecycle build includes all measured snapshot updates, not just an empty - /// allocation. A missing update measurement cannot become free ingestion. - #[test] - fn lifecycle_build_requires_complete_snapshot_ingestion() { - let (mut artifact, _, _) = fixture(); - let row = &mut artifact.records[0]; - row.metrics.resources.cpu.build_cpu_ns = Some(Measurement { - value: 10.0, - stddev: None, - samples: 1, - method: None, - }); - assert_eq!(snapshot_build_cpu(row), Some(20010.0)); - row.metrics.resources.cpu.prepare_cpu_ns = Some(Measurement { - value: 17.0, - stddev: None, - samples: 1, - method: None, - }); - assert_eq!(snapshot_build_cpu(row), Some(20027.0)); - row.metrics.resources.cpu.update_cpu_ns = None; - assert_eq!(snapshot_build_cpu(row), None); - } - /// Newly shared optional dimensions receive the same numeric validation. #[test] fn optional_prepare_and_scan_measurements_are_validated() { @@ -536,32 +442,6 @@ mod tests { } } - /// Only the established frequency-sketch contract can omit preparation. - #[test] - fn unmeasured_preparation_for_other_families_keeps_build_unknown() { - let (mut artifact, _, _) = fixture(); - let row = &mut artifact.records[0]; - row.metrics.resources.cpu.build_cpu_ns = Some(Measurement { - value: 10.0, - stddev: None, - samples: 1, - method: None, - }); - assert_eq!(snapshot_build_cpu(row), Some(20010.0)); - row.algorithm = SketchAlgorithm::CountSketch; - assert_eq!(snapshot_prepare_cpu(row), Some(0.0)); - row.algorithm = SketchAlgorithm::Kll; - row.params = SketchParams::Kll { k: 269 }; - assert_eq!(snapshot_build_cpu(row), None); - row.metrics.resources.cpu.prepare_cpu_ns = Some(Measurement { - value: 17.0, - stddev: None, - samples: 1, - method: None, - }); - assert_eq!(snapshot_build_cpu(row), Some(20027.0)); - } - fn fixture() -> (EvidenceArtifact, EvidenceContext, AggIntent) { let distribution = DistributionDescriptor { id: "unit-test-uniform".into(), @@ -605,8 +485,8 @@ mod tests { repetitions: 3, }, metrics: ResourceMeasurements { - resources: asap_types::resources::MeasuredResources { - cpu: asap_types::resources::MeasuredCpu { + resources: asap_types::workload::resources::MeasuredResources { + cpu: asap_types::workload::resources::MeasuredCpu { update_cpu_ns: Some(Measurement { value: cost, stddev: Some(1.0), @@ -644,33 +524,6 @@ mod tests { ) } - /// Real replacement generation follows measured update ranking while keeping - /// every candidate and the same formally sized parameter configurations. - #[test] - fn public_cost_model_changes_replacement_order_without_changing_guarantees() { - let (artifact, context, intent) = fixture(); - let model = - EmpiricalCostModel::new(EmpiricalEvidenceProvider::new(artifact, context).unwrap()); - let default = realizations_for_intent(&intent, &DefaultCostModel); - let measured = realizations_for_intent(&intent, &model); - assert_eq!(default.len(), measured.len()); - let Realization::Sketch(first_default) = &default[0] else { - panic!("expected sketch") - }; - let Realization::Sketch(first_measured) = &measured[0] else { - panic!("expected sketch") - }; - assert_eq!(first_default.algorithm(), &SketchAlgorithm::Cms); - assert_eq!(first_measured.algorithm(), &SketchAlgorithm::CountSketch); - for candidate in &measured { - assert!(default.contains(candidate)); - } - assert_eq!( - model.size_params(SketchAlgorithm::Cms, &intent, 0.001, 0.001), - DefaultCostModel.size_params(SketchAlgorithm::Cms, &intent, 0.001, 0.001) - ); - } - /// Missing, mismatched and expired evidence preserve the original ranking; /// a measurement from another configuration is never extrapolated. #[test] diff --git a/crates/asap-aware-mapping/src/empirical_resources.rs b/crates/plan-selection/src/cost/empirical_resources.rs similarity index 96% rename from crates/asap-aware-mapping/src/empirical_resources.rs rename to crates/plan-selection/src/cost/empirical_resources.rs index 716392f38..63bf9077f 100644 --- a/crates/asap-aware-mapping/src/empirical_resources.rs +++ b/crates/plan-selection/src/cost/empirical_resources.rs @@ -1,12 +1,12 @@ //! Measured resource payloads and compatibility with the v1 benchmark wire format. //! -//! Physical dimensions live in `asap_types::resources`; flat wire structs below +//! Physical dimensions live in `asap_types::workload::resources`; flat wire structs below //! exist only to keep archived artifacts readable and preserve their field names. -use asap_types::resources::PhysicalResources; +use asap_types::workload::resources::PhysicalResources; use serde::{Deserialize, Serialize}; -pub use asap_types::resources::{MeasuredCpu, MeasuredResources, Measurement}; +pub use asap_types::workload::resources::{MeasuredCpu, MeasuredResources, Measurement}; #[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)] #[serde(from = "SketchWire", into = "SketchWire")] diff --git a/crates/plan-selection/src/cost/mod.rs b/crates/plan-selection/src/cost/mod.rs new file mode 100644 index 000000000..f057ad296 --- /dev/null +++ b/crates/plan-selection/src/cost/mod.rs @@ -0,0 +1,18 @@ +//! Stage 3 pricing: the [`CostModel`](cost_model::CostModel) trait every +//! deployment's cost-based selection plugs into, with the built-in +//! [`DefaultCostModel`](cost_model::DefaultCostModel); recurring and one-shot +//! cost rates ([`recurrence`]); analytical and evidence-based pricing +//! ([`analytical_cost`], [`physical_plan_cost_model`], [`empirical_cost`]); and +//! the physical lowering and storage I/O profiles they price +//! ([`query_physical_lowering`], [`storage_io`]). + +pub mod analytical_cost; +pub mod cost_model; +pub mod empirical_cost; +pub mod empirical_resources; +pub mod physical_handoff_cost; +pub mod physical_operator_statistics; +pub mod physical_plan_cost_model; +pub mod query_physical_lowering; +pub mod recurrence; +pub mod storage_io; diff --git a/crates/asap-aware-mapping/src/physical_handoff_cost.rs b/crates/plan-selection/src/cost/physical_handoff_cost.rs similarity index 97% rename from crates/asap-aware-mapping/src/physical_handoff_cost.rs rename to crates/plan-selection/src/cost/physical_handoff_cost.rs index 207d812f1..a5a442aa5 100644 --- a/crates/asap-aware-mapping/src/physical_handoff_cost.rs +++ b/crates/plan-selection/src/cost/physical_handoff_cost.rs @@ -1,11 +1,11 @@ //! Byte estimates at deployment-declared physical handoffs. -use crate::analytical_cost::{ +use crate::cost::analytical_cost::{ estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDAG, ExecutionMultiplicity, PhysicalDAGNode, }; -use crate::physical_operator_statistics::{ComparisonScope, OperatorStatistics}; -pub use asap_types::resources::{PhysicalHandoffBytes, PhysicalHandoffKind}; +use crate::cost::physical_operator_statistics::{ComparisonScope, OperatorStatistics}; +pub use asap_types::workload::resources::{PhysicalHandoffBytes, PhysicalHandoffKind}; use serde::{Deserialize, Serialize}; use std::collections::{HashMap, HashSet}; diff --git a/crates/asap-aware-mapping/src/physical_operator_statistics.rs b/crates/plan-selection/src/cost/physical_operator_statistics.rs similarity index 97% rename from crates/asap-aware-mapping/src/physical_operator_statistics.rs rename to crates/plan-selection/src/cost/physical_operator_statistics.rs index 9cf8045bb..c63d77bf8 100644 --- a/crates/asap-aware-mapping/src/physical_operator_statistics.rs +++ b/crates/plan-selection/src/cost/physical_operator_statistics.rs @@ -6,14 +6,16 @@ use std::collections::HashMap; -use asap_types::pre_asap::query_expr::{InfoMatcher, Predicate, Source}; +use asap_types::ir::operator::operator_properties::{InfoMatcher, Source}; +use asap_types::ir::Predicate; + use asap_types::workload::{ DataArrival, DataWorkload, DurationMs, QueryRecurrence, QueryWorkloadEntry, RepeatedDemand, TimeSelection, TimestampMs, }; use serde::{Deserialize, Serialize}; -use crate::analytical_cost::AnalyticalCostError; +use crate::cost::analytical_cost::AnalyticalCostError; /// The semantic and workload boundary within which two resource estimates /// may be compared. Canonical workload and query-IR types remain authoritative; @@ -26,12 +28,12 @@ pub struct ComparisonScope { pub horizon: DurationMs, pub recurrence: QueryRecurrence, pub time_selection: TimeSelection, - pub sources: Vec, + pub sources: Vec, } /// Exact source selection covered by a physical plan. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct SourceCoverage { +pub struct ScanSelection { pub source: Source, /// Provider-owned stable identifier for the physical source contents, /// such as a catalog snapshot, table version, or object generation. @@ -53,7 +55,7 @@ impl ComparisonScope { query: &QueryWorkloadEntry, planning_time: TimestampMs, horizon: DurationMs, - sources: Vec, + sources: Vec, ) -> Result { let scope = Self { data_arrival: data.arrival, @@ -87,7 +89,7 @@ impl ComparisonScope { .any(|(index, source)| self.sources[..index].contains(source)) { return Err(AnalyticalCostError::MissingComparisonScope( - "duplicate source coverage", + "duplicate scan selection", )); } if self @@ -274,12 +276,12 @@ pub struct PartitionStatistics { } /// Workload-dependent evidence for one operator in an already-lowered -/// physical DAG. [`PhysicalOperator`](crate::analytical_cost::PhysicalOperator) +/// physical DAG. [`PhysicalOperator`](crate::cost::analytical_cost::PhysicalOperator) /// is the authoritative operator vocabulary: every one of its variants has a /// matching statistics variant here. /// -/// This enum intentionally does not mirror either logical IR. `QueryExpr` and -/// `SummaryExpr` are inputs to physical lowering, and one logical node may +/// This enum intentionally does not mirror the logical IR. `OperatorNode`s +/// are inputs to physical lowering, and one logical node may /// expand into several physical nodes or choose among several algorithms. /// Physical configuration such as a Top-K limit or hash-join build side lives /// on `PhysicalOperator`; this enum contains only workload/catalog evidence diff --git a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs b/crates/plan-selection/src/cost/physical_plan_cost_model.rs similarity index 88% rename from crates/asap-aware-mapping/src/physical_plan_cost_model.rs rename to crates/plan-selection/src/cost/physical_plan_cost_model.rs index 307fb9a61..5e1835f33 100644 --- a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs +++ b/crates/plan-selection/src/cost/physical_plan_cost_model.rs @@ -2,21 +2,22 @@ use std::{cell::RefCell, rc::Rc}; -use asap_types::post_asap::{SketchAlgorithm, SummaryExpr, SummaryNode}; -use asap_types::pre_asap::{AggIntent, QueryExpr}; -use asap_types::resources::CacheProfile; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::schema::SketchAlgorithm; +use asap_types::ir::OperatorNode; +use asap_types::workload::resources::CacheProfile; -use crate::analytical_cost::{ +use crate::cost::analytical_cost::{ estimate_physical_dag_comparison, AnalyticalCostError, EvidenceBackedPhysicalDAG as PhysicalDAG, PhysicalDAGComparisonEstimate, PhysicalDAGEstimateRequest, PhysicalNodeEvidence, ResourceCalibration, }; -use crate::cost_model::{Cost, CostModel, DefaultCostModel}; -use crate::physical_operator_statistics::ComparisonScope; -use crate::query_physical_lowering::{ +use crate::cost::cost_model::{Cost, CostModel, DefaultCostModel}; +use crate::cost::physical_operator_statistics::ComparisonScope; +use crate::cost::query_physical_lowering::{ lower_query_physical_dag, PhysicalNodeEvidenceProvider, PhysicalNodeRequest, }; -use crate::replacement::{Replacement, ReplacementSubDAG, TargetSubDAG}; +use asap_logical_optimizer::pass1::replacement::{Replacement, ReplacementSubDAG, TargetSubDAG}; /// One immutable generation of deployment evidence for a planner target. /// @@ -29,8 +30,8 @@ pub struct PhysicalEvidenceSnapshot { pub version: String, pub scope: ComparisonScope, pub cache_profile: CacheProfile, - pub storage_io: Option, - pub handoffs: Option, + pub storage_io: Option, + pub handoffs: Option, } /// Deployment evidence needed to price one planner alternative. @@ -40,7 +41,7 @@ pub struct PhysicalEvidenceSnapshot { /// operator. Post-ASAP summary operators need a physical plan provider because their /// implementation, placement, and retained-state layout are deployment /// choices; that provider must return the complete summary DAG, including any -/// embedded `KeepPreAsap` work. +/// non-ASAP work kept inside it. pub trait PlannerPhysicalPlanProvider { /// Atomically captures the comparison scope and evidence generation. fn capture_evidence_snapshot( @@ -57,7 +58,7 @@ pub trait PlannerPhysicalPlanProvider { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result; } @@ -69,12 +70,12 @@ pub struct PhysicalPlanComparison { pub raw_cost: Cost, pub candidate_cost: Cost, pub storage_io: Option<( - crate::storage_io::StorageEstimate, - crate::storage_io::StorageEstimate, + crate::cost::storage_io::StorageEstimate, + crate::cost::storage_io::StorageEstimate, )>, pub handoffs: Option<( - crate::physical_handoff_cost::PhysicalHandoffEstimate, - crate::physical_handoff_cost::PhysicalHandoffEstimate, + crate::cost::physical_handoff_cost::PhysicalHandoffEstimate, + crate::cost::physical_handoff_cost::PhysicalHandoffEstimate, )>, } @@ -91,7 +92,7 @@ pub struct PhysicalPlanCostModel<'a> { } struct CachedTargetEvidence { - root: Rc, + root: Rc, consumer_count: usize, snapshot: PhysicalEvidenceSnapshot, raw: PhysicalDAG, @@ -188,15 +189,14 @@ impl<'a> PhysicalPlanCostModel<'a> { Replacement::ExactComposition(_) => { return Err(AnalyticalCostError::UnsupportedCandidate) } - Replacement::Rewrite(query) => lower_query_physical_dag(query, scope, &evidence)?, - Replacement::Summary(summary) => match &summary.expr { - SummaryExpr::KeepPreAsap(query) => { - lower_query_physical_dag(query, scope, &evidence)? - } - _ => self - .provider - .summary_physical_dag(&snapshot, summary, target)?, - }, + // A sub-DAG without summary state is the planner's own query + // lowering; anything with summary state is deployment-provided. + Replacement::SubDAG(sub_dag) if !sub_dag.contains_asap() => { + lower_query_physical_dag(sub_dag, scope, &evidence)? + } + Replacement::SubDAG(summary) => self + .provider + .summary_physical_dag(&snapshot, summary, target)?, }; let resources = estimate_physical_dag_comparison( PhysicalDAGEstimateRequest { @@ -219,13 +219,13 @@ impl<'a> PhysicalPlanCostModel<'a> { .as_ref() .map(|profile| { Ok(( - crate::storage_io::estimate_storage_io( + crate::cost::storage_io::estimate_storage_io( &raw, scope, profile, &snapshot.version, )?, - crate::storage_io::estimate_storage_io( + crate::cost::storage_io::estimate_storage_io( &replacement, scope, profile, @@ -239,13 +239,13 @@ impl<'a> PhysicalPlanCostModel<'a> { .as_ref() .map(|profile| { Ok(( - crate::physical_handoff_cost::estimate_physical_handoffs( + crate::cost::physical_handoff_cost::estimate_physical_handoffs( &raw, scope, profile, &snapshot.version, )?, - crate::physical_handoff_cost::estimate_physical_handoffs( + crate::cost::physical_handoff_cost::estimate_physical_handoffs( &replacement, scope, profile, @@ -356,20 +356,23 @@ impl CostModel for PhysicalPlanCostModel<'_> { #[cfg(test)] mod tests { use super::*; + use crate::candidate_selection::global_selection; use std::cell::Cell; use std::collections::HashMap; - use asap_types::pre_asap::{DataType, Field, QueryExpr, Reduction, Schema, Source}; + use asap_types::ir::operator::{Reduction, Source}; + use asap_types::ir::schema::{DataType, Field, Schema}; + use asap_types::ir::{NonASAPOp, OperatorNode}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; - use crate::analytical_cost::{ExecutionMultiplicity, PhysicalDAGNode, PhysicalOperator}; - use crate::physical_operator_statistics::{ - EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, + use crate::cost::analytical_cost::{ExecutionMultiplicity, PhysicalDAGNode, PhysicalOperator}; + use crate::cost::physical_operator_statistics::{ + EdgeStatistics, OperatorStatistics, ScanSelection, UnaryEdgeStatistics, }; - use crate::replacement::ReplacementStrategy; + use asap_logical_optimizer::pass1::replacement::ReplacementStrategy; fn edge(rows: u64, bytes: u64) -> EdgeStatistics { EdgeStatistics { rows, bytes } @@ -405,8 +408,16 @@ mod tests { } } - fn query() -> Rc { - Rc::new(QueryExpr::Aggregate { + fn query() -> Rc { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "events".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + })) + .unwrap(); + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::Count { accuracy: AccuracyTarget::Epsilon(0.01), @@ -414,14 +425,9 @@ mod tests { output_names: vec![], filters: vec![], having: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "events".into(), - }, - predicates: vec![], - schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }), - }) + child: scan, + })) + .unwrap() } fn scope() -> ComparisonScope { @@ -438,7 +444,7 @@ mod tests { lookback: Some(DurationMs(10_000)), as_of: Some(TimestampMs(1_000)), }, - sources: vec![SourceCoverage { + sources: vec![ScanSelection { source: Source::Table { table_ref: "events".into(), }, @@ -450,10 +456,10 @@ mod tests { } struct TestProvider { - storage_io: Option, + storage_io: Option, summary_available: bool, candidate_scan_bytes: u64, - handoffs: Option, + handoffs: Option, snapshot_calls: Cell, raw_evidence_calls: Cell, } @@ -506,7 +512,7 @@ mod tests { id: "candidate-scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(scope.sources[0].clone()), + scan_selection: Some(scope.sources[0].clone()), output_buffer_bytes: 8, retained_bytes: 0, execution: ExecutionMultiplicity::Once, @@ -518,7 +524,7 @@ mod tests { accumulator_count: 1, }, children: vec!["candidate-scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 8, retained_bytes: 8, execution: ExecutionMultiplicity::Once, @@ -527,7 +533,7 @@ mod tests { id: "candidate-read".into(), operator: PhysicalOperator::PassThrough, children: vec!["candidate-state".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 8, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -583,7 +589,7 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, ) -> Result { assert_eq!(snapshot.version, "test-snapshot-1"); @@ -630,7 +636,7 @@ mod tests { // zero, or invalid supplemental calibration. #[test] fn storage_only_objective_requires_positive_valid_storage_calibration() { - use crate::storage_io::*; + use crate::cost::storage_io::*; let root = query(); let target = TargetSubDAG::new(&root); let mut provider = TestProvider::new(true, 800); @@ -686,8 +692,8 @@ mod tests { cost_per_retained_byte: 0.0, version: "unused-base-v1".into(), }; - let candidates = - crate::replacement::SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let candidates = asap_logical_optimizer::pass1::replacement::ASAPStrategies::default() + .replacements(&target); provider.storage_io = Some(profile.clone()); let model = PhysicalPlanCostModel::new(&provider, base.clone()).unwrap(); let estimate = model.estimate_candidate(&candidates[0], &target).unwrap(); @@ -717,8 +723,8 @@ mod tests { fn missing_storage_profile_remains_unestimated() { let root = query(); let target = TargetSubDAG::new(&root); - let candidates = - crate::replacement::SketchAlgorithmStrategy::default_cost_model().replacements(&target); + let candidates = asap_logical_optimizer::pass1::replacement::ASAPStrategies::default() + .replacements(&target); let provider = TestProvider::new(true, 800); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); let estimate = model.estimate_candidate(&candidates[0], &target).unwrap(); @@ -750,7 +756,7 @@ mod tests { // Explicit byte pricing can rank complete plans without pricing CPU or scans. #[test] fn handoff_only_objective_ranks_complete_plans() { - use crate::physical_handoff_cost::*; + use crate::cost::physical_handoff_cost::*; let root = query(); let target = TargetSubDAG::new(&root); let mut provider = TestProvider::new(true, 800); @@ -813,12 +819,12 @@ mod tests { cost_per_retained_byte: 0.0, version: "handoff-only-v1".into(), }; - let space = crate::replacement::search_workload_with( + let space = asap_logical_optimizer::pass1::replacement::search_workload_with( vec![("q", Rc::clone(&root))], - &crate::replacement::default_strategies(), + &asap_logical_optimizer::pass1::replacement::default_strategies(), ); let model = PhysicalPlanCostModel::new(&provider, zero_base.clone()).unwrap(); - let selected = space.global_selection(&model); + let selected = global_selection(&space, &model); assert!(selected .for_target(&space.roots[0].1) .unwrap() @@ -833,8 +839,7 @@ mod tests { .calibration .cost_per_network_byte = coefficient; let model = PhysicalPlanCostModel::new(&provider, zero_base.clone()).unwrap(); - assert!(space - .global_selection(&model) + assert!(global_selection(&space, &model) .for_target(&space.roots[0].1) .unwrap() .chosen @@ -842,8 +847,7 @@ mod tests { } provider.handoffs = None; let model = PhysicalPlanCostModel::new(&provider, zero_base).unwrap(); - assert!(space - .global_selection(&model) + assert!(global_selection(&space, &model) .for_target(&space.roots[0].1) .unwrap() .chosen @@ -853,15 +857,15 @@ mod tests { #[test] fn global_selection_uses_complete_physical_comparison() { let root = query(); - let space = crate::replacement::search_workload_with( + let space = asap_logical_optimizer::pass1::replacement::search_workload_with( vec![("q", Rc::clone(&root))], - &crate::replacement::default_strategies(), + &asap_logical_optimizer::pass1::replacement::default_strategies(), ); let planned_root = Rc::clone(&space.roots[0].1); let provider = TestProvider::new(true, 800); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); - let selected = space.global_selection(&model); + let selected = global_selection(&space, &model); assert!( selected.for_target(&planned_root).unwrap().chosen.is_some(), "a fully bound build-once summary cheaper than ten raw scans must be selected" @@ -871,7 +875,7 @@ mod tests { // Ranking must retain the uncovered byte of a nearly resident buffer cache. #[test] fn tiny_buffer_misses_still_affect_global_selection() { - use crate::analytical_cost::{CacheCapacityEvidence, CacheEvidence}; + use crate::cost::analytical_cost::{CacheCapacityEvidence, CacheEvidence}; const WORKING_SET: u64 = 1_u64 << 63; struct AlmostResident(TestProvider); impl PlannerPhysicalPlanProvider for AlmostResident { @@ -913,15 +917,15 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result { self.0.summary_physical_dag(snapshot, summary, target) } } - let space = crate::replacement::search_workload_with( + let space = asap_logical_optimizer::pass1::replacement::search_workload_with( vec![("q", query())], - &crate::replacement::default_strategies(), + &asap_logical_optimizer::pass1::replacement::default_strategies(), ); let provider = AlmostResident(TestProvider::new(true, WORKING_SET)); let model = PhysicalPlanCostModel::new( @@ -934,7 +938,7 @@ mod tests { }, ) .unwrap(); - let selected = space.global_selection(&model); + let selected = global_selection(&space, &model); assert!( selected .for_target(&space.roots[0].1) @@ -964,15 +968,15 @@ mod tests { #[test] fn missing_summary_evidence_keeps_the_raw_target() { let root = query(); - let space = crate::replacement::search_workload_with( + let space = asap_logical_optimizer::pass1::replacement::search_workload_with( vec![("q", Rc::clone(&root))], - &crate::replacement::default_strategies(), + &asap_logical_optimizer::pass1::replacement::default_strategies(), ); let planned_root = Rc::clone(&space.roots[0].1); let provider = TestProvider::new(false, 800); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); - let selected = space.global_selection(&model); + let selected = global_selection(&space, &model); assert!( selected.for_target(&planned_root).unwrap().chosen.is_none(), "missing physical summary evidence must not fall back to a structural estimate" @@ -1001,12 +1005,12 @@ mod tests { fn summary_physical_dag( &self, snapshot: &PhysicalEvidenceSnapshot, - summary: &Rc, + summary: &Rc, target: &TargetSubDAG<'_>, ) -> Result { let mut dag = self.0.summary_physical_dag(snapshot, summary, target)?; dag.nodes[0] - .source_coverage + .scan_selection .as_mut() .unwrap() .source_snapshot_id = "other".into(); @@ -1015,7 +1019,7 @@ mod tests { } let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = asap_logical_optimizer::pass1::replacement::ASAPStrategies::default() .replacements(&TargetSubDAG::new(&root)); let provider = WrongScope(TestProvider::new(true, 800)); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); @@ -1053,7 +1057,7 @@ mod tests { fn summary_physical_dag( &self, _snapshot: &PhysicalEvidenceSnapshot, - _summary: &Rc, + _summary: &Rc, _target: &TargetSubDAG<'_>, ) -> Result { panic!("blank snapshot versions must fail before summary binding") @@ -1061,7 +1065,7 @@ mod tests { } let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = asap_logical_optimizer::pass1::replacement::ASAPStrategies::default() .replacements(&TargetSubDAG::new(&root)); let model = PhysicalPlanCostModel::new(&BlankVersionProvider, calibration()).unwrap(); assert_eq!( @@ -1073,22 +1077,22 @@ mod tests { #[test] fn complete_candidate_that_costs_more_than_raw_is_not_selected() { let root = query(); - let space = crate::replacement::search_workload_with( + let space = asap_logical_optimizer::pass1::replacement::search_workload_with( vec![("q", Rc::clone(&root))], - &crate::replacement::default_strategies(), + &asap_logical_optimizer::pass1::replacement::default_strategies(), ); let planned_root = Rc::clone(&space.roots[0].1); let provider = TestProvider::new(true, 100_000); let model = PhysicalPlanCostModel::new(&provider, calibration()).unwrap(); - let selected = space.global_selection(&model); + let selected = global_selection(&space, &model); assert!(selected.for_target(&planned_root).unwrap().chosen.is_none()); } #[test] fn sibling_candidates_share_one_scope_and_raw_baseline() { let root = query(); - let candidates = crate::replacement::SketchAlgorithmStrategy::default_cost_model() + let candidates = asap_logical_optimizer::pass1::replacement::ASAPStrategies::default() .replacements(&TargetSubDAG::new(&root)); assert!(candidates.len() >= 2); let provider = TestProvider::new(true, 800); diff --git a/crates/asap-aware-mapping/src/query_physical_lowering.rs b/crates/plan-selection/src/cost/query_physical_lowering.rs similarity index 77% rename from crates/asap-aware-mapping/src/query_physical_lowering.rs rename to crates/plan-selection/src/cost/query_physical_lowering.rs index 1b8e5a751..096935978 100644 --- a/crates/asap-aware-mapping/src/query_physical_lowering.rs +++ b/crates/plan-selection/src/cost/query_physical_lowering.rs @@ -1,24 +1,26 @@ -//! Recursive lowering from the canonical query IR to evidenced physical DAGs. +//! Recursive lowering from the operator IR to evidenced physical DAGs. use std::rc::Rc; -use crate::analytical_cost::{ +use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; + +use crate::cost::analytical_cost::{ validate_operator_semantics, AnalyticalCostError, EvidenceBackedPhysicalDAG, ExecutionMultiplicity, HashJoinBuildSide, PhysicalDAGNode, PhysicalNodeEvidence, PhysicalOperator, PromqlBinaryOperandMode, PromqlBinaryOperation, PromqlPresenceKind, PromqlSeriesSampleKind, PromqlVectorCardinality, }; -use crate::physical_operator_statistics::{ - ComparisonScope, EdgeStatistics, OperatorStatistics, SourceCoverage, +use crate::cost::physical_operator_statistics::{ + ComparisonScope, EdgeStatistics, OperatorStatistics, ScanSelection, }; pub struct PhysicalNodeRequest<'a> { - pub logical_node: &'a asap_types::pre_asap::QueryExpr, + pub logical_node: &'a OperatorNode, pub operator: PhysicalOperator, pub occurrence: usize, pub synthetic: bool, pub children: &'a [String], - pub source_coverage: Option<&'a SourceCoverage>, + pub scan_selection: Option<&'a ScanSelection>, } pub trait PhysicalNodeEvidenceProvider { @@ -40,19 +42,19 @@ where } } -/// Lower a resolved query operator DAG to the physical operators understood by +/// Lower a non-ASAP operator DAG to the physical operators understood by /// this cost model. The authoritative provider supplies statistics by the /// stable physical IDs owned by that provider; missing evidence makes the /// complete query unavailable. Scalar expressions remain part of their -/// containing operator's local cost. +/// containing operator's local cost. An ASAP node is unsupported here. pub fn lower_query_physical_dag( - root: &Rc, + root: &Rc, scope: &ComparisonScope, evidence: &dyn PhysicalNodeEvidenceProvider, ) -> Result { use std::collections::HashMap; - use asap_types::pre_asap::{GroupKeys, QueryExpr, RelationalSetOpKind}; + use asap_types::ir::operator::{GroupKeys, RelationalSetOpKind}; scope.validate()?; @@ -65,7 +67,7 @@ pub fn lower_query_physical_dag( } impl Lowerer<'_> { - fn lower(&mut self, query: &QueryExpr) -> Result { + fn lower(&mut self, query: &OperatorNode) -> Result { let occurrence = self.next_id; self.next_id += 1; self.lower_new(query, occurrence) @@ -73,12 +75,12 @@ pub fn lower_query_physical_dag( fn resolve( &self, - query: &QueryExpr, + query: &OperatorNode, operator: PhysicalOperator, occurrence: usize, synthetic: bool, children: &[String], - source_coverage: Option<&SourceCoverage>, + scan_selection: Option<&ScanSelection>, ) -> Result { let evidence = self.provider.evidence(PhysicalNodeRequest { logical_node: query, @@ -86,7 +88,7 @@ pub fn lower_query_physical_dag( occurrence, synthetic, children, - source_coverage, + scan_selection, })?; if evidence.physical_id.is_empty() { return Err(AnalyticalCostError::InvalidPhysicalDAG( @@ -101,14 +103,14 @@ pub fn lower_query_physical_dag( evidence: PhysicalNodeEvidence, operator: PhysicalOperator, children: Vec, - source_coverage: Option, + scan_selection: Option, ) -> Result { let id = evidence.physical_id.clone(); let node = PhysicalDAGNode { id: id.clone(), operator, children, - source_coverage, + scan_selection, output_buffer_bytes: evidence.output_buffer_bytes, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -128,12 +130,23 @@ pub fn lower_query_physical_dag( fn lower_unary( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &OperatorNode, ) -> Result { let child_id = self.lower(child)?; + self.push_unary(query, occurrence, operator, child_id) + } + + /// `operator` over an already-lowered child. + fn push_unary( + &mut self, + query: &OperatorNode, + occurrence: usize, + operator: PhysicalOperator, + child_id: String, + ) -> Result { let children = vec![child_id.clone()]; let evidence = self.resolve(query, operator, occurrence, false, &children, None)?; let statistics = &evidence.statistics; @@ -151,17 +164,17 @@ pub fn lower_query_physical_dag( fn lower_promql_unary( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, operator: PhysicalOperator, - child: &QueryExpr, + child: &OperatorNode, ) -> Result { self.lower_unary(query, occurrence, operator, child) } fn lower_promql_scalar_leaf( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, ) -> Result { let operator = PhysicalOperator::PromqlScalarLeaf; @@ -171,6 +184,28 @@ pub fn lower_query_physical_dag( self.push(evidence, operator, vec![], None) } + /// Lower the owned scalar operand of `vector(s)`. A literal or + /// `time()` is a physical scalar leaf; `scalar(v)` reads its vector + /// through `PromqlVectorToScalar`. + fn lower_scalar_operand( + &mut self, + query: &OperatorNode, + occurrence: usize, + expr: &ScalarExpr, + ) -> Result { + match expr { + ScalarExpr::Literal(asap_types::ir::scalar::ScalarValue::Float64(_)) + | ScalarExpr::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), + ScalarExpr::PromqlScalarFromVector(vector) => self.lower_promql_unary( + query, + occurrence, + PhysicalOperator::PromqlVectorToScalar, + vector, + ), + _ => Err(AnalyticalCostError::UnsupportedQueryOperator), + } + } + fn node_statistics(&self, id: &str) -> Result<&OperatorStatistics, AnalyticalCostError> { self.evidence .get(id) @@ -182,11 +217,14 @@ pub fn lower_query_physical_dag( fn lower_new( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, ) -> Result { - match query { - QueryExpr::Scan { + let Some(op) = query.non_asap() else { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + }; + match op { + NonASAPOp::Scan { source, predicates, .. } => { let coverage = bind_scan_coverage( @@ -252,13 +290,13 @@ pub fn lower_query_physical_dag( require_operator_statistics(filter_operator, &filter_evidence.statistics)?; self.push(filter_evidence, filter_operator, children, None) } - QueryExpr::Filter { pred, child } => { + NonASAPOp::Filter { pred, child } => { let operator = PhysicalOperator::Filter { predicate_operations_per_row: scalar_operation_count(&pred.0)?.max(1), }; self.lower_unary(query, occurrence, operator, child) } - QueryExpr::Project { cols, child, .. } => { + NonASAPOp::Project { cols, child, .. } => { let expression_operations_per_row = cols .iter() .try_fold(0_u64, |total, item| { @@ -280,7 +318,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Aggregate { + NonASAPOp::Aggregate { reduction, measures, filters, @@ -289,12 +327,12 @@ pub fn lower_query_physical_dag( .. } => { if having.is_some() - || asap_types::pre_asap::any_measure_filtered(filters) + || asap_types::ir::operator::non_asap::any_measure_filtered(filters) || measures.is_empty() { return Err(AnalyticalCostError::UnsupportedQueryOperator); } - if matches!(reduction, asap_types::pre_asap::Reduction::PerEntity) { + if matches!(reduction, asap_types::ir::operator::Reduction::PerEntity) { if measures.len() != 1 { return Err(AnalyticalCostError::UnsupportedQueryOperator); } @@ -320,7 +358,7 @@ pub fn lower_query_physical_dag( }; return self.lower_promql_unary(query, occurrence, operator, child); } - let asap_types::pre_asap::Reduction::Reduce(grouping) = reduction else { + let asap_types::ir::operator::Reduction::Reduce(grouping) = reduction else { unreachable!("per-entity reduction returned above") }; if grouping.is_without() || !supports_hash_aggregate(reduction, measures) { @@ -338,13 +376,9 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Dedup { cols, child } => { + NonASAPOp::Dedup { cols, child } => { let key_count = if cols.is_empty() { - child - .output_schema() - .map_err(|_| AnalyticalCostError::UnsupportedQueryOperator)? - .fields - .len() + child.schema.fields.len() } else { cols.len() }; @@ -361,7 +395,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Sort { + NonASAPOp::Sort { keys, partition_by, child, @@ -381,13 +415,26 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::Limit { n, offset, child } => { - if let QueryExpr::Sort { + NonASAPOp::Limit { + n, + offset, + partition_by: limit_partition_by, + child, + } => { + // Offset-only and per-group limits have no physical + // operator here. + let Some(n) = n else { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + }; + if limit_partition_by != &GroupKeys::none() { + return Err(AnalyticalCostError::UnsupportedQueryOperator); + } + if let Some(NonASAPOp::Sort { keys, partition_by, child: sorted_child, .. - } = child.as_ref() + }) = child.non_asap() { if !keys.is_empty() && partition_by == &GroupKeys::none() { let child_id = self.lower(sorted_child)?; @@ -442,7 +489,7 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::SQLWindowFunc { + NonASAPOp::SQLWindowFunc { func, partition_by, order_by, @@ -452,9 +499,9 @@ pub fn lower_query_physical_dag( if order_by.is_empty() || !matches!( func, - asap_types::pre_asap::WindowFuncKind::RowNumber - | asap_types::pre_asap::WindowFuncKind::Rank - | asap_types::pre_asap::WindowFuncKind::DenseRank + asap_types::ir::operator::WindowFuncKind::RowNumber + | asap_types::ir::operator::WindowFuncKind::Rank + | asap_types::ir::operator::WindowFuncKind::DenseRank ) { return Err(AnalyticalCostError::UnsupportedQueryOperator); @@ -472,7 +519,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::TimeRange { range, child } => { + NonASAPOp::TimeRange { range, child, .. } => { let range_millis = duration_millis(*range, "range")?; self.lower_promql_unary( query, @@ -481,7 +528,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlSubquery { + NonASAPOp::PromqlSubquery { range, resolution, child, @@ -513,7 +560,7 @@ pub fn lower_query_physical_dag( } Ok(id) } - QueryExpr::PromqlRelabel { value, child, .. } => self.lower_promql_unary( + NonASAPOp::PromqlRelabel { value, child, .. } => self.lower_promql_unary( query, occurrence, PhysicalOperator::PromqlRelabel { @@ -521,28 +568,28 @@ pub fn lower_query_physical_dag( }, child, ), - QueryExpr::PromqlSeriesSample { + NonASAPOp::PromqlSeriesSample { by, kind, child, .. } => { if by.is_without() { return Err(AnalyticalCostError::UnsupportedQueryOperator); } let kind = match kind { - asap_types::pre_asap::SampleKind::LimitK(k) => { + asap_types::ir::operator::SampleKind::LimitK(k) => { let k = u64::try_from(*k).map_err(|_| AnalyticalCostError::Overflow)?; if k == 0 { return Err(AnalyticalCostError::MissingOrZero("limitk")); } PromqlSeriesSampleKind::LimitK { k } } - asap_types::pre_asap::SampleKind::LimitRatio(ratio) + asap_types::ir::operator::SampleKind::LimitRatio(ratio) if ratio.is_finite() && (-1.0..=1.0).contains(ratio) => { PromqlSeriesSampleKind::LimitRatio { ratio_bits: ratio.to_bits(), } } - asap_types::pre_asap::SampleKind::LimitRatio(_) => { + asap_types::ir::operator::SampleKind::LimitRatio(_) => { return Err(AnalyticalCostError::UnsupportedQueryOperator) } }; @@ -557,7 +604,7 @@ pub fn lower_query_physical_dag( child, ) } - QueryExpr::PromqlInfoEnrich { selector, child } => { + NonASAPOp::PromqlInfoEnrich { selector, child } => { let left_id = self.lower(child)?; let coverage = bind_info_coverage( &format!("occurrence-{occurrence}-info"), @@ -609,34 +656,13 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, + NonASAPOp::BinaryOp { + operator, lhs, rhs, .. } => { - let left_scalar = is_promql_scalar(lhs); - let right_scalar = is_promql_scalar(rhs); - if left_scalar && right_scalar { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } + let op = &operator.kind; + let vector_match = &operator.vector_match; let operation = promql_binary_operation(op); - if (left_scalar || right_scalar) - && !matches!(operation, PromqlBinaryOperation::ArithmeticOrComparison) - { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } - let operand_mode = match (left_scalar, right_scalar) { - (false, false) => PromqlBinaryOperandMode::VectorVector, - (false, true) => PromqlBinaryOperandMode::VectorScalar, - (true, false) => PromqlBinaryOperandMode::ScalarVector, - (true, true) => unreachable!("scalar/scalar returned above"), - }; - if operand_mode != PromqlBinaryOperandMode::VectorVector - && vector_match.is_some() - { - return Err(AnalyticalCostError::UnsupportedQueryOperator); - } + let operand_mode = PromqlBinaryOperandMode::VectorVector; let cardinality = promql_vector_cardinality(vector_match.as_ref()); let left_id = self.lower(lhs)?; let right_id = self.lower(rhs)?; @@ -670,41 +696,31 @@ pub fn lower_query_physical_dag( require_operator_statistics(operator, &evidence.statistics)?; self.push(evidence, operator, children, None) } - QueryExpr::PromqlVectorFromScalar(child) => self.lower_promql_unary( - query, - occurrence, - PhysicalOperator::PromqlScalarToVector, - child, - ), - QueryExpr::PromqlScalarFromVector(child) => self.lower_promql_unary( - query, - occurrence, - PhysicalOperator::PromqlVectorToScalar, - child, - ), - QueryExpr::PromqlScalarBridge(inner) - if matches!( - inner.as_ref(), - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Float64(_)) - ) => - { - self.lower_promql_scalar_leaf(query, occurrence) + NonASAPOp::PromqlVectorFromScalar(scalar) => { + let scalar_occurrence = self.next_id; + self.next_id += 1; + let child_id = self.lower_scalar_operand(query, scalar_occurrence, scalar)?; + self.push_unary( + query, + occurrence, + PhysicalOperator::PromqlScalarToVector, + child_id, + ) } - QueryExpr::EvalTimestamp => self.lower_promql_scalar_leaf(query, occurrence), - QueryExpr::TimeShift { shift, child } => { + NonASAPOp::TimeShift { shift, child } => { if !shift.is_identity() { return Err(AnalyticalCostError::UnsupportedQueryOperator); } self.lower_unary(query, occurrence, PhysicalOperator::PassThrough, child) } - QueryExpr::Concat { children, .. } => { + NonASAPOp::Concat { children, .. } => { let child_ids = children .iter() .map(|child| self.lower(child)) .collect::, _>>()?; self.lower_concat(query, occurrence, child_ids) } - QueryExpr::SetOp { + NonASAPOp::SetOp { kind: RelationalSetOpKind::Union, all: true, left, @@ -714,14 +730,14 @@ pub fn lower_query_physical_dag( let right_id = self.lower(right)?; self.lower_concat(query, occurrence, vec![left_id, right_id]) } - QueryExpr::Join { + NonASAPOp::Join { kind, pred, left, right, } => { let equality_key_count = - if matches!(kind, asap_types::pre_asap::JoinKind::Cross) { + if matches!(kind, asap_types::ir::operator::JoinKind::Cross) { None } else { hash_join_key_count(&pred.0, left, right) @@ -764,7 +780,7 @@ pub fn lower_query_physical_dag( fn lower_concat( &mut self, - query: &QueryExpr, + query: &OperatorNode, occurrence: usize, child_ids: Vec, ) -> Result { @@ -837,7 +853,7 @@ fn validate_source_consumption( let consumed = nodes .iter() .filter(|node| matches!(node.operator, PhysicalOperator::Scan)) - .filter_map(|node| node.source_coverage.as_ref()) + .filter_map(|node| node.scan_selection.as_ref()) .collect::>(); for coverage in &consumed { if !scope.sources.contains(coverage) { @@ -956,10 +972,10 @@ fn require_operator_statistics( fn bind_scan_coverage( node_id: &str, - source: &asap_types::pre_asap::Source, - predicates: &[asap_types::pre_asap::Predicate], + source: &asap_types::ir::operator::Source, + predicates: &[asap_types::ir::Predicate], scope: &ComparisonScope, -) -> Result { +) -> Result { let mut matches = scope.sources.iter().filter(|coverage| { coverage.source == *source && coverage.predicates == predicates @@ -971,7 +987,7 @@ fn bind_scan_coverage( .ok_or_else(|| AnalyticalCostError::ScanOutsideComparisonScope(node_id.into()))?; if matches.any(|candidate| candidate != &coverage) { return Err(AnalyticalCostError::InvalidPhysicalDAG( - "scan source coverage is ambiguous", + "scan selection is ambiguous", )); } Ok(coverage) @@ -979,10 +995,11 @@ fn bind_scan_coverage( fn bind_info_coverage( node_id: &str, - selector: &[asap_types::pre_asap::InfoMatcher], + selector: &[asap_types::ir::operator::InfoMatcher], scope: &ComparisonScope, -) -> Result { - use asap_types::pre_asap::{CompareOpKind, Source}; +) -> Result { + use asap_types::ir::operator::Source; + use asap_types::ir::scalar::CompareOpKind; let mut metric: Option<&str> = None; for matcher in selector @@ -1009,7 +1026,7 @@ fn bind_info_coverage( .ok_or_else(|| AnalyticalCostError::ScanOutsideComparisonScope(node_id.into()))?; if matches.next().is_some() { return Err(AnalyticalCostError::InvalidPhysicalDAG( - "info source coverage is ambiguous", + "info scan selection is ambiguous", )); } Ok(coverage) @@ -1026,9 +1043,9 @@ fn duration_millis( } fn promql_binary_operation( - operation: &asap_types::pre_asap::BinaryOpKind, + operation: &asap_types::ir::operator::BinaryOpKind, ) -> PromqlBinaryOperation { - use asap_types::pre_asap::{BinaryOpKind, PromQLVectorSetOpKind}; + use asap_types::ir::operator::{BinaryOpKind, PromQLVectorSetOpKind}; match operation { BinaryOpKind::Set(PromQLVectorSetOpKind::And) => PromqlBinaryOperation::And, BinaryOpKind::Set(PromQLVectorSetOpKind::Or) => PromqlBinaryOperation::Or, @@ -1040,9 +1057,9 @@ fn promql_binary_operation( } fn promql_vector_cardinality( - vector_match: Option<&asap_types::pre_asap::VectorMatch>, + vector_match: Option<&asap_types::ir::operator::VectorMatch>, ) -> PromqlVectorCardinality { - use asap_types::pre_asap::GroupSide; + use asap_types::ir::operator::GroupSide; match vector_match.and_then(|matching| matching.grouping.as_ref()) { Some(grouping) if grouping.side == GroupSide::Left => PromqlVectorCardinality::ManyToOne, Some(_) => PromqlVectorCardinality::OneToMany, @@ -1051,17 +1068,14 @@ fn promql_vector_cardinality( } fn hash_join_key_count( - expr: &asap_types::pre_asap::QueryExpr, - left: &asap_types::pre_asap::QueryExpr, - right: &asap_types::pre_asap::QueryExpr, + expr: &ScalarExpr, + left: &OperatorNode, + right: &OperatorNode, ) -> Option { - use asap_types::pre_asap::{CompareOpKind, QueryExpr}; + use asap_types::ir::scalar::CompareOpKind; - let (Ok(left_schema), Ok(right_schema)) = (left.output_schema(), right.output_schema()) else { - return None; - }; - let left_width = left_schema.fields.len(); - let total_width = left_width.saturating_add(right_schema.fields.len()); + let left_width = left.schema.fields.len(); + let total_width = left_width.saturating_add(right.schema.fields.len()); fn column_side(column: usize, left_width: usize, total_width: usize) -> Option { if column < left_width { @@ -1073,14 +1087,15 @@ fn hash_join_key_count( } } - fn predicate(expr: &QueryExpr, left_width: usize, total_width: usize) -> Option { + fn predicate(expr: &ScalarExpr, left_width: usize, total_width: usize) -> Option { match expr { - QueryExpr::Compare { + ScalarExpr::Compare { left, op: CompareOpKind::Eq, right, + .. } => match (left.as_ref(), right.as_ref()) { - (QueryExpr::Column(left), QueryExpr::Column(right)) => match ( + (ScalarExpr::Column(left), ScalarExpr::Column(right)) => match ( column_side(*left, left_width, total_width), column_side(*right, left_width, total_width), ) { @@ -1089,7 +1104,7 @@ fn hash_join_key_count( }, _ => None, }, - QueryExpr::BoolAnd(parts) if !parts.is_empty() => { + ScalarExpr::BoolAnd(parts) if !parts.is_empty() => { parts.iter().try_fold(0_u64, |count, part| { count.checked_add(predicate(part, left_width, total_width)?) }) @@ -1101,12 +1116,8 @@ fn hash_join_key_count( predicate(expr, left_width, total_width) } -fn scalar_operation_count( - expr: &asap_types::pre_asap::QueryExpr, -) -> Result { - use asap_types::pre_asap::QueryExpr; - - let add = |parts: &[&QueryExpr]| { +fn scalar_operation_count(expr: &ScalarExpr) -> Result { + let add = |parts: &[&ScalarExpr]| { parts.iter().try_fold(0_u64, |total, part| { total .checked_add(scalar_operation_count(part)?) @@ -1119,14 +1130,14 @@ fn scalar_operation_count( .ok_or(AnalyticalCostError::Overflow) }; match expr { - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::EvalTimestamp - | QueryExpr::CurrentTimestamp => Ok(0), - QueryExpr::Compare { left, right, .. } | QueryExpr::Arithmetic { left, right, .. } => { + ScalarExpr::Column(_) + | ScalarExpr::Literal(_) + | ScalarExpr::EvalTimestamp + | ScalarExpr::CurrentTimestamp => Ok(0), + ScalarExpr::Compare { left, right, .. } | ScalarExpr::Arithmetic { left, right, .. } => { with_local(&[left, right]) } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + ScalarExpr::BoolAnd(parts) | ScalarExpr::BoolOr(parts) => { let children = parts.iter().collect::>(); add(&children)? .checked_add( @@ -1135,12 +1146,11 @@ fn scalar_operation_count( ) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::Not(child) - | QueryExpr::IsNull(child) - | QueryExpr::IsNotNull(child) - | QueryExpr::PromqlScalarBridge(child) => with_local(&[child]), - QueryExpr::Cast { expr, .. } => with_local(&[expr]), - QueryExpr::InList { expr, list, .. } => { + ScalarExpr::Not(child) | ScalarExpr::IsNull(child) | ScalarExpr::IsNotNull(child) => { + with_local(&[child]) + } + ScalarExpr::Cast { expr, .. } => with_local(&[expr]), + ScalarExpr::InList { expr, list, .. } => { let mut children = Vec::with_capacity(list.len() + 1); children.push(expr.as_ref()); children.extend(list.iter()); @@ -1148,11 +1158,11 @@ fn scalar_operation_count( .checked_add(u64::try_from(list.len()).map_err(|_| AnalyticalCostError::Overflow)?) .ok_or(AnalyticalCostError::Overflow) } - QueryExpr::FunctionCall { args, .. } => { + ScalarExpr::FunctionCall { args, .. } => { let children = args.iter().collect::>(); with_local(&children) } - QueryExpr::Case { + ScalarExpr::Case { operand, branches, else_expr, @@ -1174,10 +1184,10 @@ fn scalar_operation_count( } fn supports_hash_aggregate( - reduction: &asap_types::pre_asap::Reduction, - measures: &[asap_types::pre_asap::AggIntent], + reduction: &asap_types::ir::operator::Reduction, + measures: &[asap_types::ir::operator::AggIntent], ) -> bool { - use asap_types::pre_asap::{AggIntent, Reduction}; + use asap_types::ir::operator::{AggIntent, Reduction}; matches!(reduction, Reduction::Reduce(_)) && !measures.is_empty() @@ -1198,19 +1208,20 @@ fn supports_hash_aggregate( }) } -fn presence_intent(intent: &asap_types::pre_asap::AggIntent) -> bool { +fn presence_intent(intent: &asap_types::ir::operator::AggIntent) -> bool { matches!( intent, - asap_types::pre_asap::AggIntent::Absent | asap_types::pre_asap::AggIntent::AbsentOverTime + asap_types::ir::operator::AggIntent::Absent + | asap_types::ir::operator::AggIntent::AbsentOverTime ) } -fn present_over_time_intent(intent: &asap_types::pre_asap::AggIntent) -> bool { - matches!(intent, asap_types::pre_asap::AggIntent::PresentOverTime) +fn present_over_time_intent(intent: &asap_types::ir::operator::AggIntent) -> bool { + matches!(intent, asap_types::ir::operator::AggIntent::PresentOverTime) } -fn fixed_state_per_series_intent(intent: &asap_types::pre_asap::AggIntent) -> bool { - use asap_types::pre_asap::AggIntent; +fn fixed_state_per_series_intent(intent: &asap_types::ir::operator::AggIntent) -> bool { + use asap_types::ir::operator::AggIntent; matches!( intent, AggIntent::Rate @@ -1246,26 +1257,17 @@ fn fixed_state_per_series_intent(intent: &asap_types::pre_asap::AggIntent) -> bo ) } -fn is_promql_scalar(query: &asap_types::pre_asap::QueryExpr) -> bool { - use asap_types::pre_asap::QueryExpr; - matches!( - query, - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::PromqlScalarFromVector(_) - | QueryExpr::EvalTimestamp - ) -} - #[cfg(test)] mod tests { use super::*; - use crate::analytical_cost::{ + use crate::cost::analytical_cost::{ estimate_physical_dag, estimate_physical_dag_comparison, PhysicalDAGEstimateRequest, }; - use crate::physical_operator_statistics::{ + use crate::cost::physical_operator_statistics::{ validate_comparison_scopes, BinaryEdgeStatistics, PartitionStatistics, PromqlEdgeStatistics, PromqlUnaryEdgeStatistics, PromqlValueKind, UnaryEdgeStatistics, }; + use asap_types::ir::{BinaryOperator, ExprSemantics, Predicate, SortKey, TimeRangeKind}; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; @@ -1383,7 +1385,7 @@ mod tests { } } - fn scope(sources: Vec) -> ComparisonScope { + fn scope(sources: Vec) -> ComparisonScope { ComparisonScope { data_arrival: DataArrival::AtRest, planning_time: TimestampMs(1_000), @@ -1402,10 +1404,10 @@ mod tests { } fn coverage( - source: asap_types::pre_asap::Source, - predicates: Vec, - ) -> SourceCoverage { - SourceCoverage { + source: asap_types::ir::operator::Source, + predicates: Vec, + ) -> ScanSelection { + ScanSelection { source, source_snapshot_id: "snapshot-1".into(), predicates, @@ -1414,15 +1416,16 @@ mod tests { } #[test] - fn info_source_coverage_includes_symbolic_selector_matchers() { - use asap_types::pre_asap::{CompareOpKind, InfoMatcher, Source}; + fn info_scan_selection_includes_symbolic_selector_matchers() { + use asap_types::ir::operator::{InfoMatcher, Source}; + use asap_types::ir::scalar::CompareOpKind; let selector = vec![InfoMatcher { label: "cluster".into(), op: CompareOpKind::Eq, value: "prod".into(), }]; - let info_coverage = SourceCoverage { + let info_coverage = ScanSelection { source: Source::TimeSeries { metric: "target_info".into(), }, @@ -1451,27 +1454,31 @@ mod tests { // Correlation can be costed as an exact hash aggregate using provider-supplied state size. #[test] fn correlation_lowers_to_physical_hash_aggregate() { - use asap_types::pre_asap::{ - AggIntent, DataType, Field, QueryExpr, Reduction, Schema, Source, - }; + use asap_types::ir::operator::{AggIntent, Reduction, Source}; + use asap_types::ir::schema::{DataType, Field, Schema}; let source = Source::Table { table_ref: "pairs".into(), }; - let root = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], - output_names: vec!["r".into()], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::Scan { - source: source.clone(), - predicates: vec![], - schema: Schema::new(vec![ - Field::plain("x", DataType::Float64, true), - Field::plain("y", DataType::Float64, true), - ]), - }), - }); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], + output_names: vec!["r".into()], + filters: vec![], + having: None, + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::Scan { + source: source.clone(), + predicates: vec![], + schema: Schema::new(vec![ + Field::plain("x", DataType::Float64, true), + Field::plain("y", DataType::Float64, true), + ]), + }, + )) + .unwrap(), + })) + .unwrap(); let scope = scope(vec![coverage(source, vec![])]); let provided = HashMap::from([ ( @@ -1501,51 +1508,57 @@ mod tests { #[test] fn query_lowering_recurses_and_fuses_global_sort_limit() { - use asap_types::pre_asap::{AggIntent, GroupKeys, QueryExpr, Reduction, SortKey, Source}; - use asap_types::pre_asap::{DataType, Field, Schema}; + use asap_types::ir::operator::{AggIntent, GroupKeys, Reduction, Source}; + use asap_types::ir::schema::{DataType, Field, Schema}; use std::rc::Rc; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, - predicates: vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + predicates: vec![Predicate(ScalarExpr::Literal( + asap_types::ir::scalar::ScalarValue::Boolean(true), ))], schema: Schema::new(vec![ Field::plain("service", DataType::Utf8, false), Field::plain("value", DataType::Float64, false), ]), - }); - let aggregate = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![0]), - measures: vec![AggIntent::Sum { col: Some(1) }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::clone(&scan), - }); - let sort = Rc::new(QueryExpr::Sort { + })) + .unwrap(); + let aggregate = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![0]), + measures: vec![AggIntent::Sum { col: Some(1) }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::clone(&scan), + })) + .unwrap(); + let sort = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: ScalarExpr::Column(0), ascending: false, nulls_first: false, }], partition_by: GroupKeys::none(), child: aggregate, - }); - let root = Rc::new(QueryExpr::Limit { - n: 10, + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), offset: 5, + partition_by: GroupKeys::none(), child: sort, - }); + })) + .unwrap(); let scan_coverage = coverage( Source::Table { table_ref: "events".into(), }, - vec![asap_types::pre_asap::Predicate(Rc::new( - QueryExpr::Literal(asap_types::pre_asap::ScalarValue::Boolean(true)), + vec![Predicate(ScalarExpr::Literal( + asap_types::ir::scalar::ScalarValue::Boolean(true), ))], ); let scope = scope(vec![scan_coverage]); @@ -1611,10 +1624,7 @@ mod tests { )); let physical_scan = &dag.nodes[0]; assert_eq!(physical_scan.id, "query-2-scan"); - assert_eq!( - physical_scan.source_coverage, - Some(scope.sources[0].clone()) - ); + assert_eq!(physical_scan.scan_selection, Some(scope.sources[0].clone())); assert_eq!(physical_scan.output_buffer_bytes, 1_024); assert_ne!( physical_scan.output_buffer_bytes, @@ -1640,34 +1650,38 @@ mod tests { #[test] fn query_lowering_shares_only_provider_identified_physical_nodes() { - use asap_types::pre_asap::{CompareOpKind, DataType, Field, Schema}; - use asap_types::pre_asap::{JoinKind, Predicate, QueryExpr, Source}; + use asap_types::ir::operator::{JoinKind, Source}; + use asap_types::ir::scalar::CompareOpKind; + use asap_types::ir::schema::{DataType, Field, Schema}; use std::rc::Rc; - let shared = Rc::new(QueryExpr::Scan { + let shared = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "dimensions".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let root = Rc::new(QueryExpr::Join { + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(1)), - })), + right: Box::new(ScalarExpr::Column(1)), + semantics: ExprSemantics::Sql, + }), left: Rc::clone(&shared), right: Rc::clone(&shared), - }); - let source_coverage = coverage( + })) + .unwrap(); + let scan_selection = coverage( Source::Table { table_ref: "dimensions".into(), }, vec![], ); - let independent_scope = scope(vec![source_coverage.clone()]); + let independent_scope = scope(vec![scan_selection.clone()]); let scan_statistics = scan_stats(edge(100, 800), 800); let join_statistics = OperatorStatistics::HashJoin { edges: BinaryEdgeStatistics { @@ -1703,8 +1717,8 @@ mod tests { statistics, }) }; - let shared_scope = scope(vec![source_coverage]); - let no_cache = crate::analytical_cost::CacheProfile::no_cache(); + let shared_scope = scope(vec![scan_selection]); + let no_cache = crate::cost::analytical_cost::CacheProfile::no_cache(); let shared_dag = lower_query_physical_dag(&root, &shared_scope, &shared_provider).unwrap(); assert_eq!(shared_dag.nodes.len(), 2); assert_eq!( @@ -1784,16 +1798,19 @@ mod tests { )) ); - let invalid = Rc::new(QueryExpr::Join { - kind: JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), - op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Column(0)), - })), - left: Rc::clone(&shared), - right: Rc::clone(&shared), - }); + let invalid = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(0)), + semantics: ExprSemantics::Sql, + }), + left: Rc::clone(&shared), + right: Rc::clone(&shared), + })) + .unwrap(); assert_eq!( lower_query_physical_dag(&invalid, &shared_scope, &shared_provider), Err(AnalyticalCostError::UnsupportedQueryOperator) @@ -1802,71 +1819,83 @@ mod tests { #[test] fn query_lowering_covers_relational_unary_operators() { - use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; - use asap_types::pre_asap::{ - GroupKeys, Predicate, QueryExpr, SortKey, Source, TimeShift, WindowFuncKind, - }; - use std::rc::Rc; + use asap_types::ir::operator::{GroupKeys, Source, TimeShift, WindowFuncKind}; + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::schema::{DataType, Field, Schema}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: scan, - }); - let project = Rc::new(QueryExpr::Project { - cols: vec![], - qualifier: None, - child: filter, - }); - let dedup = Rc::new(QueryExpr::Dedup { + })) + .unwrap(); + let filter = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))), + child: scan, + })) + .unwrap(); + let project = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols: vec![], + qualifier: None, + child: filter, + })) + .unwrap(); + let dedup = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Dedup { cols: vec![0], child: project, - }); - let window = Rc::new(QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::none(), - order_by: vec![SortKey { - expr: QueryExpr::Column(0), - ascending: true, - nulls_first: false, - }], - frame: None, - output_name: "rn".into(), - child: dedup, - }); - let sort = Rc::new(QueryExpr::Sort { + })) + .unwrap(); + let window = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::SQLWindowFunc { + func: WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::none(), + order_by: vec![SortKey { + expr: ScalarExpr::Column(0), + ascending: true, + nulls_first: false, + }], + frame: None, + output_name: "rn".into(), + child: dedup, + }, + )) + .unwrap(); + let sort = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Sort { keys: vec![SortKey { - expr: QueryExpr::Column(0), + expr: ScalarExpr::Column(0), ascending: true, nulls_first: false, }], partition_by: GroupKeys::by(vec![0]), child: window, - }); - let limit = Rc::new(QueryExpr::Limit { - n: 20, + })) + .unwrap(); + let limit = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(20), offset: 0, + partition_by: GroupKeys::none(), child: sort, - }); - let root = Rc::new(QueryExpr::TimeShift { - shift: TimeShift::default(), - child: limit, - }); - - let source_coverage = coverage( + })) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeShift { + shift: TimeShift::default(), + child: limit, + })) + .unwrap(); + + let scan_selection = coverage( Source::Table { table_ref: "events".into(), }, vec![], ); - let scope = scope(vec![source_coverage]); + let scope = scope(vec![scan_selection]); let scan_statistics = scan_stats(edge(1_000, 8_000), 8_000); let dedup_statistics = OperatorStatistics::HashDeduplicate { edges: unary_edges(edge(800, 3_200), edge(500, 2_000)), @@ -1976,23 +2005,26 @@ mod tests { #[test] fn query_lowering_maps_concat_and_union_all_but_rejects_distinct_set_ops() { - use asap_types::pre_asap::{DataType, Field, Schema}; - use asap_types::pre_asap::{QueryExpr, RelationalSetOpKind, Source}; - use std::rc::Rc; + use asap_types::ir::operator::{RelationalSetOpKind, Source}; + use asap_types::ir::schema::{DataType, Field, Schema}; - let scan = |name: &str| QueryExpr::Scan { - source: Source::Table { - table_ref: name.into(), - }, - predicates: vec![], - schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + let scan = |name: &str| { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: name.into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), + })) + .unwrap() }; - let union = Rc::new(QueryExpr::SetOp { + let union = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::SetOp { kind: RelationalSetOpKind::Union, all: true, - left: Rc::new(scan("a")), - right: Rc::new(scan("b")), - }); + left: scan("a"), + right: scan("b"), + })) + .unwrap(); let scope = scope(vec![ coverage( Source::Table { @@ -2032,23 +2064,27 @@ mod tests { assert_eq!( duplicate_scope.validate(), Err(AnalyticalCostError::MissingComparisonScope( - "duplicate source coverage" + "duplicate scan selection" )) ); - let concat = Rc::new(QueryExpr::Concat { - children: vec![scan("a"), scan("b")], - discriminator_unique_key: None, - }); + let concat = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Concat { + children: vec![scan("a"), scan("b")], + discriminator_unique_key: None, + })) + .unwrap(); let dag = lower_query_physical_dag(&concat, &scope, &scripted(&provided)).unwrap(); assert_eq!(dag.nodes.last().unwrap().operator, PhysicalOperator::Concat); - let distinct_union = Rc::new(QueryExpr::SetOp { - kind: RelationalSetOpKind::Union, - all: false, - left: Rc::new(scan("a")), - right: Rc::new(scan("b")), - }); + let distinct_union = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::SetOp { + kind: RelationalSetOpKind::Union, + all: false, + left: scan("a"), + right: scan("b"), + })) + .unwrap(); assert_eq!( lower_query_physical_dag(&distinct_union, &scope, &scripted(&provided)), Err(AnalyticalCostError::UnsupportedQueryOperator) @@ -2057,22 +2093,24 @@ mod tests { #[test] fn query_lowering_fails_closed_for_missing_or_inconsistent_statistics() { - use asap_types::pre_asap::{DataType, Field, Schema}; - use asap_types::pre_asap::{QueryExpr, Source}; - use std::rc::Rc; + use asap_types::ir::operator::Source; + use asap_types::ir::schema::{DataType, Field, Schema}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let root = Rc::new(QueryExpr::Project { - cols: vec![], - qualifier: None, - child: scan, - }); + })) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { + cols: vec![], + qualifier: None, + child: scan, + })) + .unwrap(); let comparison_scope = scope(vec![coverage( Source::Table { @@ -2142,7 +2180,7 @@ mod tests { assert_eq!( lower_query_physical_dag(&root, &ambiguous_scope, &scripted(&conflicting)), Err(AnalyticalCostError::InvalidPhysicalDAG( - "scan source coverage is ambiguous" + "scan selection is ambiguous" )) ); @@ -2173,26 +2211,31 @@ mod tests { #[test] fn query_lowering_accepts_a_consistently_empty_edge() { - use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema}; - use asap_types::pre_asap::{Predicate, QueryExpr, Source}; - use std::rc::Rc; + use asap_types::ir::operator::{GroupKeys, Source}; + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::schema::{DataType, Field, Schema}; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("id", DataType::Int64, false)]), - }); - let filter = Rc::new(QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(false)))), - child: scan, - }); - let root = Rc::new(QueryExpr::Limit { - n: 10, + })) + .unwrap(); + let filter = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Literal(ScalarValue::Boolean(false))), + child: scan, + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), offset: 0, + partition_by: GroupKeys::none(), child: filter, - }); + })) + .unwrap(); let scope = scope(vec![coverage( Source::Table { @@ -2233,59 +2276,70 @@ mod tests { #[test] fn query_lowering_rejects_aggregates_without_a_hash_implementation() { - use asap_types::pre_asap::{ - AggIntent, GroupKeys, QueryExpr, Reduction, Source, WindowFuncKind, - }; - use asap_types::pre_asap::{DataType, Field, Schema}; + use asap_types::ir::operator::{AggIntent, GroupKeys, Reduction, Source, WindowFuncKind}; + use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::types::AccuracyTarget; - use std::rc::Rc; let scan = || { - Rc::new(QueryExpr::Scan { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: "events".into(), }, predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }) + })) + .unwrap() }; - let exact_quantile = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Quantile { - col: Some(0), - q: 0.99, - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: scan(), - }); - let empty_sort_limit = Rc::new(QueryExpr::Limit { - n: 10, - offset: 0, - child: Rc::new(QueryExpr::Sort { - keys: vec![], + let exact_quantile = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![AggIntent::Quantile { + col: Some(0), + q: 0.99, + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + })) + .unwrap(); + let empty_sort_limit = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), + offset: 0, partition_by: GroupKeys::none(), + child: OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::Sort { + keys: vec![], + partition_by: GroupKeys::none(), + child: scan(), + }, + )) + .unwrap(), + })) + .unwrap(); + let unsupported_window = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::SQLWindowFunc { + func: WindowFuncKind::Lag, + args: vec![ScalarExpr::Column(0)], + partition_by: GroupKeys::none(), + order_by: vec![], + frame: None, + output_name: "lag".into(), child: scan(), - }), - }); - let unsupported_window = Rc::new(QueryExpr::SQLWindowFunc { - func: WindowFuncKind::Lag, - args: vec![QueryExpr::Column(0)], - partition_by: GroupKeys::none(), - order_by: vec![], - frame: None, - output_name: "lag".into(), - child: scan(), - }); - let shifted = Rc::new(QueryExpr::TimeShift { - shift: asap_types::pre_asap::TimeShift { - offset_ms: 60_000, - at: None, }, - child: scan(), - }); + )) + .unwrap(); + let shifted = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeShift { + shift: asap_types::ir::operator::TimeShift { + offset_ms: 60_000, + at: None, + }, + child: scan(), + })) + .unwrap(); let scope = scope(vec![coverage( Source::Table { table_ref: "events".into(), @@ -2308,40 +2362,43 @@ mod tests { #[test] fn scalar_work_counts_every_local_predicate_operation() { - use asap_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; - let comparison = || QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), + let comparison = || ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(0)), op: CompareOpKind::Eq, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(1))), + semantics: ExprSemantics::Sql, }; - let predicate = QueryExpr::BoolAnd(vec![comparison(), comparison()]); + let predicate = ScalarExpr::BoolAnd(vec![comparison(), comparison()]); assert_eq!(scalar_operation_count(&predicate), Ok(3)); } #[test] fn promql_presence_is_lowered_with_a_per_step_output_bound() { - use asap_types::pre_asap::{ - AggIntent, DataType, Field, QueryExpr, Reduction, Schema, Source, - }; + use asap_types::ir::operator::{AggIntent, Reduction, Source}; + use asap_types::ir::schema::{DataType, Field, Schema}; let source = Source::TimeSeries { metric: "missing".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }); - let root = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Absent], - output_names: vec![], - filters: vec![], - having: None, - child: scan, - }); + })) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![AggIntent::Absent], + output_names: vec![], + filters: vec![], + having: None, + child: scan, + })) + .unwrap(); let vector = promql_edge(0, 2, PromqlValueKind::Vector); let scan_statistics = OperatorStatistics::Scan { edges: promql_unary_edges(edge(0, 0), edge(0, 0), vector, vector), @@ -2390,24 +2447,32 @@ mod tests { #[test] fn promql_range_and_subquery_preserve_internal_steps() { - use asap_types::pre_asap::{DataType, Field, QueryExpr, Schema, Source}; + use asap_types::ir::operator::Source; + use asap_types::ir::schema::{DataType, Field, Schema}; use std::time::Duration; let source = Source::TimeSeries { metric: "m".into() }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }); - let range = Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: scan, - }); - let root = Rc::new(QueryExpr::PromqlSubquery { - range: Duration::from_secs(300), - resolution: Some(Duration::from_secs(60)), - child: range, - }); + })) + .unwrap(); + let range = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::TimeRange { + range: Duration::from_secs(300), + kind: TimeRangeKind::Range, + child: scan, + })) + .unwrap(); + let root = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlSubquery { + range: Duration::from_secs(300), + resolution: Some(Duration::from_secs(60)), + child: range, + }, + )) + .unwrap(); let vector = promql_edge(10, 6, PromqlValueKind::Vector); let range_vector = promql_edge(10, 6, PromqlValueKind::RangeVector); let outer_range = promql_edge(10, 1, PromqlValueKind::RangeVector); @@ -2464,33 +2529,42 @@ mod tests { #[test] fn promql_binary_lowering_keeps_operation_and_matching_cardinality() { - use asap_types::pre_asap::{ - ArithmeticOpKind, BinaryOpKind, DataType, Field, GroupSide, QueryExpr, Schema, Source, - VectorGrouping, VectorMatch, VectorMatchKind, + use asap_types::ir::operator::{ + BinaryOpKind, GroupSide, Source, VectorGrouping, VectorMatch, VectorMatchKind, }; + use asap_types::ir::scalar::ArithmeticOpKind; + use asap_types::ir::schema::{DataType, Field, Schema}; let left_source = Source::TimeSeries { metric: "a".into() }; let right_source = Source::TimeSeries { metric: "b".into() }; let scan = |source| { - Rc::new(QueryExpr::Scan { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source, predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }) + })) + .unwrap() }; - let root = Rc::new(QueryExpr::BinaryOp { - op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), - lhs: scan(left_source.clone()), - rhs: scan(right_source.clone()), - vector_match: Some(VectorMatch { - kind: VectorMatchKind::On, - labels: vec!["service".into()], - grouping: Some(VectorGrouping { - side: GroupSide::Left, - labels: vec!["region".into()], - }), - }), - }); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + vector_match: Some(VectorMatch { + kind: VectorMatchKind::On, + labels: vec!["service".into()], + grouping: Some(VectorGrouping { + side: GroupSide::Left, + labels: vec!["region".into()], + }), + }), + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: scan(left_source.clone()), + rhs: scan(right_source.clone()), + })) + .unwrap(); let left_promql = promql_edge(10, 10, PromqlValueKind::Vector); let right_promql = promql_edge(5, 10, PromqlValueKind::Vector); let output_promql = promql_edge(8, 10, PromqlValueKind::Vector); @@ -2519,7 +2593,7 @@ mod tests { inputs: [left_edge, right_edge], output: output_edge, promql: Some( - crate::physical_operator_statistics::PromqlBinaryEdgeStatistics { + crate::cost::physical_operator_statistics::PromqlBinaryEdgeStatistics { inputs: [left_promql, right_promql], output: output_promql, }, @@ -2547,37 +2621,45 @@ mod tests { #[test] fn promql_relabel_sample_and_per_series_lower_as_a_complete_chain() { - use asap_types::pre_asap::{ - AggIntent, DataType, Field, GroupKeys, QueryExpr, Reduction, SampleKind, ScalarValue, - Schema, Source, - }; + use asap_types::ir::operator::{AggIntent, GroupKeys, Reduction, SampleKind, Source}; + use asap_types::ir::scalar::ScalarValue; + use asap_types::ir::schema::{DataType, Field, Schema}; let source = Source::TimeSeries { metric: "requests".into(), }; - let scan = Rc::new(QueryExpr::Scan { + let scan = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: source.clone(), predicates: vec![], schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }); - let relabel = Rc::new(QueryExpr::PromqlRelabel { - dst: "service".into(), - value: Rc::new(QueryExpr::Literal(ScalarValue::Utf8("api".into()))), - child: scan, - }); - let sample = Rc::new(QueryExpr::PromqlSeriesSample { - by: GroupKeys::none(), - kind: SampleKind::LimitK(5), - child: relabel, - }); - let root = Rc::new(QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: sample, - }); + })) + .unwrap(); + let relabel = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlRelabel { + dst: "service".into(), + value: ScalarExpr::Literal(ScalarValue::Utf8("api".into())), + child: scan, + }, + )) + .unwrap(); + let sample = OperatorNode::new_shared(asap_types::ir::Operator::NonASAP( + NonASAPOp::PromqlSeriesSample { + by: GroupKeys::none(), + kind: SampleKind::LimitK(5), + child: relabel, + }, + )) + .unwrap(); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::PerEntity, + measures: vec![AggIntent::Sum { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: sample, + })) + .unwrap(); let input = edge(100, 1_600); let sampled = edge(50, 800); diff --git a/crates/asap-aware-mapping/src/recurrence.rs b/crates/plan-selection/src/cost/recurrence.rs similarity index 87% rename from crates/asap-aware-mapping/src/recurrence.rs rename to crates/plan-selection/src/cost/recurrence.rs index 660787cdb..fa88830bf 100644 --- a/crates/asap-aware-mapping/src/recurrence.rs +++ b/crates/plan-selection/src/cost/recurrence.rs @@ -20,10 +20,10 @@ //! |---|---|---| //! | [`UpdateRate`] | Hz (updates/second) | how often the *raw* data underlying a maintained summary changes (ingest rate) | //! | [`EvaluationRate`] | Hz (evaluations/second) | how often a target is *read* — `sum(1 / query_interval_i)` over every repeating consumer | -//! | [`CostRate`] | cost units / second | a steady-state cost rate — never comparable to a bare [`Cost`](crate::cost_model::Cost) without going through [`total_cost`] | +//! | [`CostRate`] | cost units / second | a steady-state cost rate — never comparable to a bare [`Cost`](crate::cost::cost_model::Cost) without going through [`total_cost`] | //! | [`Horizon`] | seconds | the explicit evaluation window a caller supplies to compare a rate-valued cost against a one-shot cost | //! -//! `Cost` (bare, from [`crate::cost_model`]) stays a one-time, unitless +//! `Cost` (bare, from [`crate::cost::cost_model`]) stays a one-time, unitless //! magnitude — exactly what it was before this module existed, preserved //! for [`CostModel::cse_share_decision`] and everything else that already //! uses it. `CostRate` is a *new*, distinct type specifically so a rate and @@ -75,7 +75,7 @@ //! //! - [`EvaluationRate`]: derived from [`asap_types::workload::RepeatingEntry::demand`] //! values of every repeating consumer reaching a target (via -//! [`evaluation_rate_of`], or [`crate::replacement::CandidateLogicalASAPDAGs::recurrence_profiles`] +//! [`evaluation_rate_of`], or [`recurrence_profiles`](crate::candidate_selection::recurrence_profiles) //! for a whole workload). A one-shot ([`asap_types::workload::BatchEntry`]) //! consumer contributes to [`RecurrenceProfile::one_shot_consumers`] //! instead, never to this rate. @@ -106,7 +106,7 @@ use std::fmt; use asap_types::workload::{DataWorkload, RepetitionInterval}; -use crate::cost_model::{Cost, CostModel, CseCandidate, ShareDecision}; +use crate::cost::cost_model::{Cost, CostModel, CseCandidate, ShareDecision}; // ── Units ──────────────────────────────────────────────────────────────── @@ -216,7 +216,7 @@ pub enum RecurrenceError { CostRate with a one-shot Cost without distorting the comparison" )] InvalidHorizon(Horizon), - /// [`crate::replacement::CandidateLogicalASAPDAGs::recurrence_profiles`] was called + /// [`recurrence_profiles`](crate::candidate_selection::recurrence_profiles) was called /// with a `root_recurrence` slice whose length doesn't match the /// `CandidateLogicalASAPDAGs`'s own root count — a caller error, but recoverable /// (this method's whole signature promises a `Result`, so this is @@ -239,7 +239,7 @@ pub enum RecurrenceError { /// applied at every point an `UpdateRate` enters a [`RecurrenceProfile`] /// ([`RecurrenceProfile::with_update_rate`], /// [`update_rate_from_data_workload`], -/// [`crate::replacement::CandidateLogicalASAPDAGs::recurrence_profiles`]'s own parameter) +/// [`recurrence_profiles`](crate::candidate_selection::recurrence_profiles)'s own parameter) /// *and*, as a backstop that can't be bypassed by constructing a /// `RecurrenceProfile` via its public fields directly, inside [`decide`] /// itself before any comparison uses it. @@ -373,7 +373,7 @@ impl RecurrenceProfile { } /// How one workload root recurs — the opaque per-root tag -/// [`crate::replacement::CandidateLogicalASAPDAGs::recurrence_profiles`] threads down to +/// [`recurrence_profiles`](crate::candidate_selection::recurrence_profiles) threads down to /// every target reachable from that root. Mirrors /// [`asap_types::workload::QueryWorkload`]'s own `query_batch` (one-shot) /// vs. `repeating_queries` (an interval each) split, but at the @@ -394,7 +394,7 @@ pub enum RootRecurrence { // ── Explanation ────────────────────────────────────────────────────────── -/// The full readout [`CostModel::cse_share_decision_with_recurrence`] +/// The full evaluation [`CostModel::cse_share_decision_with_recurrence`] /// returns: which alternative was selected, both compared cost rates /// (and, when a [`Horizon`] was supplied, both compared totals), every /// input that went into them, their units, and provenance — meant to be @@ -632,7 +632,10 @@ pub(crate) fn decide( #[cfg(test)] mod tests { use super::*; - use crate::cost_model::DefaultCostModel; + use crate::candidate_selection::cost_sorted_with_recurrence; + use crate::candidate_selection::global_selection_with_recurrence; + use crate::candidate_selection::recurrence_profiles; + use crate::cost::cost_model::DefaultCostModel; fn interval(ms: u32) -> RepetitionInterval { RepetitionInterval(ms) @@ -778,18 +781,22 @@ mod tests { // ── decide (structural fallback) ───────────────────────────────────── - use crate::cost_model::CseCandidate; - use asap_types::post_asap::{ - ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, ResultGuarantee, Schema, - SummaryExpr, SummaryNode, + use crate::cost::cost_model::CseCandidate; + use asap_types::ir::operator::operator_properties::{Reduction, Source}; + use asap_types::ir::properties::ResultGuarantee; + use asap_types::ir::scalar::ColumnRef; + use asap_types::ir::schema::DataType; + use asap_types::ir::schema::{ + ExactKind, ExactParams, Field, FieldDataType, GroupingStrategy, Schema, }; - use asap_types::pre_asap::expr_ir::ColumnRef; - use asap_types::pre_asap::query_expr::{QueryExpr, Reduction, Source}; - use asap_types::pre_asap::schema::DataType; + use asap_types::ir::{ + ASAPOp, BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, Predicate, ScalarExpr, + }; + use std::rc::Rc; - fn scan() -> QueryExpr { - QueryExpr::Scan { + fn scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -800,28 +807,34 @@ mod tests { 0, vec![], ), - } + })) + .unwrap() } - fn summary_node(family: FieldDataType) -> SummaryNode { - SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(scan())), - schema: Schema::lifted(vec![], None), - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), + /// A summary of `family` over the kept pre-ASAP scan. + fn summary_node(family: FieldDataType) -> Rc { + let kept = Rc::new( + scan() + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("RetainedExact"))), + ); + std::rc::Rc::new( + OperatorNode::with_schema( + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: kept, + family: family.clone(), + input: asap_types::ir::schema::SummaryUpdate::column(ColumnRef::Named( + "value".into(), + )), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, }), - family: family.clone(), - input: asap_types::post_asap::SummaryUpdate::column(ColumnRef::Named( - "value".into(), - )), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![Field::new("state", family, false)], None), - guarantee: None, - } + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(None), + ) } #[test] @@ -907,9 +920,9 @@ mod tests { fn rank_candidates( &self, - _intent: &asap_types::pre_asap::agg_intent::AggIntent, - candidates: &[asap_types::post_asap::SketchAlgorithm], - ) -> Vec { + _intent: &asap_types::ir::operator::agg_intent::AggIntent, + candidates: &[asap_types::ir::schema::SketchAlgorithm], + ) -> Vec { candidates.to_vec() } fn maintenance_cost_per_update(&self, _candidate: &CseCandidate) -> Cost { @@ -1119,21 +1132,21 @@ mod tests { // ── multiple roots sharing a sub-DAG, via CandidateLogicalASAPDAGs ────────────────── - use crate::replacement::search_workload; - use asap_types::pre_asap::agg_intent::AggIntent; - use asap_types::pre_asap::expr_ir::ScalarValue; - use asap_types::pre_asap::query_expr::{Predicate, Reduction as QueryReduction}; + use asap_logical_optimizer::pass1::replacement::search_workload; + use asap_types::ir::operator::agg_intent::AggIntent; + use asap_types::ir::operator::operator_properties::Reduction as QueryReduction; + use asap_types::ir::scalar::{CompareOpKind, ScalarValue}; /// Like `scan()`, plus a "job" label column to group by — CSE's /// sharing legality gate requires a provable unique key /// (`Schema::has_unique_key`), and an *ungrouped* aggregate's empty - /// `by` reports none (see `asap_types::pre_asap::cse`'s own "Legality" + /// `by` reports none (see `asap_types::ir::cse`'s own "Legality" /// module docs); grouping by a label column gives `sum_agg()` below a /// real one, matching the pattern /// `replacement.rs`'s own CSE fixtures already use (`metric_scan`/`agg` /// grouped by a label column). - fn labeled_scan() -> QueryExpr { - QueryExpr::Scan { + fn labeled_scan() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::TimeSeries { metric: "m".into() }, predicates: vec![], schema: Schema::with_time_index( @@ -1145,18 +1158,20 @@ mod tests { 0, vec![], ), - } + })) + .unwrap() } - fn sum_agg() -> QueryExpr { - QueryExpr::Aggregate { + fn sum_agg() -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { reduction: QueryReduction::by(vec![2]), measures: vec![AggIntent::Sum { col: Some(1) }], output_names: vec![], filters: vec![], having: None, - child: Rc::new(labeled_scan()), - } + child: labeled_scan(), + })) + .unwrap() } /// A root wrapping a fresh, independently-built (but structurally @@ -1167,28 +1182,34 @@ mod tests { /// `shared_aggregate_across_two_roots_gets_both_strategies_candidates`'s /// own doc) while letting `share_common_sub_dags` unify their /// identical `sum_agg()` children onto one shared `Rc`. - fn filtered_root(distinguishing_literal: i64) -> QueryExpr { - QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Int64( - distinguishing_literal, - )))), - child: Rc::new(sum_agg()), - } + fn filtered_root(distinguishing_literal: i64) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(1)), + op: CompareOpKind::Gt, + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64( + distinguishing_literal, + ))), + semantics: ExprSemantics::Sql, + }), + child: sum_agg(), + })) + .unwrap() } /// Three workload roots share one underlying `sum_agg()` sub-DAG: two /// repeating consumers with different intervals, one one-shot batch - /// consumer. `CandidateLogicalASAPDAGs::recurrence_profiles` must aggregate all three + /// consumer. `candidate_selection::recurrence_profiles` must aggregate all three /// onto the shared sub-DAG's own profile: `evaluation_rate = 1/t1 + /// 1/t2`, `one_shot_consumers = 1` — issue #287's "support a shared /// sub-DAG consumed by queries with different intervals" and "multiple /// roots sharing a sub-DAG" acceptance criteria. #[test] fn recurrence_profiles_aggregates_mixed_intervals_across_roots_sharing_a_subdag() { - let roots: Vec<(&str, Rc)> = vec![ - ("root_a", Rc::new(filtered_root(1))), - ("root_b", Rc::new(filtered_root(2))), - ("root_c", Rc::new(filtered_root(3))), + let roots: Vec<(&str, Rc)> = vec![ + ("root_a", filtered_root(1)), + ("root_b", filtered_root(2)), + ("root_c", filtered_root(3)), ]; let space = search_workload(roots); @@ -1206,7 +1227,7 @@ mod tests { ); let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("the shared sum_agg() is a discovered target"); assert_eq!(shared_group.consumer_count, 3, "shared by all 3 roots"); @@ -1215,9 +1236,8 @@ mod tests { repeating(10_000), // 0.1 Hz RootRecurrence::OneShotCount(1), ]; - let profiles = space - .recurrence_profiles(&root_recurrence, Some(UpdateRate(5.0))) - .unwrap(); + let profiles = + recurrence_profiles(&space, &root_recurrence, Some(UpdateRate(5.0))).unwrap(); let profile = profiles.for_target(&shared_group.target); let expected_rate = 1.0 / 1.0 + 1.0 / 10.0; // Hz @@ -1250,32 +1270,31 @@ mod tests { #[test] fn plan_selection_uses_recurrence_profiles_for_cse_choices() { - let roots = vec![ - ("a", Rc::new(filtered_root(1))), - ("b", Rc::new(filtered_root(2))), - ]; + let roots = vec![("a", filtered_root(1)), ("b", filtered_root(2))]; let space = search_workload(roots); let shared = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("the aggregate is shared by both roots"); let update_rate = Some(UpdateRate(10.0)); - let frequent = space - .recurrence_profiles(&[repeating(10), repeating(10)], update_rate) - .unwrap(); - let infrequent = space - .recurrence_profiles(&[repeating(100_000), repeating(100_000)], update_rate) - .unwrap(); + let frequent = + recurrence_profiles(&space, &[repeating(10), repeating(10)], update_rate).unwrap(); + let infrequent = recurrence_profiles( + &space, + &[repeating(100_000), repeating(100_000)], + update_rate, + ) + .unwrap(); - let frequent_ranked = space - .cost_sorted_with_recurrence(&DeterministicUnitCostModel, &frequent, None) - .unwrap(); - let infrequent_ranked = space - .cost_sorted_with_recurrence(&DeterministicUnitCostModel, &infrequent, None) - .unwrap(); + let frequent_ranked = + cost_sorted_with_recurrence(&space, &DeterministicUnitCostModel, &frequent, None) + .unwrap(); + let infrequent_ranked = + cost_sorted_with_recurrence(&space, &DeterministicUnitCostModel, &infrequent, None) + .unwrap(); let first_provenance = - |ranked: &[crate::replacement::RankedTargetSubDAGCandidates<'_>]| { + |ranked: &[crate::candidate_selection::RankedTargetSubDAGCandidates<'_>]| { ranked .iter() .find(|group| Rc::ptr_eq(group.target, &shared.target)) @@ -1284,43 +1303,50 @@ mod tests { }; assert_eq!( first_provenance(&frequent_ranked), - Some(crate::replacement::ReplacementProvenance::CseShare) + Some(asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseShare) ); assert_eq!( first_provenance(&infrequent_ranked), - Some(crate::replacement::ReplacementProvenance::CseRecompute) + Some(asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseRecompute) ); - let frequent_selected = space - .global_selection_with_recurrence(&DeterministicUnitCostModel, &frequent, None) - .unwrap(); - let infrequent_selected = space - .global_selection_with_recurrence(&DeterministicUnitCostModel, &infrequent, None) - .unwrap(); + let frequent_selected = + global_selection_with_recurrence(&space, &DeterministicUnitCostModel, &frequent, None) + .unwrap(); + let infrequent_selected = global_selection_with_recurrence( + &space, + &DeterministicUnitCostModel, + &infrequent, + None, + ) + .unwrap(); assert_eq!( frequent_selected .for_target(&shared.target) .and_then(|group| group.chosen) .map(|candidate| candidate.provenance), - Some(crate::replacement::ReplacementProvenance::CseShare) + Some(asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseShare) ); assert_eq!( infrequent_selected .for_target(&shared.target) .and_then(|group| group.chosen) .map(|candidate| candidate.provenance), - Some(crate::replacement::ReplacementProvenance::CseRecompute) + Some(asap_logical_optimizer::pass1::replacement::ReplacementProvenance::CseRecompute) ); } #[test] fn recurrence_profiles_rejects_an_invalid_evaluation_rate() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); - let err = space - .recurrence_profiles(&[RootRecurrence::Repeating(EvaluationRate(f64::NAN))], None) - .unwrap_err(); + let err = recurrence_profiles( + &space, + &[RootRecurrence::Repeating(EvaluationRate(f64::NAN))], + None, + ) + .unwrap_err(); assert!(matches!(err, RecurrenceError::InvalidEvaluationRate(_))); } @@ -1329,10 +1355,10 @@ mod tests { /// signature promises a `Result`. #[test] fn recurrence_profiles_reports_a_root_count_mismatch_as_an_error_not_a_panic() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); - let err = space.recurrence_profiles(&[], None).unwrap_err(); + let err = recurrence_profiles(&space, &[], None).unwrap_err(); assert_eq!( err, RecurrenceError::RootCountMismatch { @@ -1344,15 +1370,15 @@ mod tests { #[test] fn recurrence_profiles_rejects_an_invalid_update_rate() { - let root = Rc::new(scan()); - let roots: Vec<(&str, Rc)> = vec![("only", root)]; + let root = scan(); + let roots: Vec<(&str, Rc)> = vec![("only", root)]; let space = search_workload(roots); - let err = space - .recurrence_profiles( - &[RootRecurrence::OneShotCount(1)], - Some(UpdateRate(f64::NAN)), - ) - .unwrap_err(); + let err = recurrence_profiles( + &space, + &[RootRecurrence::OneShotCount(1)], + Some(UpdateRate(f64::NAN)), + ) + .unwrap_err(); assert!(matches!(err, RecurrenceError::InvalidUpdateRate(_))); } @@ -1373,23 +1399,25 @@ mod tests { /// `consumer_count`. #[test] fn recurrence_profiles_does_not_stamp_update_rate_on_a_site_unreachable_from_any_root() { - let avg_root = QueryExpr::Aggregate { - reduction: QueryReduction::by(vec![]), - measures: vec![AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let roots: Vec<(&str, Rc)> = vec![("q", Rc::new(avg_root))]; + let avg_root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: QueryReduction::by(vec![]), + measures: vec![AggIntent::Avg { col: None }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + })) + .unwrap(); + let roots: Vec<(&str, Rc)> = vec![("q", avg_root)]; let space = search_workload(roots); let count_group = space .target_subdag_candidates() .find(|g| { matches!( - g.target.as_ref(), - QueryExpr::Aggregate { measures, .. } + g.target.non_asap(), + Some(NonASAPOp::Aggregate { measures, .. }) if measures.iter().any(|m| matches!(m, AggIntent::Count { .. })) ) }) @@ -1399,9 +1427,8 @@ mod tests { ); let root_recurrence = vec![repeating(1_000)]; - let profiles = space - .recurrence_profiles(&root_recurrence, Some(UpdateRate(5.0))) - .unwrap(); + let profiles = + recurrence_profiles(&space, &root_recurrence, Some(UpdateRate(5.0))).unwrap(); let count_profile = profiles.for_target(&count_group.target); assert_eq!( @@ -1421,26 +1448,33 @@ mod tests { /// Issue #287 review (lower-priority item): a parent referencing the /// same shared child twice (`BinaryOp{lhs: X, rhs: X}`, the same shape - /// `pre_asap::cse`'s own within-one-query sharing collapses onto one + /// `ir::cse`'s own within-one-query sharing collapses onto one /// `Rc`) must credit that child with 2 contributions per repeating /// root, matching how `TargetSubDAGCandidates::consumer_count` already counts that /// exact structural occurrence twice — not 1, which a plain /// reachability-set walk would (wrongly) collapse it to. #[test] fn recurrence_profiles_credits_a_direct_repeated_reference_by_its_multiplicity() { - let root = QueryExpr::BinaryOp { - op: asap_types::pre_asap::query_expr::BinaryOpKind::Compare( - asap_types::pre_asap::expr_ir::CompareOpKind::Eq, - ), - lhs: Rc::new(sum_agg()), - rhs: Rc::new(sum_agg()), - vector_match: None, - }; - let space = search_workload(vec![("q", Rc::new(root))]); + let root = + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: asap_types::ir::operator::operator_properties::BinaryOpKind::Compare( + CompareOpKind::Eq, + ), + vector_match: None, + }, + return_bool: false, + lhs: sum_agg(), + rhs: sum_agg(), + })) + .unwrap(); + let space = search_workload(vec![("q", root)]); let shared_group = space .target_subdag_candidates() - .find(|g| matches!(g.target.as_ref(), QueryExpr::Aggregate { .. })) + .find(|g| matches!(g.target.non_asap(), Some(NonASAPOp::Aggregate { .. }))) .expect("sum_agg() should merge onto one shared Rc, referenced twice from BinaryOp"); assert_eq!( shared_group.consumer_count, 2, @@ -1448,7 +1482,7 @@ mod tests { ); let root_recurrence = vec![repeating(1_000)]; // 1 Hz - let profiles = space.recurrence_profiles(&root_recurrence, None).unwrap(); + let profiles = recurrence_profiles(&space, &root_recurrence, None).unwrap(); let profile = profiles.for_target(&shared_group.target); // Referenced twice from the one root: evaluation_rate should be @@ -1463,7 +1497,7 @@ mod tests { let scan_group = space .target_subdag_candidates() - .find(|group| matches!(group.target.as_ref(), QueryExpr::Scan { .. })) + .find(|group| matches!(group.target.non_asap(), Some(NonASAPOp::Scan { .. }))) .expect("the shared aggregate has a scan descendant"); assert_eq!( profiles diff --git a/crates/asap-aware-mapping/src/storage_io.rs b/crates/plan-selection/src/cost/storage_io.rs similarity index 97% rename from crates/asap-aware-mapping/src/storage_io.rs rename to crates/plan-selection/src/cost/storage_io.rs index 646a3532c..ba2363ffe 100644 --- a/crates/asap-aware-mapping/src/storage_io.rs +++ b/crates/plan-selection/src/cost/storage_io.rs @@ -5,13 +5,13 @@ use std::collections::{HashMap, HashSet}; use serde::{Deserialize, Serialize}; // Keep the original public import path while sharing the sole type definition. -pub use asap_types::resources::StorageResources; +pub use asap_types::workload::resources::StorageResources; -use crate::analytical_cost::{ +use crate::cost::analytical_cost::{ estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDAG, ExecutionMultiplicity, PhysicalDAGNode, PhysicalOperator, }; -use crate::physical_operator_statistics::{ComparisonScope, OperatorStatistics}; +use crate::cost::physical_operator_statistics::{ComparisonScope, OperatorStatistics}; pub const STORAGE_IO_MODEL_VERSION: &str = "storage-requests-v1"; @@ -121,7 +121,7 @@ pub fn estimate_storage_io( profile: &StorageIoProfile, evidence_version: &str, ) -> Result { - // Also prove source coverage, edge consistency, execution legality and DAG + // Also prove scan selection, edge consistency, execution legality and DAG // identity before using supplementary deployment evidence. estimate_physical_dag(&dag.nodes, &dag.root, scope, dag)?; let evaluations = scope.validate()?; diff --git a/crates/plan-selection/src/lib.rs b/crates/plan-selection/src/lib.rs new file mode 100644 index 000000000..7d27323fe --- /dev/null +++ b/crates/plan-selection/src/lib.rs @@ -0,0 +1,1720 @@ +//! `asap-plan-selection` — #509 Stage 3 (MVP): plan selection, the only stage +//! that computes cost. Cargo enforces the stage order: this crate depends on +//! `asap-types`, Stage 1 and Stage 2, never on the facade or the executor. +//! +//! - [`cost`] — the [`CostModel`] trait, analytical and evidence-based +//! pricing, recurrence, and the physical lowering and storage I/O they price. +//! - [`candidate_selection`] — the legacy cost-ranked selection over a Stage 1 +//! search (deleted under #580). +//! +//! Each Stage 2 candidate is checked against every query's accuracy target +//! with the accuracy model, and Count-Min is admitted only over weights proven +//! non-negative; a miss rejects the candidate as invalid, with a reason. +//! Every valid candidate is priced node by node over its physical DAG, so a +//! node shared by several queries is charged once, and the cheapest is +//! selected. The rest are reported valid but costlier. A candidate that +//! cannot be built or priced is rejected with its reason; it does not fail +//! the selection. +//! +//! Prices come from [`crate::cost::analytical_cost::estimate_operator`] over edge +//! statistics derived from the [`DataWorkload`] and a fixed default group +//! count; summary build and estimation are priced as rows × sketch depth and +//! rows read out. These numbers are illustrative, not calibrated. Latency +//! bounds and deployment capabilities are not checked yet. +//! +//! [`select_plan`] chooses over Stage 1's sharing variants without building +//! every combination: per variant, a dynamic program over target nesting (see +//! there). [`select_exhaustive`] builds and prices every combination, for +//! display and for checking the program. [`plan_stages`] runs the whole +//! pipeline from the frontends' roots. +pub mod candidate_selection; +pub mod cost; +#[cfg(test)] +mod test_support; + +pub use candidate_selection::{ + CompositionDecision, CostedGlobalSelection, RankedTargetSubDAGCandidates, RecurrenceProfileMap, +}; +pub use cost::cost_model::{ + maintenance_operation_plan_cost_rate, raw_recompute_cost_rate, read_operation_plan_cost_rate, + CostModel, CostProvenance, CostUnit, DefaultCostModel, ExactCompositionCostInputs, + ExactCompositionCostRequest, ValueOperationCapabilities, +}; +pub use cost::recurrence::{ + evaluation_rate_of, total_cost, update_rate_from_data_workload, CostRate, EvaluationRate, + Horizon, RecurrenceCostExplanation, RecurrenceError, RecurrenceProfile, RootRecurrence, + UpdateRate, +}; + +use std::collections::{BTreeMap, HashMap}; +use std::rc::Rc; + +use asap_types::ir::export::{ + NonASAPOpKind, PhysicalASAPDAG, PhysicalASAPNodeId, PhysicalASAPOperatorPayload as Payload, +}; +use asap_types::ir::operator::Reduction; +use asap_types::ir::schema::{DataType, Schema}; +use asap_types::ir::schema::{ + FieldDataType, SketchAlgorithm, SketchParams, SketchStatistic, WeightDomain, +}; +use asap_types::ir::{ASAPOp, Operator, OperatorNode, QueryRoot}; +use asap_types::types::AccuracyTarget; +use asap_types::workload::DataWorkload; +use thiserror::Error; + +use crate::cost::analytical_cost::{ + estimate_operator, AnalyticalCostError, PhysicalOperator, ResourceCalibration, ResourceEstimate, +}; +use crate::cost::physical_operator_statistics::{ + EdgeStatistics, OperatorStatistics, PartitionStatistics, UnaryEdgeStatistics, +}; +use asap_logical_optimizer::accuracy::{ + AccuracyEvidenceProvider, AccuracyModel, DefaultAccuracyModel, NoAccuracyEvidence, +}; +use asap_logical_optimizer::pass1::logical_candidates::{ + choice_index, combination_count, compose_logical_candidate, enumerate_choices, nested_targets, + read_targets, LocalLogicalCandidates, LogicalCandidateError, +}; +pub use asap_logical_optimizer::pass2::identical_expressions::SharingVariant; +use asap_logical_optimizer::pass2::identical_expressions::{ + share_identical_expressions, stage1_logical_candidates, +}; +use asap_physical_optimizer::implementation::physical_candidates::{ + stage2_physical, PhysicalCandidate, +}; + +pub const COST_UNIT: &str = "cpu_ms_per_workload_evaluation"; +pub const COST_SOURCE: &str = "analytical-cost-v1 (illustrative statistics)"; + +/// Groups assumed for every `by (...)` reduction, absent group-count evidence. +const DEFAULT_GROUP_COUNT: u64 = 100; +/// Used only when the data workload does not declare them. +const DEFAULT_SERIES: u64 = 1_000; +const DEFAULT_ROWS_PER_SECOND: f64 = 1_000.0; +const DEFAULT_LOOKBACK_MS: u64 = 60_000; +/// Most combinations built for display, and for selection when the dynamic +/// program's assumptions do not hold. +pub const MAX_ENUMERATED_CANDIDATES: usize = 64; + +static DEFAULT_COST_MODEL: DefaultCostModel = DefaultCostModel; +static DEFAULT_ACCURACY_MODEL: DefaultAccuracyModel = DefaultAccuracyModel; +static NO_ACCURACY_EVIDENCE: NoAccuracyEvidence = NoAccuracyEvidence; + +/// Planning logic, as opposed to the scoped facts it consumes: a model can have +/// a built-in default, evidence about a particular deployment cannot. +/// +/// Stage 3 prices plans analytically, so the stage pipeline does not read +/// `cost`; only the legacy replacement search does (#580). +#[derive(Clone, Copy)] +#[non_exhaustive] +pub struct PlanningModels<'a> { + pub cost: &'a dyn CostModel, + pub accuracy: &'a dyn AccuracyModel, + pub evidence: &'a dyn AccuracyEvidenceProvider, +} + +impl<'a> PlanningModels<'a> { + pub fn new( + cost: &'a dyn CostModel, + accuracy: &'a dyn AccuracyModel, + evidence: &'a dyn AccuracyEvidenceProvider, + ) -> Self { + Self { + cost, + accuracy, + evidence, + } + } + + /// The built-in models. `DefaultCostModel` does not override + /// `estimate_cost`, so this configuration ranks structurally and is not a + /// measured deployment cost. + pub fn builtin() -> PlanningModels<'static> { + PlanningModels { + cost: &DEFAULT_COST_MODEL, + accuracy: &DEFAULT_ACCURACY_MODEL, + evidence: &NO_ACCURACY_EVIDENCE, + } + } + + pub fn with_cost(mut self, cost: &'a dyn CostModel) -> Self { + self.cost = cost; + self + } + + pub fn with_accuracy(mut self, accuracy: &'a dyn AccuracyModel) -> Self { + self.accuracy = accuracy; + self + } + + pub fn with_evidence(mut self, evidence: &'a dyn AccuracyEvidenceProvider) -> Self { + self.evidence = evidence; + self + } +} + +#[derive(Debug, Clone, PartialEq)] +pub struct NodeCost { + pub cost: f64, + /// Estimated output rows. + pub rows: u64, + pub detail: String, +} + +/// Whole-workload cost of one candidate: one entry per DAG node, `total` is +/// their sum. +#[derive(Debug, Clone, PartialEq)] +pub struct CandidateCost { + pub total: f64, + pub unit: &'static str, + pub source: &'static str, + pub per_node: BTreeMap, +} + +/// A candidate that was not selected. `valid == false`: it failed a check or +/// could not be built or priced; `true`: it lost on cost. +#[derive(Debug, Clone, PartialEq)] +pub struct Rejection { + pub id: String, + pub valid: bool, + pub reason: String, +} + +/// How the selected candidate was found. +#[derive(Debug, Clone, PartialEq)] +pub enum SelectionMethod { + /// Every candidate was built and priced. + Exhaustive, + /// The dynamic program over target nesting, whose assumptions held. + TreeDp, + /// The dynamic program's result although its assumptions did not hold + /// and there were too many combinations to enumerate. + TreeDpNotGuaranteedOptimal { reason: String }, +} + +/// Costs are present for valid candidates only. +#[derive(Debug, Clone, PartialEq)] +pub struct Selection { + pub selected: String, + pub costs: BTreeMap, + pub rejected: Vec, + pub method: SelectionMethod, +} + +impl Selection { + /// Whether no valid candidate is cheaper than the selected one. + pub fn guaranteed_optimal(&self) -> bool { + !matches!( + self.method, + SelectionMethod::TreeDpNotGuaranteedOptimal { .. } + ) + } +} + +#[derive(Debug, Error)] +pub enum SelectionError { + #[error("no valid candidate: {0:?}")] + NoValidCandidate(Vec), + #[error("Stage 1: {0}")] + Stage1(#[from] LogicalCandidateError), +} + +/// Reject candidates that miss a query's accuracy target or cannot be priced, +/// price the rest and select the cheapest (the first on ties). `targets[i]` +/// is the requirement of `candidate.roots[i]`; `None` imposes none. +pub fn stage3_select( + cands: &[PhysicalCandidate], + targets: &[Option], + data: &DataWorkload, + models: PlanningModels<'_>, +) -> Result { + let mut costs = BTreeMap::new(); + let mut rejected = Vec::new(); + let mut best: Option<(&str, f64)> = None; + for candidate in cands { + match assess(candidate, targets, data, &models) { + Ok(cost) => { + if best.is_none_or(|(_, total)| cost.total < total) { + best = Some((&candidate.id, cost.total)); + } + costs.insert(candidate.id.clone(), cost); + } + Err(reason) => rejected.push(Rejection { + id: candidate.id.clone(), + valid: false, + reason, + }), + } + } + let Some((selected, best_total)) = best else { + return Err(SelectionError::NoValidCandidate(rejected)); + }; + for candidate in cands { + if candidate.id != selected && costs.contains_key(&candidate.id) { + rejected.push(Rejection { + id: candidate.id.clone(), + valid: true, + reason: format!( + "costlier: {:.3} vs {:.3} {COST_UNIT}", + costs[&candidate.id].total, best_total + ), + }); + } + } + Ok(Selection { + selected: selected.to_string(), + costs, + rejected, + method: SelectionMethod::Exhaustive, + }) +} + +/// Stage 3's checks and price for one candidate; `Err` is the rejection reason. +fn assess( + candidate: &PhysicalCandidate, + targets: &[Option], + data: &DataWorkload, + models: &PlanningModels<'_>, +) -> Result { + if candidate.roots.len() != targets.len() { + return Err(format!( + "{} roots but {} accuracy targets", + candidate.roots.len(), + targets.len() + )); + } + if let Some(reason) = accuracy_violation(candidate, targets, models) { + return Err(reason); + } + price(&candidate.dag, data).map_err(|(node, error)| format!("node {node:?}: {error}")) +} + +/// One Stage 1 sharing variant as selection sees it: candidates of variant +/// `v` are numbered after every candidate of the variants before it, so ids +/// stay unique across variants. +struct Variant<'a, Id> { + inventory: &'a LocalLogicalCandidates, + /// Identical sub-DAGs are merged, after composition too. + shared: bool, + /// Candidates numbered before this variant's. + offset: usize, +} + +// Manual impls: a derive would require `Id: Copy`. +impl Clone for Variant<'_, Id> { + fn clone(&self) -> Self { + *self + } +} +impl Copy for Variant<'_, Id> {} + +fn variants(stage1: &[SharingVariant]) -> Vec> { + let mut offset = 0; + stage1 + .iter() + .map(|v| { + let variant = Variant { + inventory: &v.inventory, + shared: v.shared, + offset, + }; + offset = offset.saturating_add(combination_count(&v.inventory)); + variant + }) + .collect() +} + +impl Variant<'_, Id> { + /// 1-based candidate number of `choice`: `L` and `P`. + fn number(&self, choice: &[usize]) -> usize { + self.offset + choice_index(self.inventory, choice) + 1 + } +} + +/// The independent variant alone: Pass 1 without Pass 2. +pub fn independent(inventory: LocalLogicalCandidates) -> Vec> { + vec![SharingVariant { + shared: false, + inventory, + }] +} + +/// Stage 1 → Stage 2 for `choice` in the independent variant, named +/// `P` from `L`. `Err` is the reason the candidate cannot +/// be built. +pub fn realize_choice( + inventory: &LocalLogicalCandidates, + choice: &[usize], +) -> Result<(Vec<(Id, QueryRoot)>, PhysicalCandidate), String> { + realize( + Variant { + inventory, + shared: false, + offset: 0, + }, + choice, + ) +} + +/// Stage 1 → Stage 2 for `choice` in `variant`. A shared variant also +/// merges identical sub-DAGs after composition, so queries that chose the +/// same summary producer reach one node; the returned Stage 1 candidate is +/// the merged one. +fn realize( + variant: Variant<'_, Id>, + choice: &[usize], +) -> Result<(Vec<(Id, QueryRoot)>, PhysicalCandidate), String> { + let index = variant.number(choice); + let mut logical = compose_logical_candidate(variant.inventory, choice) + .map_err(|e| format!("Stage 1: {e}"))?; + if variant.shared { + if let Some(merged) = share_identical_expressions(&logical) { + logical = merged; + } + } + let roots = logical + .iter() + .map(|(_, root)| match root { + QueryRoot::Operator(node) => Ok(node.clone()), + QueryRoot::Scalar(_) => Err("Stage 2: scalar query roots are not physical yet"), + }) + .collect::, _>>()?; + let mut candidate = + stage2_physical(&format!("L{index}"), &roots).map_err(|e| format!("Stage 2: {e}"))?; + candidate.id = format!("P{index}"); + candidate.label = format!("{choice:?}"); + Ok((logical, candidate)) +} + +/// One built combination; `physical` is `None` when it could not be built. +#[derive(Debug, Clone)] +pub struct EnumeratedCandidate { + /// From the shared variant (Pass 2's identical-expression rule). + pub shared: bool, + pub choice: Vec, + pub logical: Option>, + pub physical: Option, +} + +/// Every combination [`select_exhaustive`] built, and Stage 3 over them. +#[derive(Debug, Clone)] +pub struct Enumeration { + /// Over every variant. + pub combinations: usize, + pub candidates: Vec>, + pub selection: Selection, +} + +/// Build the first `max` combinations, variant by variant in enumeration +/// order, and select over them. One that cannot be built is rejected with +/// its reason. +pub fn select_exhaustive( + stage1: &[SharingVariant], + targets: &[Option], + data: &DataWorkload, + models: PlanningModels<'_>, + max: usize, +) -> Result, SelectionError> { + exhaustive(&variants(stage1), targets, data, models, max) +} + +fn exhaustive( + variants: &[Variant<'_, Id>], + targets: &[Option], + data: &DataWorkload, + models: PlanningModels<'_>, + max: usize, +) -> Result, SelectionError> { + let mut candidates = Vec::new(); + let mut failed = Vec::new(); + for variant in variants { + let left = max.saturating_sub(candidates.len()); + for choice in enumerate_choices(variant.inventory, left) { + let (logical, physical) = match realize(*variant, &choice) { + Ok((logical, physical)) => (Some(logical), Some(physical)), + Err(reason) => { + failed.push(Rejection { + id: format!("P{}", variant.number(&choice)), + valid: false, + reason, + }); + // Composition may have succeeded: keep it for display. + ( + compose_logical_candidate(variant.inventory, &choice).ok(), + None, + ) + } + }; + candidates.push(EnumeratedCandidate { + shared: variant.shared, + choice, + logical, + physical, + }); + } + } + let physical: Vec<_> = candidates + .iter() + .filter_map(|c| c.physical.clone()) + .collect(); + let selection = match stage3_select(&physical, targets, data, models) { + Ok(mut selection) => { + selection.rejected.extend(failed); + selection + } + Err(SelectionError::NoValidCandidate(mut rejected)) => { + rejected.extend(failed); + return Err(SelectionError::NoValidCandidate(rejected)); + } + Err(other) => return Err(other), + }; + Ok(Enumeration { + combinations: variants.iter().fold(0usize, |n, v| { + n.saturating_add(combination_count(v.inventory)) + }), + candidates, + selection, + }) +} + +/// The plan [`select_plan`] chose. +#[derive(Debug, Clone)] +pub struct SelectedPlan { + /// From the shared variant (Pass 2's identical-expression rule). + pub shared: bool, + pub choice: Vec, + /// The chosen Stage 1 candidate (identical sub-DAGs merged when `shared`). + pub logical: Vec<(Id, QueryRoot)>, + /// Stage 2 of `logical`; `roots` follow `logical`'s order. + pub physical: PhysicalCandidate, + pub selection: Selection, +} + +/// Relative tolerance when checking that costs add up. +const ADDITIVITY_TOLERANCE: f64 = 1e-9; + +/// Choose a sharing variant and one alternative per target. Sharing prices a +/// shared node once, which is not a sum of per-target changes, so the +/// variant is an outer choice: the dynamic program of [`select_variant`] +/// runs once per variant and the cheapest result wins (the first on ties). +pub fn select_plan( + stage1: &[SharingVariant], + targets: &[Option], + data: &DataWorkload, + models: PlanningModels<'_>, +) -> Result, SelectionError> { + let mut best: Option> = None; + let mut costs = BTreeMap::new(); + let mut rejected = Vec::new(); + let mut not_optimal = None; + let mut failures = Vec::new(); + for variant in variants(stage1) { + let evaluate = |choice: &[usize]| -> Result { + let (_, candidate) = realize(variant, choice)?; + assess(&candidate, targets, data, &models).map(|cost| cost.total) + }; + let plan = match select_variant(variant, targets, data, models, &evaluate) { + Ok(plan) => plan, + Err(SelectionError::NoValidCandidate(reasons)) => { + failures.extend(reasons); + continue; + } + Err(other) => return Err(other), + }; + if let SelectionMethod::TreeDpNotGuaranteedOptimal { reason } = &plan.selection.method { + not_optimal.get_or_insert_with(|| reason.clone()); + } + costs.extend(plan.selection.costs.clone()); + rejected.extend(plan.selection.rejected.clone()); + let total = |p: &SelectedPlan| p.selection.costs[&p.selection.selected].total; + match &best { + Some(current) if total(current) <= total(&plan) => rejected.push(costlier( + &plan.selection.selected, + total(&plan), + total(current), + )), + _ => { + if let Some(previous) = best.take() { + rejected.push(costlier( + &previous.selection.selected, + total(&previous), + total(&plan), + )); + } + best = Some(plan); + } + } + } + let Some(mut plan) = best else { + return Err(SelectionError::NoValidCandidate(failures)); + }; + rejected.extend(failures); + plan.selection.costs = costs; + plan.selection.rejected = rejected; + if let Some(reason) = not_optimal { + plan.selection.method = SelectionMethod::TreeDpNotGuaranteedOptimal { reason }; + } + Ok(plan) +} + +fn costlier(id: &str, total: f64, best: f64) -> Rejection { + Rejection { + id: id.to_string(), + valid: true, + reason: format!("costlier: {total:.3} vs {best:.3} {COST_UNIT}"), + } +} + +/// Choose one alternative per target of one variant by a dynamic program +/// over target nesting, then build the winner and check it in full. +/// +/// `best(t, c) = local(t, c) + Σ_{u beneath t, read by c} min_c' best(u, c')`, +/// where `local(t, c)` is the change in workload cost when only `t` takes +/// alternative `c`, and a choice is admissible when that one-target +/// candidate builds and passes Stage 3's checks. Most realizations read +/// their target's rewritten input, so they read every target beneath it; a +/// whole-expression alternative absorbs the target beneath instead +/// ([`read_targets`]), which then contributes nothing and takes its +/// pass-through. +/// +/// The result is the exhaustive minimum when (1) cost is a sum over nodes, +/// (2) a choice changes only its target's own nodes, and (3) shared nodes do +/// not depend on choices. Stage 3 prices per node and sizes every +/// realization of a target alike, so these hold unless a target's choice +/// changes what a target reading its output costs or whether it can be +/// built, or, in a shared variant, two targets reading one input build +/// identical producers that are then merged. That coupling is checked: +/// every pair of choices for a target and a target beneath it (and, when +/// shared, for two targets reading a common input) is built, and must cost +/// the sum of their single changes and be admissible exactly when both are. +/// On coupling, or when the winner fails the full check, every combination +/// of the variant is built instead if there are at most +/// [`MAX_ENUMERATED_CANDIDATES`]; otherwise the result is flagged as not +/// guaranteed optimal. +fn select_variant( + variant: Variant<'_, Id>, + targets: &[Option], + data: &DataWorkload, + models: PlanningModels<'_>, + evaluate: &dyn Fn(&[usize]) -> Result, +) -> Result, SelectionError> { + let inventory = variant.inventory; + let width = inventory.targets.len(); + let with = |changes: &[(usize, usize)]| { + let mut choice = vec![0; width]; + for &(target, alternative) in changes { + choice[target] = alternative; + } + choice + }; + // Alternative 0 is always the pass-through, so the base is the raw plan. + let base = match evaluate(&with(&[])) { + Ok(base) => base, + Err(reason) => { + return fallback(variant, targets, data, models, with(&[]), reason); + } + }; + let local: Vec>> = inventory + .targets + .iter() + .enumerate() + .map(|(t, target)| { + (0..target.alternatives.len()) + .map(|c| match c { + 0 => Ok(0.0), + _ => evaluate(&with(&[(t, c)])).map(|total| total - base), + }) + .collect() + }) + .collect(); + let beneath = nested_targets(inventory); + let reads: Vec>> = (0..width) + .map(|t| { + (0..local[t].len()) + .map(|c| read_targets(inventory, &beneath, t, c)) + .collect() + }) + .collect(); + let mut pairs: Vec<(usize, usize)> = Vec::new(); + for (t, by_choice) in reads.iter().enumerate() { + for &u in by_choice.iter().flatten() { + if !pairs.contains(&(t, u)) { + pairs.push((t, u)); + } + } + } + if variant.shared { + pairs.extend(common_input_pairs(inventory, &beneath)); + } + let mut coupling = None; + 'pairs: for &(t, u) in &pairs { + for c in 1..local[t].len() { + if !reads[t][c].contains(&u) { + // `c` does not read `u`'s output (it absorbs it, or reads a + // target beneath it only through another). + continue; + } + for d in 1..local[u].len() { + let joint = evaluate(&with(&[(t, c), (u, d)])); + let coupled = match (&local[t][c], &local[u][d], &joint) { + (Ok(a), Ok(b), Ok(joint)) => { + let expected = base + a + b; + (joint - expected).abs() + > ADDITIVITY_TOLERANCE * joint.abs().max(expected.abs()).max(1.0) + } + (Ok(_), Ok(_), Err(_)) | (Err(_), _, Ok(_)) | (_, Err(_), Ok(_)) => true, + _ => false, + }; + if coupled { + coupling = Some(format!( + "target {t} alternative {c} and target {u} alternative {d} do not \ + combine additively" + )); + break 'pairs; + } + } + } + } + let mut best: Vec> = vec![None; width]; + for t in 0..width { + best_choice(t, &local, &reads, &mut best); + } + let mut choice: Vec = best.iter().map(|b| b.map_or(0, |(_, c)| c)).collect(); + for t in 0..width { + if let Some(u) = inventory.targets[t].absorbs[choice[t]] { + choice[u] = 0; + } + } + if let Some(reason) = coupling { + return fallback(variant, targets, data, models, choice, reason); + } + match finish( + variant, + targets, + data, + &models, + choice.clone(), + SelectionMethod::TreeDp, + ) { + Ok(plan) => Ok(plan), + Err(reason) => fallback(variant, targets, data, models, choice, reason), + } +} + +/// Pairs of targets, neither beneath the other, that read a common input +/// node: in a shared variant their producers may be merged. +fn common_input_pairs( + inventory: &LocalLogicalCandidates, + beneath: &[Vec], +) -> Vec<(usize, usize)> { + let inputs: Vec> = inventory + .targets + .iter() + .map(|t| t.target.children().into_iter().map(Rc::as_ptr).collect()) + .collect(); + let mut pairs = Vec::new(); + for t in 0..inputs.len() { + for u in t + 1..inputs.len() { + let nested = beneath[t].contains(&u) || beneath[u].contains(&t); + if !nested && inputs[t].iter().any(|p| inputs[u].contains(p)) { + pairs.push((t, u)); + } + } + } + pairs +} + +/// `best(t) = min_c local(t, c) + Σ_{u read by c} best(u)`, memoized; the +/// first alternative wins ties, as in enumeration order. Alternative 0 (the +/// pass-through) is admissible whenever the base plan is. `reads[t][c]` is +/// [`read_targets`]. +fn best_choice( + t: usize, + local: &[Vec>], + reads: &[Vec>], + best: &mut [Option<(f64, usize)>], +) -> f64 { + if let Some((cost, _)) = best[t] { + return cost; + } + let mut inner = Vec::with_capacity(local[t].len()); + for read in &reads[t] { + inner.push( + read.iter() + .map(|&u| best_choice(u, local, reads, best)) + .sum::(), + ); + } + let (cost, choice) = local[t] + .iter() + .enumerate() + .filter_map(|(c, cost)| cost.as_ref().ok().map(|cost| (cost + inner[c], c))) + .fold( + (f64::INFINITY, 0), + |min, next| if next.0 < min.0 { next } else { min }, + ); + best[t] = Some((cost, choice)); + cost +} + +/// Build `choice` in `variant` and run Stage 3 on it. +fn finish( + variant: Variant<'_, Id>, + targets: &[Option], + data: &DataWorkload, + models: &PlanningModels<'_>, + choice: Vec, + method: SelectionMethod, +) -> Result, String> { + let (logical, physical) = realize(variant, &choice)?; + let cost = assess(&physical, targets, data, models)?; + Ok(SelectedPlan { + shared: variant.shared, + choice, + logical, + selection: Selection { + selected: physical.id.clone(), + costs: BTreeMap::from([(physical.id.clone(), cost)]), + rejected: Vec::new(), + method, + }, + physical, + }) +} + +/// Selection when the dynamic program's result cannot be trusted: every +/// combination of the variant if there are few, else `choice` flagged with +/// `reason`. +fn fallback( + variant: Variant<'_, Id>, + targets: &[Option], + data: &DataWorkload, + models: PlanningModels<'_>, + choice: Vec, + reason: String, +) -> Result, SelectionError> { + if combination_count(variant.inventory) <= MAX_ENUMERATED_CANDIDATES { + let enumeration = exhaustive(&[variant], targets, data, models, MAX_ENUMERATED_CANDIDATES)?; + let winner = enumeration + .candidates + .iter() + .find(|c| { + c.physical + .as_ref() + .is_some_and(|p| p.id == enumeration.selection.selected) + }) + .expect("the selected candidate was built"); + let mut plan = finish( + variant, + targets, + data, + &models, + winner.choice.clone(), + SelectionMethod::Exhaustive, + ) + .map_err(|reason| { + SelectionError::NoValidCandidate(vec![Rejection { + id: enumeration.selection.selected.clone(), + valid: false, + reason, + }]) + })?; + plan.selection = Selection { + costs: enumeration.selection.costs, + rejected: enumeration.selection.rejected, + ..plan.selection + }; + return Ok(plan); + } + let method = SelectionMethod::TreeDpNotGuaranteedOptimal { + reason: reason.clone(), + }; + finish(variant, targets, data, &models, choice.clone(), method).map_err(|failure| { + SelectionError::NoValidCandidate(vec![Rejection { + id: format!("P{}", variant.number(&choice)), + valid: false, + reason: format!("{reason}; {failure}"), + }]) + }) +} + +/// What [`plan_stages`] produced. +#[derive(Debug, Clone)] +pub struct StagePipelineRun { + /// Stage 1: Pass 1's alternatives per sharing variant (Pass 2). + pub stage1: Vec>, + /// Stages 2 and 3 for the selected candidate ([`select_plan`]). + pub plan: SelectedPlan, + /// Every candidate built and priced, when requested for display. + pub enumeration: Option>, +} + +/// The #509 stage pipeline over `roots`: Stage 1 (Pass 1 and Pass 2's +/// identical-expression rule), Stage 2 and Stage 3. The facade and the +/// `stage_pipeline` devtool both run this. `display` builds and prices up to +/// that many candidates for display as well (0: none). +pub fn plan_stages( + roots: Vec<(Id, QueryRoot)>, + targets: &[Option], + data: &DataWorkload, + models: PlanningModels<'_>, + display: usize, +) -> Result, SelectionError> { + let stage1 = stage1_logical_candidates(roots)?; + let plan = select_plan(&stage1, targets, data, models)?; + let enumeration = match display { + 0 => None, + max => Some(select_exhaustive(&stage1, targets, data, models, max)?), + }; + Ok(StagePipelineRun { + stage1, + plan, + enumeration, + }) +} + +/// The first summary estimate that misses its query's target, as a reason. +fn accuracy_violation( + candidate: &PhysicalCandidate, + targets: &[Option], + models: &PlanningModels<'_>, +) -> Option { + for (query, (root, target)) in candidate.roots.iter().zip(targets).enumerate() { + let Some(target) = target else { continue }; + for node in OperatorNode::reachable(root) { + let Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input, + query: statistic, + }) = &node.operator + else { + continue; + }; + let Operator::ASAP(ASAPOp::SummaryAgg { family, input, .. }) = &summary_input.operator + else { + return Some(format!("q{}: estimate over a non-summary input", query + 1)); + }; + let name = family_name(family); + // Count-Min's one-sided error bound assumes no negative updates. + if matches!(family, FieldDataType::Sketch(kind, _) + if matches!(kind.algorithm(), SketchAlgorithm::Cms | SketchAlgorithm::CmsWithHeap)) + && !matches!(input.weight_domain, WeightDomain::NonNegative { .. }) + { + return Some(format!( + "q{}: {name} needs non-negative update weights, and these are not proven \ + non-negative", + query + 1 + )); + } + let Some(guarantee) = models.accuracy.local_guarantee(family, statistic) else { + return Some(format!( + "q{}: no accuracy model for {name}; target {target:?}", + query + 1 + )); + }; + if !models.accuracy.satisfies(&guarantee, target) { + return Some(format!( + "q{}: {name} guarantees bound {:?}, failure probability {:?}, which misses \ + target {target:?} (analytical guarantee; no accuracy evidence)", + query + 1, + guarantee.bound.evaluate(), + guarantee.failure_probability.evaluate(), + )); + } + } + } + None +} + +fn family_name(family: &FieldDataType) -> String { + match family { + FieldDataType::Sketch(kind, _) => format!("{:?}", kind.algorithm()), + FieldDataType::ExactAggregate(kind, _) => format!("exact {kind:?} accumulator"), + other => format!("{other:?}"), + } +} + +/// Statistics the analytical model needs, derived once per workload. +struct Shape { + series: u64, + rows_per_ms: f64, +} + +/// Price every node of `dag` once. Nodes are exported children first, so +/// each node's input statistics are known when it is reached. +fn price( + dag: &PhysicalASAPDAG, + data: &DataWorkload, +) -> Result { + let series = data + .input_cardinality + .value + .unwrap_or(DEFAULT_SERIES) + .max(1); + let shape = Shape { + series, + rows_per_ms: data + .ingestion_rate + .value + .map_or(DEFAULT_ROWS_PER_SECOND, |rate| rate.0) + / 1_000.0, + }; + let calibration = ResourceCalibration { + cost_per_cpu_op: 1e-6, + cost_per_scan_byte: 1e-7, + cost_per_retained_byte: 0.0, + version: "illustrative-v1".into(), + }; + let nodes: HashMap<_, _> = dag.nodes.iter().map(|n| (n.id, n)).collect(); + let mut output: HashMap = HashMap::new(); + let mut per_node = BTreeMap::new(); + for node in &dag.nodes { + let inputs: Vec<_> = dag + .edges + .iter() + .filter(|e| e.consumer == node.id) + .map(|e| output[&e.producer]) + .collect(); + let input = inputs + .first() + .copied() + .unwrap_or(EdgeStatistics { rows: 1, bytes: 1 }); + let width = row_bytes(&node.output_schema); + let edge = |rows: u64| EdgeStatistics { + rows, + bytes: rows * width, + }; + let unary = |output| UnaryEdgeStatistics { + input, + output, + promql: None, + }; + let groups = |reduction: &Reduction| { + match reduction { + Reduction::Reduce(keys) if keys.keys().is_empty() && !keys.is_without() => 1, + Reduction::Reduce(keys) if !keys.is_without() => DEFAULT_GROUP_COUNT, + _ => shape.series, + } + .min(input.rows.max(1)) + }; + let (out, estimate, detail) = match &node.payload { + Payload::Relational { operator } => match operator { + NonASAPOpKind::Scan { .. } => { + // A scan reads what its time range keeps. + let lookback = dag + .edges + .iter() + .filter(|e| e.producer == node.id) + .filter_map(|e| match &nodes[&e.consumer].payload { + Payload::Relational { + operator: NonASAPOpKind::TimeRange { range, .. }, + } => Some(range.as_millis() as u64), + _ => None, + }) + .max() + .unwrap_or(DEFAULT_LOOKBACK_MS); + let out = edge(((shape.rows_per_ms * lookback as f64).round() as u64).max(1)); + let estimate = estimate_operator( + PhysicalOperator::Scan, + OperatorStatistics::Scan { + edges: UnaryEdgeStatistics { + input: out, + output: out, + promql: None, + }, + source_read_bytes: out.bytes, + }, + ); + (out, estimate, format!("scan {} samples", out.rows)) + } + NonASAPOpKind::Aggregate { + reduction, + measures, + .. + } => { + let group_count = groups(reduction); + let keys = match reduction { + Reduction::Reduce(keys) => keys.keys().len() as u64, + _ => 1, + }; + let out = edge(group_count); + let estimate = estimate_operator( + PhysicalOperator::HashAggregate { + grouping_key_count: keys, + accumulator_count: measures.len().max(1) as u64, + }, + OperatorStatistics::HashAggregate { + edges: unary(out), + group_count, + key_bytes: 16 * keys, + accumulator_bytes_per_group: 8, + }, + ); + ( + out, + estimate, + format!( + "hash aggregate {} rows into {group_count} groups", + input.rows + ), + ) + } + NonASAPOpKind::Sort { keys, partition_by } => { + let partitions = if partition_by.keys().is_empty() { + 1 + } else { + DEFAULT_GROUP_COUNT.min(input.rows.max(1)) + }; + let estimate = estimate_operator( + PhysicalOperator::InMemoryComparisonSort { + ordering_key_count: keys.len() as u64, + partitioned: partitions > 1, + }, + OperatorStatistics::InMemoryComparisonSort { + edges: unary(input), + input_partitioning: split(input, partitions), + }, + ); + ( + input, + estimate, + format!("sort {} rows in {partitions} partitions", input.rows), + ) + } + NonASAPOpKind::Limit { + n, + offset, + partition_by, + } => { + let partitions = partition_count(!partition_by.keys().is_empty()); + let limit = n.map_or(u64::MAX, |n| (n as u64).saturating_mul(partitions)); + let offset = (*offset as u64).saturating_mul(partitions); + let out = edge(selected_rows(input.rows.saturating_sub(offset), limit)); + let estimate = estimate_operator( + PhysicalOperator::Limit { limit, offset }, + OperatorStatistics::Limit { edges: unary(out) }, + ); + (out, estimate, format!("limit to {} rows", out.rows)) + } + other => { + let estimate = estimate_operator( + PhysicalOperator::PassThrough, + OperatorStatistics::PassThrough { + edges: unary(input), + }, + ); + let name = match other { + NonASAPOpKind::TimeRange { range, .. } => format!("time range {range:?}"), + _ => "operator".into(), + }; + (input, estimate, format!("{name}: pass {} rows", input.rows)) + } + }, + Payload::SummaryAgg { + family, reduction, .. + } => { + let group_count = groups(reduction); + let (depth, state_bytes) = summary_shape(family); + let out = EdgeStatistics { + rows: group_count, + bytes: group_count * state_bytes, + }; + let ops = input.rows as f64 * depth as f64; + ( + out, + Ok(ResourceEstimate::new(ops, out.bytes, 0)), + format!( + "build {} into {group_count} states: {} rows x depth {depth}", + family_name(family), + input.rows + ), + ) + } + Payload::SummaryEstimate { query } => { + let rows = match query { + // The logical result, as an exact Sort → Limit sizes it. + // Items are at most the series: a whole-expression + // sketch reads several samples per ranked item. + SketchStatistic::TopK { k } => { + let (summarized, grouped) = summarized_rows(dag, &output, node.id); + selected_rows( + summarized.min(shape.series), + (*k as u64).saturating_mul(partition_count(grouped)), + ) + } + _ => input.rows, + }; + let out = edge(rows); + ( + out, + Ok(ResourceEstimate::new(out.rows as f64, 0, 0)), + format!("estimate {} rows from {} states", out.rows, input.rows), + ) + } + Payload::FinalizeExactAccumulator => ( + edge(input.rows), + Ok(ResourceEstimate::new(input.rows as f64, 0, 0)), + format!("finalize {} accumulators", input.rows), + ), + _ => ( + edge(input.rows), + Ok(ResourceEstimate::new(input.rows as f64, 0, 0)), + format!("{} rows", input.rows), + ), + }; + let cost = estimate + .and_then(|estimate| estimate.calibrated_cost(&calibration)) + .map_err(|error| (node.id, error))?; + output.insert(node.id, out); + per_node.insert( + node.id, + NodeCost { + cost, + rows: out.rows, + detail, + }, + ); + } + Ok(CandidateCost { + total: per_node.values().map(|n| n.cost).sum(), + unit: COST_UNIT, + source: COST_SOURCE, + per_node, + }) +} + +/// Partitions a per-group ranking assumes, absent group-count evidence. +fn partition_count(grouped: bool) -> u64 { + if grouped { + DEFAULT_GROUP_COUNT + } else { + 1 + } +} + +/// Rows a limit of `limit` keeps from `input` rows. Every top-k realization +/// is sized by this, so its consumers are priced alike whichever is chosen. +fn selected_rows(input: u64, limit: u64) -> u64 { + input.min(limit) +} + +/// The rows the summary under estimate `id` read, and whether it groups them. +fn summarized_rows( + dag: &PhysicalASAPDAG, + output: &HashMap, + id: PhysicalASAPNodeId, +) -> (u64, bool) { + let producer = |consumer| { + dag.edges + .iter() + .find(|e| e.consumer == consumer) + .map(|e| e.producer) + }; + let state = producer(id); + let grouped = state + .and_then(|state| dag.nodes.iter().find(|n| n.id == state)) + .is_some_and(|n| match &n.payload { + Payload::SummaryAgg { reduction, .. } => { + !matches!(reduction, Reduction::Reduce(keys) if keys.keys().is_empty() && !keys.is_without()) + } + _ => false, + }); + let rows = state + .and_then(producer) + .and_then(|input| output.get(&input)) + .map_or(1, |edge| edge.rows); + (rows, grouped) +} + +/// `input` split as evenly as integers allow into `partitions` parts. +fn split(input: EdgeStatistics, partitions: u64) -> PartitionStatistics { + let part = |total: u64, i: u64| total / partitions + u64::from(i < total % partitions); + PartitionStatistics { + partitions: (0..partitions) + .map(|i| EdgeStatistics { + rows: part(input.rows, i), + bytes: part(input.bytes, i), + }) + .collect(), + } +} + +/// Plain values are 8 bytes, strings 16; summary columns are sized apart. +fn row_bytes(schema: &Schema) -> u64 { + schema + .fields + .iter() + .map(|f| match &f.dtype { + FieldDataType::Plain(DataType::Utf8) => 16, + _ => 8, + }) + .sum::() + .max(1) +} + +/// Update operations per input row and bytes per state. +fn summary_shape(family: &FieldDataType) -> (u64, u64) { + match family { + FieldDataType::Sketch(kind, _) => match kind.params() { + SketchParams::Cms { width, depth } | SketchParams::CountSketch { width, depth } => { + (u64::from(*depth), 8 * u64::from(*width) * u64::from(*depth)) + } + SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } + | SketchParams::CountSketchWithHeap { + width, + depth, + heap_size, + } => ( + u64::from(*depth) + 1, + 8 * u64::from(*width) * u64::from(*depth) + 24 * u64::from(*heap_size), + ), + _ => (1, 1_024), + }, + _ => (1, 8), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::test_support::lower_promql; + use asap_physical_optimizer::implementation::physical_candidates::stage2_physical; + use asap_types::ir::QueryRoot; + use asap_types::workload::{Evidence, Rate}; + + fn data() -> DataWorkload { + DataWorkload { + ingestion_rate: Evidence { + value: Some(Rate(10_000.0)), + ..Default::default() + }, + input_cardinality: Evidence { + value: Some(10_000), + ..Default::default() + }, + ..Default::default() + } + } + + /// Exact (P1), CMS+heap (P2) and CountSketch+heap (P3) realizations of + /// one approximate top-k query, in Pass 1 catalog order. + fn candidates() -> Vec { + let root = lower_promql( + "topk by (job) (10, sum_over_time(m[1m]))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }, + ); + let inventory = + asap_logical_optimizer::pass1::logical_candidates::enumerate_local_logical_candidates( + vec![(0, QueryRoot::Operator(root))], + ) + .unwrap(); + let topk = inventory + .targets + .iter() + .position(|t| t.alternatives.len() > 2) + .unwrap(); + [0, 1, 2] + .into_iter() + .map(|alternative| { + let mut choice = vec![0; inventory.targets.len()]; + choice[topk] = alternative; + let roots: Vec<_> = + asap_logical_optimizer::pass1::logical_candidates::compose_logical_candidate( + &inventory, &choice, + ) + .unwrap() + .into_iter() + .map(|(_, root)| match root { + QueryRoot::Operator(node) => node, + QueryRoot::Scalar(_) => panic!("operator root"), + }) + .collect(); + let mut candidate = stage2_physical("L", &roots).unwrap(); + candidate.id = format!("P{}", alternative + 1); + candidate + }) + .collect() + } + + /// A summary whose analytical guarantee misses the target is rejected as + /// invalid with a reason; the exact plan is then selected. + #[test] + fn accuracy_failing_candidate_is_rejected_as_invalid() { + let candidates = candidates(); + let strict = AccuracyTarget::EpsilonDelta { + epsilon: 1e-6, + delta: 1e-9, + }; + let selection = stage3_select( + &candidates, + &[Some(strict)], + &data(), + PlanningModels::builtin(), + ) + .unwrap(); + assert_eq!(selection.selected, "P1"); + let rejected: BTreeMap<_, _> = selection + .rejected + .iter() + .map(|r| (r.id.as_str(), r)) + .collect(); + assert_eq!(rejected.len(), 2); + assert!(!rejected["P3"].valid); + assert!( + rejected["P3"].reason.contains("misses target"), + "{}", + rejected["P3"].reason + ); + assert!(!selection.costs.contains_key("P3")); + } + + /// Count-Min over weights not proven non-negative is invalid whatever + /// the target. + #[test] + fn count_min_over_signed_weights_is_rejected_as_invalid() { + let target = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }; + let selection = stage3_select( + &candidates(), + &[Some(target)], + &data(), + PlanningModels::builtin(), + ) + .unwrap(); + let p2 = selection.rejected.iter().find(|r| r.id == "P2").unwrap(); + assert!(!p2.valid); + assert!(p2.reason.contains("non-negative"), "{}", p2.reason); + } + + fn inventory(queries: &[&str]) -> LocalLogicalCandidates { + let target = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }; + let roots = queries + .iter() + .enumerate() + .map(|(i, query)| { + let root = lower_promql(query, target.clone()); + let root = + asap_types::ir::schema_support::with_promql_series_identity(&root).unwrap(); + (i, QueryRoot::Operator(root)) + }) + .collect(); + asap_logical_optimizer::pass1::logical_candidates::enumerate_local_logical_candidates(roots) + .unwrap() + } + + fn no_targets(inventory: &LocalLogicalCandidates) -> Vec> { + vec![None; inventory.roots.len()] + } + + /// Real Stage 1 → 3 cost plus a penalty whenever the two named targets + /// both leave their pass-through: an inner choice that changes the cost + /// of the target reading it. + fn coupled<'a>( + inventory: &'a LocalLogicalCandidates, + targets: &'a [Option], + data: &'a DataWorkload, + (outer, inner): (usize, usize), + ) -> impl Fn(&[usize]) -> Result + 'a { + move |choice: &[usize]| { + let (_, candidate) = realize_choice(inventory, choice)?; + let total = assess(&candidate, targets, data, &PlanningModels::builtin())?.total; + Ok(total + + if choice[outer] > 0 && choice[inner] > 0 { + 1.0 + } else { + 0.0 + }) + } + } + + /// The outer target and the target beneath it. + fn nested_pair(inventory: &LocalLogicalCandidates) -> (usize, usize) { + let beneath = nested_targets(inventory); + beneath + .iter() + .enumerate() + .find_map(|(t, inner)| inner.first().map(|&u| (t, u))) + .unwrap() + } + + /// Coupling between a target and the target it reads, over at most 64 + /// combinations, falls back to building every combination. + #[test] + fn coupling_over_few_combinations_selects_exhaustively() { + let inventory = inventory(&["count(topk by (job) (10, sum_over_time(m[1m])))"]); + assert!(combination_count(&inventory) <= MAX_ENUMERATED_CANDIDATES); + let targets = no_targets(&inventory); + let data = data(); + let evaluate = coupled(&inventory, &targets, &data, nested_pair(&inventory)); + let plan = select_variant( + Variant { + inventory: &inventory, + shared: false, + offset: 0, + }, + &targets, + &data, + PlanningModels::builtin(), + &evaluate, + ) + .unwrap(); + assert_eq!(plan.selection.method, SelectionMethod::Exhaustive); + let exhaustive = select_exhaustive( + &independent(inventory.clone()), + &targets, + &data, + PlanningModels::builtin(), + MAX_ENUMERATED_CANDIDATES, + ) + .unwrap(); + assert_eq!(plan.selection.selected, exhaustive.selection.selected); + } + + /// Coupling over more than 64 combinations keeps the dynamic program's + /// result and flags it as not guaranteed optimal. + #[test] + fn coupling_over_many_combinations_is_flagged() { + let inventory = inventory(&[ + "count(topk by (job) (10, sum_over_time(m[1m])))", + "sum by (job) (rate(m[1m]))", + ]); + assert!(combination_count(&inventory) > MAX_ENUMERATED_CANDIDATES); + let targets = no_targets(&inventory); + let data = data(); + let evaluate = coupled(&inventory, &targets, &data, nested_pair(&inventory)); + let plan = select_variant( + Variant { + inventory: &inventory, + shared: false, + offset: 0, + }, + &targets, + &data, + PlanningModels::builtin(), + &evaluate, + ) + .unwrap(); + assert!( + matches!( + &plan.selection.method, + SelectionMethod::TreeDpNotGuaranteedOptimal { reason } if reason.contains("additively") + ), + "{:?}", + plan.selection.method + ); + assert!(!plan.selection.guaranteed_optimal()); + } + + /// Without coupling the dynamic program's result stands, and it is the + /// exhaustive minimum. + #[test] + fn uncoupled_selection_uses_the_dynamic_program() { + let inventory = inventory(&["count(topk by (job) (10, sum_over_time(m[1m])))"]); + let targets = no_targets(&inventory); + let plan = select_plan( + &independent(inventory.clone()), + &targets, + &data(), + PlanningModels::builtin(), + ) + .unwrap(); + assert_eq!(plan.selection.method, SelectionMethod::TreeDp); + let exhaustive = select_exhaustive( + &independent(inventory.clone()), + &targets, + &data(), + PlanningModels::builtin(), + MAX_ENUMERATED_CANDIDATES, + ) + .unwrap(); + assert_eq!(plan.selection.selected, exhaustive.selection.selected); + } + + /// A whole-expression top-k absorbs the `sum_over_time` beneath it. When + /// it is the cheapest choice, the dynamic program selects it with the + /// inner target at its pass-through (it contributes nothing), and agrees + /// with brute force over every valid choice under the same costs. + #[test] + fn absorbing_alternative_drops_the_inner_target() { + let inventory = inventory(&["topk by (job) (10, sum_over_time(m[1m]))"]); + let (t, c, u) = inventory + .targets + .iter() + .enumerate() + .find_map(|(t, target)| { + let c = target.alternatives.iter().enumerate().position(|(c, a)| { + target.absorbs[c].is_some() + && matches!(a, asap_logical_optimizer::Realization::Sketch(kind) + if *kind.algorithm() == SketchAlgorithm::CountSketchWithHeap) + })?; + Some((t, c, target.absorbs[c]?)) + }) + .expect("a whole-expression CountSketch alternative"); + let targets = no_targets(&inventory); + let data = data(); + for bonus in [0.0, 1e4] { + let evaluate = |choice: &[usize]| -> Result { + let (_, candidate) = realize_choice(&inventory, choice)?; + let total = assess(&candidate, &targets, &data, &PlanningModels::builtin())?.total; + Ok(total - if choice[t] == c { bonus } else { 0.0 }) + }; + let plan = select_variant( + Variant { + inventory: &inventory, + shared: false, + offset: 0, + }, + &targets, + &data, + PlanningModels::builtin(), + &evaluate, + ) + .unwrap(); + assert_eq!(plan.selection.method, SelectionMethod::TreeDp); + let brute = enumerate_choices(&inventory, usize::MAX) + .into_iter() + .map(|choice| (evaluate(&choice).unwrap(), choice)) + .fold(None::<(f64, Vec)>, |best, next| match best { + Some(best) if best.0 <= next.0 => Some(best), + _ => Some(next), + }) + .unwrap(); + assert_eq!(plan.choice, brute.1, "bonus {bonus}"); + if bonus > 0.0 { + assert_eq!((plan.choice[t], plan.choice[u]), (c, 0)); + } + } + } + + /// In a shared variant, two queries that chose structurally identical + /// summary producers reach one state. + #[test] + fn identical_producers_are_merged_after_composition() { + let inventory = inventory(&[ + "quantile_over_time(0.5, m[5m])", + "quantile_over_time(0.99, m[5m])", + ]); + let kll = |t: &asap_logical_optimizer::pass1::logical_candidates::LocalLogicalTarget| { + t.alternatives + .iter() + .position(|a| matches!(a, asap_logical_optimizer::Realization::Sketch(kind) if *kind.algorithm() == SketchAlgorithm::Kll)) + .unwrap() + }; + let choice: Vec<_> = inventory.targets.iter().map(kll).collect(); + let plan = finish( + Variant { + inventory: &inventory, + shared: true, + offset: 0, + }, + &no_targets(&inventory), + &data(), + &PlanningModels::builtin(), + choice, + SelectionMethod::Exhaustive, + ) + .unwrap(); + let states: std::collections::HashSet<_> = plan + .physical + .roots + .iter() + .flat_map(OperatorNode::reachable) + .filter(|n| matches!(n.operator, Operator::ASAP(ASAPOp::SummaryAgg { .. }))) + .map(|n| std::rc::Rc::as_ptr(&n)) + .collect(); + assert_eq!(states.len(), 1); + } + + /// A candidate that cannot be checked is rejected with its reason; the + /// others are still selected among. + #[test] + fn a_candidate_that_cannot_be_checked_is_rejected_not_fatal() { + let mut candidates = candidates(); + let mut extra = candidates[0].clone(); + extra.id = "P4".into(); + extra.roots.push(extra.roots[0].clone()); + candidates.push(extra); + let target = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }; + let selection = stage3_select( + &candidates, + &[Some(target)], + &data(), + PlanningModels::builtin(), + ) + .unwrap(); + let p4 = selection.rejected.iter().find(|r| r.id == "P4").unwrap(); + assert!(!p4.valid); + assert!(p4.reason.contains("accuracy targets"), "{}", p4.reason); + } + + /// Every top-k realization reports the same output rows, as the logical + /// result sizes them, whether the input holds fewer rows than k × groups + /// or more. + #[test] + fn every_topk_realization_reports_the_same_output_rows() { + for series in [3, 1_000_000] { + let data = DataWorkload { + input_cardinality: asap_types::workload::Evidence { + value: Some(series), + ..Default::default() + }, + ..data() + }; + let rows: Vec = candidates() + .iter() + .map(|candidate| { + let cost = price(&candidate.dag, &data).unwrap(); + cost.per_node[&candidate.dag.roots[0]].rows + }) + .collect(); + assert!(rows.windows(2).all(|w| w[0] == w[1]), "{series}: {rows:?}"); + } + } + + /// Each node is priced once, under its DAG id, and the total is the sum; + /// every candidate is either selected or rejected as costlier. + #[test] + fn per_node_costs_cover_the_dag_and_sum_to_the_total() { + let candidates = candidates(); + let target = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }; + let selection = stage3_select( + &candidates, + &[Some(target)], + &data(), + PlanningModels::builtin(), + ) + .unwrap(); + for candidate in candidates.iter().filter(|c| c.id != "P2") { + let cost = &selection.costs[&candidate.id]; + let ids: Vec<_> = candidate.dag.nodes.iter().map(|n| n.id).collect(); + let mut keys: Vec<_> = cost.per_node.keys().copied().collect(); + keys.sort(); + let mut sorted = ids.clone(); + sorted.sort(); + assert_eq!(keys, sorted); + let sum: f64 = cost.per_node.values().map(|n| n.cost).sum(); + assert_eq!(cost.total, sum); + assert!(cost.total > 0.0); + } + let costlier: Vec<_> = selection.rejected.iter().filter(|r| r.valid).collect(); + assert_eq!(costlier.len(), 1); + assert_ne!(costlier[0].id, selection.selected); + } +} diff --git a/crates/plan-selection/src/test_support.rs b/crates/plan-selection/src/test_support.rs new file mode 100644 index 000000000..1c0cefb41 --- /dev/null +++ b/crates/plan-selection/src/test_support.rs @@ -0,0 +1,124 @@ +// Fixture helpers for this crate's tests: the subset of +// `asap-logical-optimizer`'s `test_support` that Stage 2/3 tests use. + +use std::rc::Rc; + +use asap_types::ir::OperatorNode; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, + Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, +}; + +pub(crate) fn lower_promql(query: &str, accuracy: AccuracyTarget) -> Rc { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .pop() + .unwrap() +} + +// ── Shared pre-ASAP fixture builders ───────────────────────────────────── +// +// Every builder returns an `Rc` whose schema is derived by +// `OperatorNode::new_shared`, so a fixture is exactly what a front end +// would hand the planner. + +use asap_types::ir::operator::agg_intent::AggIntent; +use asap_types::ir::operator::operator_properties::{Reduction, Source}; +use asap_types::ir::schema::{ColumnId, DataType, Field, Schema}; +use asap_types::ir::{NonASAPOp, Predicate}; + +/// A `TimeSeries("m")` scan over `[ts(0), value(1), labels...]`, time index 0, +/// no unique key. +pub(crate) fn metric_scan(labels: &[&str]) -> Rc { + metric_scan_with_keys(labels, vec![]) +} + +/// [`metric_scan`] with explicit `unique_keys` (a `[[0]]` key makes CSE +/// willing to hoist the scan). +pub(crate) fn metric_scan_with_keys( + labels: &[&str], + unique_keys: Vec>, +) -> Rc { + let mut columns = vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + ]; + columns.extend( + labels + .iter() + .map(|n| Field::plain(*n, DataType::Utf8, true)), + ); + scan("m", Schema::with_time_index(columns, 0, unique_keys)) +} + +/// A predicate-free `TimeSeries(metric)` scan with the given schema. +pub(crate) fn scan(metric: &str, schema: Schema) -> Rc { + scan_from( + Source::TimeSeries { + metric: metric.into(), + }, + schema, + ) +} + +pub(crate) fn scan_from(source: Source, schema: Schema) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { + source, + predicates: vec![], + schema, + })) + .unwrap() +} + +/// A general aggregate node. +pub(crate) fn aggregate( + reduction: Reduction, + measures: Vec, + output_names: Vec, + having: Option, + child: Rc, +) -> Rc { + OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction, + measures, + output_names, + filters: vec![], + having, + child, + })) + .unwrap() +} + +/// `intent by (by)` — a single-measure, `HAVING`-free grouped aggregate. +pub(crate) fn agg( + by: Vec, + intent: AggIntent, + child: Rc, +) -> Rc { + aggregate(Reduction::by(by), vec![intent], vec![], None, child) +} diff --git a/crates/asap-aware-mapping/tests/data/offline-evidence-synthetic.json b/crates/plan-selection/tests/data/offline-evidence-synthetic.json similarity index 100% rename from crates/asap-aware-mapping/tests/data/offline-evidence-synthetic.json rename to crates/plan-selection/tests/data/offline-evidence-synthetic.json diff --git a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs b/crates/plan-selection/tests/physical_handoff_cost.rs similarity index 95% rename from crates/asap-aware-mapping/tests/physical_handoff_cost.rs rename to crates/plan-selection/tests/physical_handoff_cost.rs index 0885db71c..79fa1bcf3 100644 --- a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs +++ b/crates/plan-selection/tests/physical_handoff_cost.rs @@ -1,18 +1,18 @@ -use asap_aware_mapping::analytical_cost::{ +use asap_plan_selection::cost::analytical_cost::{ EvidenceBackedPhysicalDAG, ExecutionMultiplicity, PhysicalDAGNode, PhysicalNodeEvidence, PhysicalOperator, }; -use asap_aware_mapping::physical_operator_statistics::{ - ComparisonScope, EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, +use asap_plan_selection::cost::physical_operator_statistics::{ + ComparisonScope, EdgeStatistics, OperatorStatistics, ScanSelection, UnaryEdgeStatistics, }; -use asap_types::pre_asap::query_expr::Source; +use asap_types::ir::operator::operator_properties::Source; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; use std::collections::HashMap; fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { - let coverage = SourceCoverage { + let coverage = ScanSelection { source: Source::Table { table_ref: "events".into(), }, @@ -49,7 +49,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -58,7 +58,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "left".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -67,7 +67,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "right".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -76,7 +76,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "root".into(), operator: PhysicalOperator::Concat, children: vec!["left".into(), "right".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -136,12 +136,12 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { ) } -use asap_aware_mapping::physical_handoff_cost::*; +use asap_plan_selection::cost::physical_handoff_cost::*; // Legacy mapping imports and the shared resource API are the very same Rust types. #[test] fn mapping_resource_reexports_are_wire_compatible_shared_types() { - let shared = asap_types::resources::PhysicalHandoffBytes { + let shared = asap_types::workload::resources::PhysicalHandoffBytes { network_bytes: 480, materialization_bytes: 40, }; @@ -150,7 +150,7 @@ fn mapping_resource_reexports_are_wire_compatible_shared_types() { serde_json::to_value(legacy).unwrap(), serde_json::json!({"network_bytes": 480, "materialization_bytes": 40}) ); - let shared_kind = asap_types::resources::PhysicalHandoffKind::Materialization; + let shared_kind = asap_types::workload::resources::PhysicalHandoffKind::Materialization; let mut handoff = transfer("persist", None); handoff.kind = shared_kind; let json = serde_json::to_value(&handoff).unwrap(); @@ -387,7 +387,7 @@ fn missing_stale_and_non_finite_evidence_is_rejected() { // Individual actions may fit while accumulation across actions or nodes overflows. #[test] fn handoff_accumulation_and_calibrated_cost_overflow_are_rejected() { - use asap_aware_mapping::analytical_cost::AnalyticalCostError; + use asap_plan_selection::cost::analytical_cost::AnalyticalCostError; let (mut dag, mut scope) = fixture(); scope.recurrence = QueryRecurrence::OneTime { invocations: 1, diff --git a/crates/plan-selection/tests/stage3_dependencies.rs b/crates/plan-selection/tests/stage3_dependencies.rs new file mode 100644 index 000000000..0c40fd9f9 --- /dev/null +++ b/crates/plan-selection/tests/stage3_dependencies.rs @@ -0,0 +1,25 @@ +//! Stage 3 (plan selection) depends on neither the facade nor the executor: Cargo enforces +//! the one-way #509 stage flow (#572). + +use std::path::Path; + +/// Crates Stage 3 must not depend on: the facade and the executor. +const FORBIDDEN_CRATES: &[&str] = &["asap-planner", "asap-executor"]; + +/// The manifest names neither the facade nor the executor crate, so Cargo +/// rejects any import of them. +#[test] +fn stage3_manifest_has_no_path_to_the_facade_or_executor() { + let manifest = + std::fs::read_to_string(Path::new(env!("CARGO_MANIFEST_DIR")).join("Cargo.toml")).unwrap(); + let offenders: Vec<&str> = manifest + .lines() + .filter(|line| !line.trim_start().starts_with('#')) + .filter(|line| FORBIDDEN_CRATES.iter().any(|name| line.contains(name))) + .collect(); + assert!( + offenders.is_empty(), + "asap-plan-selection must not depend on the facade or the executor:\n{}", + offenders.join("\n") + ); +} diff --git a/crates/asap-aware-mapping/tests/storage_io.rs b/crates/plan-selection/tests/storage_io.rs similarity index 93% rename from crates/asap-aware-mapping/tests/storage_io.rs rename to crates/plan-selection/tests/storage_io.rs index a12118f09..e2bddff52 100644 --- a/crates/asap-aware-mapping/tests/storage_io.rs +++ b/crates/plan-selection/tests/storage_io.rs @@ -1,18 +1,18 @@ -use asap_aware_mapping::analytical_cost::{ +use asap_plan_selection::cost::analytical_cost::{ EvidenceBackedPhysicalDAG, ExecutionMultiplicity, PhysicalDAGNode, PhysicalNodeEvidence, PhysicalOperator, }; -use asap_aware_mapping::physical_operator_statistics::{ - ComparisonScope, EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, +use asap_plan_selection::cost::physical_operator_statistics::{ + ComparisonScope, EdgeStatistics, OperatorStatistics, ScanSelection, UnaryEdgeStatistics, }; -use asap_types::pre_asap::query_expr::Source; +use asap_types::ir::operator::operator_properties::Source; use asap_types::workload::{ DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; use std::collections::HashMap; fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { - let coverage = SourceCoverage { + let coverage = ScanSelection { source: Source::Table { table_ref: "events".into(), }, @@ -49,7 +49,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], - source_coverage: Some(coverage), + scan_selection: Some(coverage), output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -58,7 +58,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "left".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -67,7 +67,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "right".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -76,7 +76,7 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { id: "root".into(), operator: PhysicalOperator::Concat, children: vec!["left".into(), "right".into()], - source_coverage: None, + scan_selection: None, output_buffer_bytes: 0, retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, @@ -136,15 +136,15 @@ fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { ) } -use asap_aware_mapping::storage_io::*; +use asap_plan_selection::cost::storage_io::*; // The compatibility import and shared resource namespace expose one Rust type. #[test] fn storage_estimates_use_the_shared_resource_type_without_wire_changes() { let (dag, scope) = fixture(); let estimate = estimate_storage_io(&dag, &scope, &profile(&dag), "evidence-v1").unwrap(); - let shared: asap_types::resources::StorageResources = estimate.total; - let legacy: asap_aware_mapping::storage_io::StorageResources = shared; + let shared: asap_types::workload::resources::StorageResources = estimate.total; + let legacy: asap_plan_selection::cost::storage_io::StorageResources = shared; assert_eq!(shared, legacy); let wire = serde_json::to_value(&estimate).unwrap(); assert_eq!( @@ -319,7 +319,7 @@ fn multiplicity_and_cross_node_overflow_are_unavailable() { .push(large.clone()); assert_eq!( estimate_storage_io(&dag, &scope, &profile, "evidence-v1"), - Err(asap_aware_mapping::analytical_cost::AnalyticalCostError::Overflow) + Err(asap_plan_selection::cost::analytical_cost::AnalyticalCostError::Overflow) ); scope.recurrence = QueryRecurrence::OneTime { invocations: 1, @@ -336,7 +336,7 @@ fn multiplicity_and_cross_node_overflow_are_unavailable() { }); assert_eq!( estimate_storage_io(&dag, &scope, &profile, "evidence-v1"), - Err(asap_aware_mapping::analytical_cost::AnalyticalCostError::Overflow) + Err(asap_plan_selection::cost::analytical_cost::AnalyticalCostError::Overflow) ); } @@ -353,7 +353,7 @@ fn storage_node_identity_statistics_and_calibration_provenance_are_bound() { .get_mut("scan") .unwrap() .node - .source_coverage + .scan_selection .as_mut() .unwrap() .source_snapshot_id = "another-source".into(); diff --git a/crates/planner/Cargo.toml b/crates/planner/Cargo.toml index 59825a3fb..694b62c4c 100644 --- a/crates/planner/Cargo.toml +++ b/crates/planner/Cargo.toml @@ -6,12 +6,12 @@ edition = "2021" # The library facade (issue #429): one entry point that takes prepared input and # returns the selected DAG. It is the only crate that depends on every frontend # — before it, the sole facade re-exporting more than one was `asap-devtools`, -# a developer-tools crate. Everything below the frontend boundary lives in -# `asap-aware-mapping`, so a third party writing an optimization pass (issue -# #430) depends on that crate alone and never pulls in DataFusion. +# a developer-tools crate. It also owns the optimization pass (pass/, issue +# #430) that runs the #509 stage crates. [dependencies] asap-types = { path = "../types" } -asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-logical-optimizer = { path = "../logical-optimizer" } +asap-plan-selection = { path = "../plan-selection" } asap-frontend-sql = { path = "../frontend-sql" } asap-frontend-promql = { path = "../frontend-promql" } asap-frontend-metricsql = { path = "../frontend-metricsql" } diff --git a/crates/planner/src/lib.rs b/crates/planner/src/lib.rs index 88a085bf8..de71cbbf4 100644 --- a/crates/planner/src/lib.rs +++ b/crates/planner/src/lib.rs @@ -3,32 +3,32 @@ //! //! ```text //! PlanningWorkload ──lowering──▶ ParsedWorkload ──optimization pass──▶ PlanOutput -//! (this crate) (asap-aware-mapping) +//! (front ends) ([`pass`]: #509 Stages 1–3) //! ``` //! //! [`e2e_plan`] runs both stages. A caller who already holds pre-ASAP IR — a //! new frontend, a deserialized plan, a test that does not want to build SQL -//! and a catalog — skips this crate and calls -//! [`asap_aware_mapping::optimize`] directly. +//! and a catalog — calls [`optimize`] directly. -use std::rc::Rc; - -use asap_types::parsed_workload::{ParsedWorkload, ParsedWorkloadError}; -use asap_types::pre_asap::query_expr::QueryExpr; +use asap_types::workload::parsed_workload::{ParsedWorkload, ParsedWorkloadError}; use asap_types::workload::{PlanningWorkload, QueryLanguage, SqlDialect, WorkloadError}; -use asap_frontend_metricsql::{lower_metricsql, MetricsqlError}; +use asap_frontend_metricsql::{lower_metricsql_query, MetricsqlError}; use asap_frontend_promql::{ - lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, + lower_promql_query_workload, lower_promql_query_workload_with_histograms, HistogramCatalog, + PromqlError, }; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog, SqlError}; +pub mod pass; + // The optimization stage's vocabulary is this facade's vocabulary too: a caller // configures the same models and reads the same output whether it goes through // `e2e_plan` or straight to `optimize`. -pub use asap_aware_mapping::pass::{ - optimize, LifecycleInput, MajorPass, OptimizationInput, OptimizationPass, OptimizeError, - PassRegistry, PlanOutput, PlanningModels, QueryLifecyclePlan, +pub use asap_plan_selection::PlanningModels; +pub use pass::{ + optimize, OptimizationInput, OptimizationInputError, OptimizationPass, OptimizeError, + PassNameConflict, PassRegistry, PlanOutput, QueryPlan, StagePipeline, }; // ── Input ──────────────────────────────────────────────────────────────── @@ -55,10 +55,7 @@ pub struct UserInput<'a> { pub workload: &'a PlanningWorkload, pub frontend_specific: FrontendInput<'a>, pub models: PlanningModels<'a>, - /// Planning clock and runtime capabilities for the - /// maintenance-versus-recomputation decision every plan carries. - pub lifecycle: LifecycleInput, - /// `None` uses [`MajorPass`]. A black-box caller never sets this. + /// `None` uses [`StagePipeline`]. A black-box caller never sets this. pub pass: Option<&'a dyn OptimizationPass>, } @@ -67,13 +64,11 @@ impl<'a> UserInput<'a> { workload: &'a PlanningWorkload, frontend_specific: FrontendInput<'a>, models: PlanningModels<'a>, - lifecycle: LifecycleInput, ) -> Self { Self { workload, frontend_specific, models, - lifecycle, pass: None, } } @@ -103,21 +98,6 @@ impl<'a> UserInput<'a> { }); } - if let Some(horizon) = self.lifecycle.horizon { - if !horizon.0.is_finite() || horizon.0 <= 0.0 { - return Err(UserInputError::InvalidHorizon(horizon.0)); - } - } - // Two clocks would let the DAG be built for one instant and priced - // for another, with neither stage able to notice. - if let FrontendInput::Promql { now_ms, .. } = &self.frontend_specific { - if *now_ms != self.lifecycle.now_ms { - return Err(UserInputError::PlanningTimeMismatch { - frontend: *now_ms, - lifecycle: self.lifecycle.now_ms, - }); - } - } Ok(()) } } @@ -144,10 +124,6 @@ pub enum UserInputError { language: String, frontend: &'static str, }, - #[error("planning horizon must be finite and positive, got {0}")] - InvalidHorizon(f64), - #[error("frontend planning time {frontend} ms disagrees with lifecycle planning time {lifecycle} ms")] - PlanningTimeMismatch { frontend: u64, lifecycle: u64 }, } #[derive(Debug, thiserror::Error)] @@ -191,12 +167,12 @@ pub async fn e2e_plan(input: UserInput<'_>) -> Result { input.validate()?; let exprs = lower(&input).await?; - let parsed = ParsedWorkload::new(input.workload.clone(), exprs)?; + let parsed = ParsedWorkload::from_roots(input.workload.clone(), exprs)?; - let fallback = MajorPass; + let fallback = StagePipeline; let pass: &dyn OptimizationPass = input.pass.unwrap_or(&fallback); - let optimization = OptimizationInput::new(&parsed, input.models, input.lifecycle); + let optimization = OptimizationInput::new(&parsed, input.models); optimize(pass, optimization).map_err(PlanError::Optimize) } @@ -204,9 +180,8 @@ pub async fn e2e_plan(input: UserInput<'_>) -> Result { /// /// The SQL and MetricsQL frontends are driven one entry at a time rather than /// through `lower_sql_batch`, which walks `query_batch` alone and would drop -/// every repeating query — exactly the entries whose recurrence the lifecycle -/// stage needs. -async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { +/// every repeating query, and the output must cover every entry. +async fn lower(input: &UserInput<'_>) -> Result, PlanError> { let entries = || input.workload.query_workload.entries(); match &input.frontend_specific { @@ -229,7 +204,7 @@ async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { entry_index: Some(index), source: LoweringError::Sql(source), })?; - lowered.push(Rc::new(expr)); + lowered.push(expr.into()); } Ok(lowered) } @@ -237,28 +212,29 @@ async fn lower(input: &UserInput<'_>) -> Result>, PlanError> { now_ms, histograms, .. } => { let lowered = match histograms { - Some(histograms) => lower_promql_workload_with_histograms( + Some(histograms) => lower_promql_query_workload_with_histograms( input.workload, histograms.clone(), *now_ms, ), - None => lower_promql_workload(input.workload, *now_ms), + None => lower_promql_query_workload(input.workload, *now_ms), } .map_err(|source| PlanError::Lowering { entry_index: None, source: LoweringError::Promql(source), })?; - Ok(lowered.into_iter().map(Rc::new).collect()) + Ok(lowered) } FrontendInput::Metricsql => { let mut lowered = Vec::new(); for (index, entry) in entries().enumerate() { - let expr = lower_metricsql(&entry.query.0, entry.requirements.accuracy.target()) - .map_err(|source| PlanError::Lowering { - entry_index: Some(index), - source: LoweringError::Metricsql(source), - })?; - lowered.push(Rc::new(expr)); + let expr = + lower_metricsql_query(&entry.query.0, entry.requirements.accuracy.target()) + .map_err(|source| PlanError::Lowering { + entry_index: Some(index), + source: LoweringError::Metricsql(source), + })?; + lowered.push(expr); } Ok(lowered) } diff --git a/crates/asap-aware-mapping/src/pass/mod.rs b/crates/planner/src/pass/mod.rs similarity index 62% rename from crates/asap-aware-mapping/src/pass/mod.rs rename to crates/planner/src/pass/mod.rs index a0b993d97..7f090e646 100644 --- a/crates/asap-aware-mapping/src/pass/mod.rs +++ b/crates/planner/src/pass/mod.rs @@ -6,150 +6,50 @@ //! `TargetSubDAGCandidates`, no `ReplacementStrategy` — so an algorithm with no //! candidate-generation phase at all (a greedy MQO loop, say) can implement it //! without pretending to have phases it does not have. The shipped algorithm is -//! one implementation, [`MajorPass`]. +//! one implementation, [`StagePipeline`]. //! //! Call [`optimize`] rather than [`OptimizationPass::optimize`] directly: it //! validates the input once for every pass and checks the output contract that //! downstream consumers rely on. -mod major; +mod stage_pipeline; use std::collections::BTreeMap; use std::rc::Rc; -use asap_types::parsed_workload::ParsedWorkload; -use asap_types::post_asap::SummaryNode; -use asap_types::workload::WorkloadError; - -use crate::accuracy::{ - AccuracyEvidenceProvider, AccuracyModel, DefaultAccuracyModel, NoAccuracyEvidence, -}; -use crate::cost_model::{CostModel, DefaultCostModel}; -use crate::recurrence::Horizon; -use crate::replacement::RealizationError; -use crate::summary_maintenance_lifecycle::{ - SummaryMaintenanceLifecycleAssemblyError, SummaryMaintenanceLifecycleCapabilities, - SummaryMaintenanceLifecyclePlan, SummaryMaintenanceLifecycleSelectionError, +use asap_types::ir::export::compile_physical_asap_workload; +use asap_types::ir::properties::timing::{ + apply_materialization_timings, MaterializationAssignment, TimingMemo, }; +use asap_types::ir::properties::ExecutionDataStateError; +use asap_types::ir::OperatorNode; +use asap_types::workload::parsed_workload::ParsedWorkload; +use asap_types::workload::WorkloadError; -pub use major::MajorPass; +use asap_logical_optimizer::pass1::logical_candidates::LogicalCandidateError; +use asap_plan_selection::{Selection, SelectionError}; -static DEFAULT_COST_MODEL: DefaultCostModel = DefaultCostModel; -static DEFAULT_ACCURACY_MODEL: DefaultAccuracyModel = DefaultAccuracyModel; -static NO_ACCURACY_EVIDENCE: NoAccuracyEvidence = NoAccuracyEvidence; +use asap_plan_selection::PlanningModels; +pub use stage_pipeline::StagePipeline; // ── Input ──────────────────────────────────────────────────────────────── -/// Planning logic, as opposed to the scoped facts it consumes: a model can have -/// a built-in default, evidence about a particular deployment cannot. -#[derive(Clone, Copy)] -#[non_exhaustive] -pub struct PlanningModels<'a> { - pub cost: &'a dyn CostModel, - pub accuracy: &'a dyn AccuracyModel, - pub evidence: &'a dyn AccuracyEvidenceProvider, -} - -impl<'a> PlanningModels<'a> { - pub fn new( - cost: &'a dyn CostModel, - accuracy: &'a dyn AccuracyModel, - evidence: &'a dyn AccuracyEvidenceProvider, - ) -> Self { - Self { - cost, - accuracy, - evidence, - } - } - - /// The built-in models. `DefaultCostModel` does not override - /// `estimate_cost`, so this configuration ranks structurally and is not a - /// measured deployment cost. - pub fn builtin() -> PlanningModels<'static> { - PlanningModels { - cost: &DEFAULT_COST_MODEL, - accuracy: &DEFAULT_ACCURACY_MODEL, - evidence: &NO_ACCURACY_EVIDENCE, - } - } - - pub fn with_cost(mut self, cost: &'a dyn CostModel) -> Self { - self.cost = cost; - self - } - - pub fn with_accuracy(mut self, accuracy: &'a dyn AccuracyModel) -> Self { - self.accuracy = accuracy; - self - } - - pub fn with_evidence(mut self, evidence: &'a dyn AccuracyEvidenceProvider) -> Self { - self.evidence = evidence; - self - } -} - -/// Supplying this asks the pass to also decide summary maintenance versus raw -/// recomputation; leaving it out asks only for the logical DAG. -#[derive(Clone, Copy)] -#[non_exhaustive] -pub struct LifecycleInput { - /// Planning clock, Unix milliseconds. - pub now_ms: u64, - /// Seconds. Required to turn recurring demand into a finite total. - pub horizon: Option, - pub capabilities: SummaryMaintenanceLifecycleCapabilities, -} - -impl LifecycleInput { - pub fn new(now_ms: u64, capabilities: SummaryMaintenanceLifecycleCapabilities) -> Self { - Self { - now_ms, - horizon: None, - capabilities, - } - } - - pub fn with_horizon(mut self, horizon: Horizon) -> Self { - self.horizon = Some(horizon); - self - } -} - #[derive(Clone, Copy)] #[non_exhaustive] pub struct OptimizationInput<'a> { pub workload: &'a ParsedWorkload, pub models: PlanningModels<'a>, - /// Every plan carries the maintenance-versus-recomputation decision, so - /// the planning clock and runtime capabilities are always required. - pub lifecycle: LifecycleInput, } impl<'a> OptimizationInput<'a> { - pub fn new( - workload: &'a ParsedWorkload, - models: PlanningModels<'a>, - lifecycle: LifecycleInput, - ) -> Self { - Self { - workload, - models, - lifecycle, - } + pub fn new(workload: &'a ParsedWorkload, models: PlanningModels<'a>) -> Self { + Self { workload, models } } pub fn validate(&self) -> Result<(), OptimizationInputError> { self.workload .validate() - .map_err(OptimizationInputError::Workload)?; - if let Some(horizon) = self.lifecycle.horizon { - if !horizon.0.is_finite() || horizon.0 <= 0.0 { - return Err(OptimizationInputError::InvalidHorizon(horizon.0)); - } - } - Ok(()) + .map_err(OptimizationInputError::Workload) } } @@ -158,51 +58,125 @@ impl<'a> OptimizationInput<'a> { pub enum OptimizationInputError { #[error("workload: {0}")] Workload(WorkloadError), - #[error("planning horizon must be finite and positive, got {0}")] - InvalidHorizon(f64), } // ── Output ─────────────────────────────────────────────────────────────── -/// One query's selected post-ASAP DAG plus the maintenance decisions taken -/// for it. The DAG is `plan.root`. +/// One query's selected post-ASAP DAG. #[derive(Debug, Clone)] -pub struct QueryLifecyclePlan { +pub struct QueryPlan { /// Index into `QueryWorkload::entries()`. pub entry_index: usize, - pub plan: SummaryMaintenanceLifecyclePlan, + pub root: Rc, } -/// One plan per workload entry, in `QueryWorkload::entries()` order; +/// One multi-root workload DAG with query bindings in entry order; /// [`check_contract`] enforces that. /// /// Plans are not deduplicated across entries: a summary state that several -/// queries share appears in each of their plans as the same `Rc` (with the -/// same lifecycle), so a consumer that deploys or costs the workload must -/// dedupe deployments by `Rc::ptr_eq` on the summary node. +/// queries share appears in each of their plans as the same `Rc`, so a +/// consumer that deploys or costs the workload must dedupe by `Rc::ptr_eq`. #[derive(Debug, Clone)] #[non_exhaustive] pub struct PlanOutput { - pub plans: Vec, + pub plans: Vec, + /// Exact scalar expressions, keyed by workload entry; embedded plan reads remain visible. + pub scalar_roots: Vec<(usize, asap_types::ir::ScalarExpr)>, + /// How the plan was chosen, when the pass selects among priced candidates. + pub selection: Option, } impl PlanOutput { - pub fn new(plans: Vec) -> Self { - Self { plans } + pub fn new(plans: Vec) -> Self { + Self { + plans, + scalar_roots: Vec::new(), + selection: None, + } } /// Entry indices in output order. pub fn entry_indices(&self) -> Vec { - self.plans.iter().map(|p| p.entry_index).collect() + let mut indices: Vec<_> = self + .plans + .iter() + .map(|p| p.entry_index) + .chain(self.scalar_roots.iter().map(|(i, _)| *i)) + .collect(); + if !self.scalar_roots.is_empty() { + indices.sort_unstable(); + } + indices + } + + /// All query roots in workload order, including standalone scalars. + pub fn roots(&self) -> Vec { + let mut roots: Vec<_> = self + .plans + .iter() + .map(|p| { + ( + p.entry_index, + asap_types::ir::QueryRoot::Operator(Rc::clone(&p.root)), + ) + }) + .chain( + self.scalar_roots + .iter() + .map(|(i, expr)| (*i, asap_types::ir::QueryRoot::Scalar(expr.clone()))), + ) + .collect(); + roots.sort_by_key(|(i, _)| *i); + roots.into_iter().map(|(_, root)| root).collect() + } + + /// The selected operator roots. Use `roots()` to include scalar queries. + pub fn operator_roots(&self) -> Vec> { + self.plans.iter().map(|p| Rc::clone(&p.root)).collect() + } + + /// Unique operators in the entire workload DAG, including scalar-plan dependencies. + /// Several query roots can reach the same operator; it is returned once. + pub fn operators(&self) -> Vec> { + let mut seen = std::collections::HashSet::new(); + let mut nodes = Vec::new(); + for root in self.roots() { + let inputs = match root { + asap_types::ir::QueryRoot::Operator(node) => vec![node], + asap_types::ir::QueryRoot::Scalar(expr) => { + expr.operator_refs().into_iter().cloned().collect() + } + }; + for input in inputs { + for node in OperatorNode::reachable(&input) { + if seen.insert(Rc::as_ptr(&node)) { + nodes.push(node); + } + } + } + } + nodes } - /// The selected DAG root per query. - pub fn dags(&self) -> Vec> { - self.plans.iter().map(|p| Rc::clone(&p.plan.root)).collect() + /// The workload as one physical ASAP DAG: a root per operator query, in + /// plan order, with shared sub-DAGs exported once. Standalone scalar + /// roots have no physical form yet and are left out. + pub fn execution_timed_dag( + &self, + ) -> Result { + // One memo, so a node shared by several roots is timed and exported once. + let mut memo = TimingMemo::new(); + let assignment = MaterializationAssignment::all_query_time(); + let timed = self + .plans + .iter() + .map(|p| apply_materialization_timings(&p.root, &assignment, &mut memo)) + .collect::, _>>()?; + compile_physical_asap_workload(&timed) } pub fn len(&self) -> usize { - self.plans.len() + self.plans.len() + self.scalar_roots.len() } pub fn is_empty(&self) -> bool { @@ -215,18 +189,10 @@ impl PlanOutput { pub enum OptimizeError { #[error("optimization input: {0}")] Input(#[from] OptimizationInputError), - #[error("entry {entry_index}: {source}")] - Realization { - entry_index: usize, - source: RealizationError, - }, - #[error("summary-maintenance-lifecycle selection: {0}")] - LifecycleSelection(SummaryMaintenanceLifecycleSelectionError), - #[error("entry {entry_index}: {source}")] - LifecycleAssembly { - entry_index: usize, - source: SummaryMaintenanceLifecycleAssemblyError, - }, + #[error("Stage 1: {0}")] + LogicalCandidates(#[from] LogicalCandidateError), + #[error("plan selection: {0}")] + Selection(#[from] SelectionError), /// The pass returned something the downstream contract forbids. This is a /// defect in the pass, not in its input. #[error("pass `{pass}` violated the output contract: {detail}")] @@ -310,11 +276,11 @@ impl PassRegistry { Self::default() } - /// Only [`MajorPass`], under the name `major`. + /// Only [`StagePipeline`], under the name `stage-pipeline`. pub fn with_builtin() -> Self { let mut registry = Self::new(); registry - .register(Box::new(MajorPass)) + .register(Box::new(StagePipeline)) .expect("empty registry cannot conflict"); registry } @@ -393,14 +359,17 @@ mod tests { assert_eq!(err.0, "greedy"); } - /// The builtin registry resolves `major`, and names come back sorted so a + /// The builtin registry resolves `stage-pipeline`, and names come back sorted so a /// sweep over every registered pass is reproducible. #[test] fn registry_resolves_builtin_and_lists_names_in_order() { let mut registry = PassRegistry::with_builtin(); registry.register(Box::new(Stub("alpha"))).unwrap(); - assert!(registry.get("major").is_some()); + assert!(registry.get("stage-pipeline").is_some()); assert!(registry.get("absent").is_none()); - assert_eq!(registry.names().collect::>(), vec!["alpha", "major"]); + assert_eq!( + registry.names().collect::>(), + vec!["alpha", "stage-pipeline"] + ); } } diff --git a/crates/planner/src/pass/stage_pipeline.rs b/crates/planner/src/pass/stage_pipeline.rs new file mode 100644 index 000000000..f0f6d2b43 --- /dev/null +++ b/crates/planner/src/pass/stage_pipeline.rs @@ -0,0 +1,68 @@ +//! [`StagePipeline`] — the #509 planner stages behind the +//! [`OptimizationPass`](super::OptimizationPass) trait. +//! +//! [`plan_stages`] runs them: Stage 1 lists each target's local alternatives +//! (Pass 1) with and without identical sub-DAGs shared across queries (Pass +//! 2's identical-expression rule), Stage 2 implements a candidate physically +//! (everything at query time), and Stage 3 checks accuracy, prices it and +//! chooses. Pass 2's other rules are not planned yet. + +use asap_types::ir::schema_support::with_promql_series_identity; +use asap_types::ir::QueryRoot; +use asap_types::workload::QueryLanguage; + +use super::{OptimizationInput, OptimizationPass, OptimizeError, PlanOutput, QueryPlan}; +use asap_plan_selection::plan_stages; + +#[derive(Debug, Default, Clone, Copy)] +pub struct StagePipeline; + +impl OptimizationPass for StagePipeline { + fn name(&self) -> &'static str { + "stage-pipeline" + } + + fn optimize(&self, input: OptimizationInput<'_>) -> Result { + let workload = input.workload; + let mut output = PlanOutput::new(Vec::new()); + output.scalar_roots = workload.scalar_roots().to_vec(); + if workload.exprs().is_empty() { + return Ok(output); + } + let promql = workload.query_workload().language == QueryLanguage::PromQL; + // PromQL rows carry each series' full identity as a column: the row + // representation per-series state needs at runtime. A query with no + // such representation is planned over its labels alone. + let roots: Vec<(usize, QueryRoot)> = workload + .operator_indices() + .iter() + .copied() + .zip(workload.exprs()) + .map(|(index, root)| { + let root = match promql { + true => with_promql_series_identity(root).unwrap_or_else(|_| root.clone()), + false => root.clone(), + }; + (index, QueryRoot::Operator(root)) + }) + .collect(); + let targets: Vec<_> = workload + .entries() + .map(|(entry, _)| Some(entry.requirements.accuracy.target())) + .collect(); + let data = workload.data_workload().cloned().unwrap_or_default(); + let plan = plan_stages(roots, &targets, &data, input.models, 0)?.plan; + + output.plans = plan + .logical + .iter() + .zip(plan.physical.roots) + .map(|((entry_index, _), root)| QueryPlan { + entry_index: *entry_index, + root, + }) + .collect(); + output.selection = Some(plan.selection); + Ok(output) + } +} diff --git a/crates/planner/tests/e2e_plan.rs b/crates/planner/tests/e2e_plan.rs index ff8a8dde0..ea2eb615c 100644 --- a/crates/planner/tests/e2e_plan.rs +++ b/crates/planner/tests/e2e_plan.rs @@ -3,17 +3,13 @@ use std::rc::Rc; -use asap_aware_mapping::pass::{ - OptimizationInput, OptimizationPass, OptimizeError, PlanOutput, PlanningModels, -}; -use asap_aware_mapping::replacement::default_strategies_with_evidence; -use asap_aware_mapping::{ - search_workload_with_targets, Horizon, LifecycleInput, SummaryMaintenanceLifecycleCapabilities, -}; use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; +use asap_logical_optimizer::pass2::identical_expressions::stage1_logical_candidates; +use asap_plan_selection::{select_exhaustive, PlanningModels, MAX_ENUMERATED_CANDIDATES}; +use asap_planner::pass::{OptimizationInput, OptimizationPass, OptimizeError, PlanOutput}; use asap_planner::{e2e_plan, FrontendInput, PlanError, UserInput, UserInputError}; -use asap_types::post_asap::SummaryExpr; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::QueryRoot; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataArrival, DataWorkload, DurationMs, Evidence, @@ -41,12 +37,6 @@ fn batch(sql: &str) -> BatchEntry { } } -/// The planning clock and default capabilities, no horizon: the least a -/// caller can supply. -fn lifecycle() -> LifecycleInput { - LifecycleInput::new(NOW_MS, SummaryMaintenanceLifecycleCapabilities::default()) -} - fn lineitem_catalog() -> SqlCatalog { SqlCatalog::new().with_table( "lineitem", @@ -90,7 +80,6 @@ async fn plans_every_query_in_entry_order() { &workload, FrontendInput::Sql { catalog: &catalog }, PlanningModels::builtin(), - lifecycle(), ); let output = e2e_plan(input).await.expect("workload plans"); @@ -98,13 +87,10 @@ async fn plans_every_query_in_entry_order() { assert_eq!(output.entry_indices(), vec![0, 1]); } -/// With the built-in cost model no lifecycle cost is ever known, and -/// lifecycle-aware selection then finalizes every summary target as raw -/// recompute: the cost-only selection picks a sketch for the same workload. -/// This pins that behavior so the facade's output is not mistaken for a -/// decision; it is a defect of `DefaultCostModel`, not addressed here. +/// The facade selects what exhaustive Stage 1 → 3 selection selects over the +/// same inventory. #[tokio::test] -async fn builtin_cost_model_cannot_price_lifecycles_and_falls_back_to_raw_recompute() { +async fn facade_plans_match_exhaustive_stage_pipeline_selection() { let workload = sql_workload( vec![ batch("SELECT COUNT(DISTINCT l_orderkey) FROM lineitem"), @@ -119,14 +105,14 @@ async fn builtin_cost_model_cannot_price_lifecycles_and_falls_back_to_raw_recomp &workload, FrontendInput::Sql { catalog: &catalog }, models, - lifecycle(), )) .await .expect("workload plans"); - // The cost-only selection over the same search space, the way a caller - // reaches it without the facade. + // Every combination built and priced, the way a caller reaches it + // without the facade. let mut roots = Vec::new(); + let mut targets = Vec::new(); for (index, entry) in workload.query_workload.entries().enumerate() { let accuracy = entry.requirements.accuracy.target(); let expr = lower_sql_dialect( @@ -137,38 +123,43 @@ async fn builtin_cost_model_cannot_price_lifecycles_and_falls_back_to_raw_recomp ) .await .expect("lowers"); - roots.push((index, Rc::new(expr), Some(accuracy))); + roots.push((index, QueryRoot::Operator(expr))); + targets.push(Some(accuracy)); } - let strategies = default_strategies_with_evidence(models.cost, models.evidence); - let space = search_workload_with_targets(roots, &strategies, models.accuracy); - let selection = space.global_selection(models.cost); - - assert_eq!(output.plans.len(), space.roots.len()); - for (plan, (_, root)) in output.plans.iter().zip(&space.roots) { - let cost_only = selection - .assemble_selected_dag(root) - .expect("assembles") - .expect("root has a group"); - assert!( - !matches!(cost_only.expr, SummaryExpr::KeepPreAsap(_)), - "entry {}: cost-only selection was expected to pick a summary", - plan.entry_index - ); - assert!( - matches!(plan.plan.root.expr, SummaryExpr::KeepPreAsap(_)) - && plan.plan.selected_raw_recompute - && plan.plan.deployments.is_empty() - && plan.plan.summary_total_cost.is_none() - && plan.plan.raw_recompute_total_cost.is_none(), - "entry {}: the built-in model priced a lifecycle", + let inventory = stage1_logical_candidates(roots).expect("Stage 1"); + let data = workload.data_workload.clone().unwrap_or_default(); + let enumeration = select_exhaustive( + &inventory, + &targets, + &data, + models, + MAX_ENUMERATED_CANDIDATES, + ) + .expect("selects"); + assert!(enumeration.combinations <= MAX_ENUMERATED_CANDIDATES); + let exhaustive = enumeration + .candidates + .iter() + .filter_map(|c| c.physical.as_ref()) + .find(|p| p.id == enumeration.selection.selected) + .expect("selected candidate"); + + let selection = output.selection.as_ref().expect("stage pipeline selection"); + assert_eq!(selection.selected, enumeration.selection.selected); + assert!(selection.guaranteed_optimal()); + assert_eq!(output.plans.len(), exhaustive.roots.len()); + for (plan, root) in output.plans.iter().zip(&exhaustive.roots) { + assert_eq!( + &plan.root, root, + "entry {}: the facade selected a different DAG", plan.entry_index ); } } /// A repeating SQL query reaches the optimizer. `lower_sql_batch` walks -/// `query_batch` alone, so driving the frontend through it would drop exactly -/// the entries whose recurrence the lifecycle stage reads. +/// `query_batch` alone, so driving the frontend through it would drop the +/// repeating entries. #[tokio::test] async fn lowers_repeating_sql_entries_too() { let workload = sql_workload( @@ -188,7 +179,6 @@ async fn lowers_repeating_sql_entries_too() { &workload, FrontendInput::Sql { catalog: &catalog }, PlanningModels::builtin(), - lifecycle(), ); let output = e2e_plan(input).await.expect("workload plans"); @@ -230,7 +220,6 @@ async fn runs_a_caller_supplied_pass_instead_of_the_shipped_one() { &workload, FrontendInput::Sql { catalog: &catalog }, PlanningModels::builtin(), - lifecycle(), ) .with_pass(&pass); @@ -253,7 +242,7 @@ async fn harness_rejects_a_pass_that_mislabels_entry_indices() { "mangling" } fn optimize(&self, input: OptimizationInput<'_>) -> Result { - let mut output = asap_aware_mapping::MajorPass.optimize(input)?; + let mut output = asap_planner::StagePipeline.optimize(input)?; for plan in output.plans.iter_mut() { plan.entry_index += 1; } @@ -271,7 +260,6 @@ async fn harness_rejects_a_pass_that_mislabels_entry_indices() { &workload, FrontendInput::Sql { catalog: &catalog }, PlanningModels::builtin(), - lifecycle(), ) .with_pass(&pass); @@ -303,7 +291,6 @@ async fn rejects_a_frontend_that_does_not_match_the_workload_language() { histograms: None, }, PlanningModels::builtin(), - lifecycle(), ); let err = e2e_plan(input).await.unwrap_err(); @@ -313,48 +300,9 @@ async fn rejects_a_frontend_that_does_not_match_the_workload_language() { )); } -/// Two planning clocks would let the DAG be built for one instant and priced -/// for another; the input check refuses that before lowering. -#[test] -fn rejects_disagreeing_planning_clocks() { - let workload = PlanningWorkload { - query_workload: QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: Some(vec![batch("up")]), - repeating_queries: None, - }, - data_workload: Some(DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - data_ingestion_interval: Evidence { - value: Some(DurationMs(15_000)), - ..Default::default() - }, - ..Default::default() - }), - }; - let input = UserInput::new( - &workload, - FrontendInput::Promql { - now_ms: NOW_MS, - histograms: None, - }, - PlanningModels::builtin(), - LifecycleInput::new( - NOW_MS + 1, - SummaryMaintenanceLifecycleCapabilities::default(), - ), - ); - - assert!(matches!( - input.validate(), - Err(UserInputError::PlanningTimeMismatch { .. }) - )); -} - -/// The maintenance decisions ride inside each plan, and the DAG is still -/// there — inside the plan's `root`, not alongside it. +/// A repeating PromQL query yields one plan carrying its selected DAG root. #[tokio::test] -async fn lifecycle_decisions_ride_inside_each_plan() { +async fn each_plan_carries_its_selected_root() { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -383,57 +331,61 @@ async fn lifecycle_decisions_ride_inside_each_plan() { histograms: None, }, PlanningModels::builtin(), - lifecycle().with_horizon(Horizon(3_600.0)), ); let output = e2e_plan(input).await.expect("workload plans"); assert_eq!(output.plans.len(), 1); assert_eq!(output.plans[0].entry_index, 0); - let _: &Rc<_> = &output.plans[0].plan.root; - assert_eq!(output.dags().len(), 1); + let _: &Rc<_> = &output.plans[0].root; + assert_eq!(output.operator_roots().len(), 1); } -/// Each root's lifecycle is planned against the entries that read it: a -/// query polled every minute and an unrelated one polled every ten minutes -/// each see only their own reads over the hour, not the workload's 66. +/// Scalar-only and mixed workloads preserve entry bindings without wrapper nodes. #[tokio::test] -async fn each_plan_counts_only_its_own_entries_reads() { - let repeating = |query: &str, interval_ms: u32| RepeatingEntry { - query: Query(query.into()), - demand: RepeatedDemand::FixedInterval(RepetitionInterval(interval_ms)), - requirements: approximate(), - predictability: Predictability::Unknown, - time_selection: TimeSelection::default(), - }; - let workload = PlanningWorkload { - query_workload: QueryWorkload { - language: QueryLanguage::PromQL, - query_batch: None, - repeating_queries: Some(vec![ - repeating("count_over_time(up[5m])", 60_000), - repeating("sum_over_time(latency[5m])", 600_000), - ]), - }, - data_workload: Some(DataWorkload { - arrival: DataArrival::ContinuouslyIngesting, - data_ingestion_interval: Evidence { - value: Some(DurationMs(15_000)), +async fn scalar_roots_survive_planning_in_workload_order() { + for queries in [ + vec!["2", "time()"], + vec!["2", "up * 2", "scalar(sum(up)) + 1"], + ] { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(queries.iter().map(|q| batch(q)).collect()), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, ..Default::default() + }), + }; + let output = e2e_plan(UserInput::new( + &workload, + FrontendInput::Promql { + now_ms: NOW_MS, + histograms: None, }, - ..Default::default() - }), - }; - let input = UserInput::new( - &workload, - FrontendInput::Promql { - now_ms: NOW_MS, - histograms: None, - }, - PlanningModels::builtin(), - lifecycle().with_horizon(Horizon(3_600.0)), - ); - - let output = e2e_plan(input).await.expect("workload plans"); - let reads: Vec<_> = output.plans.iter().map(|p| p.plan.expected_reads).collect(); - assert_eq!(reads, vec![Some(60.0), Some(6.0)]); + PlanningModels::builtin(), + )) + .await + .unwrap(); + assert_eq!( + output.entry_indices(), + (0..queries.len()).collect::>() + ); + assert!(matches!( + output.roots()[0], + asap_types::ir::QueryRoot::Scalar(_) + )); + assert_eq!(output.roots().len(), queries.len()); + if queries.len() == 3 { + assert_eq!(output.plans[0].entry_index, 1); + let asap_types::ir::QueryRoot::Scalar(expr) = &output.roots()[2] else { + panic!() + }; + assert_eq!(expr.operator_refs().len(), 1); + } + } } diff --git a/crates/planner/tests/stage_pipeline_selection.rs b/crates/planner/tests/stage_pipeline_selection.rs new file mode 100644 index 000000000..8be40c301 --- /dev/null +++ b/crates/planner/tests/stage_pipeline_selection.rs @@ -0,0 +1,310 @@ +//! The stage pipeline's dynamic program selects the exhaustive minimum +//! (#572): on #509 Example 1 and on small nested, top-k, shared-input and SQL +//! workloads, the sharing variant and combination it picks are the ones that +//! building and pricing every combination of every variant picks. + +use asap_frontend_sql::{lower_sql_dialect, SqlCatalog}; +use asap_logical_optimizer::pass2::identical_expressions::{ + stage1_logical_candidates, SharingVariant, +}; +use asap_plan_selection::PlanningModels; +use asap_plan_selection::{ + select_exhaustive, select_plan, SelectionMethod, MAX_ENUMERATED_CANDIDATES, +}; +use asap_planner::{e2e_plan, FrontendInput, UserInput}; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::schema_support::with_promql_series_identity; +use asap_types::ir::QueryRoot; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, BatchEntry, DataArrival, DataDistribution, DataWorkload, DurationMs, + Evidence, EvidenceSource, LatencyRequirement, PlanningWorkload, Predictability, Query, + QueryLanguage, QueryRequirements, QueryTimeScope, QueryWorkload, Rate, RepeatedDemand, + RepeatingEntry, RepetitionInterval, SqlDialect, TimeSelection, +}; + +type Inventory = Vec>; + +fn declared(value: T) -> Evidence { + Evidence { + value: Some(value), + source: EvidenceSource::Declared, + ..Default::default() + } +} + +/// #509 Example 1 over its shared data workload, as `stage_pipeline` builds it. +fn example1() -> PlanningWorkload { + let panel = |query: &str, accuracy, response_latency| RepeatingEntry { + query: Query(query.into()), + demand: RepeatedDemand::FixedInterval(RepetitionInterval(10_000)), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + response_latency, + }, + predictability: Predictability::Predictable { known_at: None }, + time_selection: TimeSelection { + scope: QueryTimeScope::RealTime, + lookback: Some(DurationMs(60_000)), + as_of: None, + }, + }; + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: None, + repeating_queries: Some(vec![ + panel( + "sum by (job) (rate(http_requests_total[1m]))", + AccuracyTarget::Exact, + LatencyRequirement::Unspecified, + ), + panel( + "topk by (job) (10, sum_over_time(http_requests_total[1m]))", + AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }, + LatencyRequirement::ExplicitMaxMs(100.0), + ), + ]), + }, + data_workload: Some(DataWorkload { + arrival: DataArrival::ContinuouslyIngesting, + data_ingestion_interval: declared(DurationMs(15_000)), + ingestion_volume: Evidence::default(), + ingestion_rate: declared(Rate(1_000_000.0 / 15.0)), + input_cardinality: declared(1_000_000), + distribution: declared(DataDistribution::Zipf), + }), + } +} + +fn batch(query: &str, accuracy: AccuracyTarget) -> BatchEntry { + BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + } +} + +fn promql(queries: &[&str], series: u64) -> PlanningWorkload { + let accuracy = AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.001, + }; + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(queries.iter().map(|q| batch(q, accuracy.clone())).collect()), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: declared(DurationMs(15_000)), + ingestion_rate: declared(Rate(10.0)), + input_cardinality: declared(series), + ..Default::default() + }), + } +} + +/// Stage 1 as the stage pipeline builds it: series identity, then Pass 1 +/// with and without identical sub-DAGs merged. +fn promql_inventory(workload: &PlanningWorkload) -> Inventory { + let roots = asap_frontend_promql::lower_promql_query_workload(workload, 0) + .expect("lowers") + .into_iter() + .enumerate() + .map(|(index, root)| match root { + QueryRoot::Operator(node) => ( + index, + QueryRoot::Operator(with_promql_series_identity(&node).unwrap()), + ), + QueryRoot::Scalar(_) => panic!("operator roots"), + }) + .collect(); + stage1_logical_candidates(roots).expect("Stage 1") +} + +fn targets(workload: &PlanningWorkload) -> Vec> { + workload + .query_workload + .entries() + .map(|entry| Some(entry.requirements.accuracy.target())) + .collect() +} + +/// The dynamic program's choice is the exhaustive winner's; returns that +/// winner's id. +fn assert_dp_matches_exhaustive( + inventory: &Inventory, + workload: &PlanningWorkload, + combinations: usize, +) -> String { + let targets = targets(workload); + let data = workload.data_workload.clone().unwrap_or_default(); + let models = PlanningModels::builtin(); + let exhaustive = select_exhaustive( + inventory, + &targets, + &data, + models, + MAX_ENUMERATED_CANDIDATES, + ) + .expect("exhaustive selection"); + assert_eq!(exhaustive.combinations, combinations); + assert_eq!(exhaustive.candidates.len(), combinations); + let winner = exhaustive + .candidates + .iter() + .find(|c| { + c.physical + .as_ref() + .is_some_and(|p| p.id == exhaustive.selection.selected) + }) + .expect("winner was built"); + + let plan = select_plan(inventory, &targets, &data, models).expect("selects"); + assert_eq!(plan.selection.method, SelectionMethod::TreeDp); + assert_eq!((plan.shared, &plan.choice), (winner.shared, &winner.choice)); + assert_eq!(plan.selection.selected, exhaustive.selection.selected); + exhaustive.selection.selected +} + +/// #509 Example 1: the dynamic program picks the cheapest of its 64 +/// combinations (32 per sharing variant, including Q2's whole-expression +/// top-k sketches, which absorb its `sum_over_time`): P58, all exact with +/// the range selector shared. +#[test] +fn example1_dp_equals_exhaustive() { + let workload = example1(); + let selected = assert_dp_matches_exhaustive(&promql_inventory(&workload), &workload, 64); + assert_eq!(selected, "P58"); +} + +/// When sharing merges a whole target (`rate(x[1m])` read by both queries), +/// the shared variant has one target for it, and the dynamic program still +/// selects the exhaustive minimum over both variants (16 + 8 combinations). +#[test] +fn shared_target_dp_equals_exhaustive() { + let workload = promql( + &["sum by (job) (rate(x[1m]))", "max by (job) (rate(x[1m]))"], + 1_000, + ); + let stage1 = promql_inventory(&workload); + let targets: Vec<_> = stage1.iter().map(|v| v.inventory.targets.len()).collect(); + assert_eq!(targets, [4, 3]); + let selected = assert_dp_matches_exhaustive(&stage1, &workload, 24); + assert!( + selected.trim_start_matches('P').parse::().unwrap() > 16, + "a shared candidate wins: {selected}" + ); +} + +/// Nested targets (`sum` over `rate`) select the exhaustive minimum. +#[test] +fn nested_sum_over_rate_dp_equals_exhaustive() { + let workload = promql(&["sum by (job) (rate(x[1m]))"], 1_000); + assert_dp_matches_exhaustive(&promql_inventory(&workload), &workload, 4); +} + +/// An aggregate over a top-k, with inputs both below and above k × groups +/// rows, selects the exhaustive minimum over all 40 combinations. +#[test] +fn count_over_topk_dp_equals_exhaustive() { + for series in [3, 1_000_000] { + let workload = promql(&["count(topk by (job) (10, sum_over_time(m[1m])))"], series); + assert_dp_matches_exhaustive(&promql_inventory(&workload), &workload, 40); + } +} + +/// Two PromQL queries sharing a source select the exhaustive minimum. +#[test] +fn two_query_promql_dp_equals_exhaustive() { + let workload = promql( + &[ + "sum by (job) (rate(x[1m]))", + "topk by (job) (10, sum_over_time(x[1m]))", + ], + 1_000, + ); + assert_dp_matches_exhaustive(&promql_inventory(&workload), &workload, 64); +} + +/// A SQL workload (distinct count and percentile) selects the exhaustive minimum. +#[tokio::test] +async fn sql_dp_equals_exhaustive() { + let accuracy = AccuracyTarget::Epsilon(0.01); + let queries = [ + "SELECT COUNT(DISTINCT l_orderkey) FROM lineitem", + "SELECT approx_percentile_cont(l_extendedprice, 0.99) FROM lineitem", + ]; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::SQL(SqlDialect::DataFusionSQL), + query_batch: Some(queries.iter().map(|q| batch(q, accuracy.clone())).collect()), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + arrival: DataArrival::AtRest, + ..Default::default() + }), + }; + let catalog = SqlCatalog::new().with_table( + "lineitem", + Schema::new(vec![ + Field::plain("l_orderkey", DataType::Int64, false), + Field::plain("l_extendedprice", DataType::Float64, false), + ]), + ); + let mut roots = Vec::new(); + for (index, query) in queries.iter().enumerate() { + let root = lower_sql_dialect(query, &catalog, SqlDialect::DataFusionSQL, accuracy.clone()) + .await + .expect("lowers"); + roots.push((index, QueryRoot::Operator(root))); + } + let inventory = stage1_logical_candidates(roots).expect("Stage 1"); + assert_dp_matches_exhaustive(&inventory, &workload, 15); +} + +/// Through the facade, Example 1 selects the exhaustive winner, P58: both +/// queries exact, Q1's rate and sum and Q2's sum as exact accumulators, over +/// one shared range selector. +#[tokio::test] +async fn facade_selects_the_example1_exhaustive_winner() { + let workload = example1(); + let output = e2e_plan(UserInput::new( + &workload, + FrontendInput::Promql { + now_ms: 0, + histograms: None, + }, + PlanningModels::builtin(), + )) + .await + .expect("plans"); + let selection = output.selection.as_ref().expect("selection"); + assert_eq!(selection.selected, "P58"); + assert_eq!(output.plans.len(), 2); + let scans = |root: &std::rc::Rc| { + asap_types::ir::OperatorNode::reachable(root) + .into_iter() + .filter(|n| matches!(n.non_asap(), Some(asap_types::ir::NonASAPOp::Scan { .. }))) + .map(|n| std::rc::Rc::as_ptr(&n)) + .collect::>() + }; + assert_eq!(scans(&output.plans[0].root), scans(&output.plans[1].root)); + assert_eq!(selection.method, SelectionMethod::TreeDp); + assert_eq!(output.entry_indices(), vec![0, 1]); + // The plans are already timed at query time; exporting them again + // re-times nothing. + let dag = output.execution_timed_dag().expect("exports"); + assert_eq!(dag.roots.len(), 2); +} diff --git a/crates/planner/tests/summary_sharing.rs b/crates/planner/tests/summary_sharing.rs index 41b489e9c..5aee0c3db 100644 --- a/crates/planner/tests/summary_sharing.rs +++ b/crates/planner/tests/summary_sharing.rs @@ -1,35 +1,33 @@ //! Structurally identical summary producers chosen by different queries are -//! shared after Pass 1: one `Rc` across their plans, costed once. +//! shared after Pass 1: one `Rc` across their plans. +use asap_types::ir::cse::share_common_sub_dags; +use asap_types::ir::{ASAPOp, OperatorNode}; use std::rc::Rc; -use asap_aware_mapping::accuracy::{ +use asap_frontend_promql::lower_promql_workload; +use asap_frontend_sql::SqlCatalog; +use asap_logical_optimizer::accuracy::{ AccuracyModel, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, }; -use asap_aware_mapping::cost_model::Cost; -use asap_aware_mapping::pass::{PlanOutput, PlanningModels}; -use asap_aware_mapping::replacement::{default_size_params, DEFAULT_DELTA}; -use asap_aware_mapping::{ - global_selection_with_summary_maintenance_lifecycles, search_workload_with_targets, - ReplacementStrategy, SketchAlgorithmStrategy, WorkloadDemand, +use asap_logical_optimizer::pass1::replacement::{default_size_params, DEFAULT_DELTA}; +use asap_logical_optimizer::{ + search_workload_with_targets, ASAPStrategies, Replacement, ReplacementStrategy, + ReplacementSubDAG, TargetSubDAG, }; -use asap_aware_mapping::{ - CostModel, CostRate, DefaultCostModel, Horizon, LifecycleInput, SummaryMaintenanceCapabilities, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecycleCostInputs, -}; -use asap_frontend_promql::lower_promql_workload; -use asap_frontend_sql::SqlCatalog; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::PlanningModels; +use asap_plan_selection::{CostModel, DefaultCostModel}; +use asap_planner::pass::{PlanOutput, QueryPlan}; use asap_planner::{e2e_plan, FrontendInput, UserInput}; -use asap_types::post_asap::{ - share_common_summary_sub_dags, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, - ProbabilityExpr, ResultGuarantee, SketchStatistic, -}; -use asap_types::post_asap::{ - FieldDataType, SketchAlgorithm, SketchParams, SummaryExpr, SummaryNode, +use asap_types::ir::operator::agg_intent::default_quantile; +use asap_types::ir::operator::AggIntent; +use asap_types::ir::properties::{ + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, }; -use asap_types::pre_asap::agg_intent::default_quantile; -use asap_types::pre_asap::schema::{DataType, Field, Schema}; -use asap_types::pre_asap::{AggIntent, QueryExpr}; +use asap_types::ir::schema::SketchStatistic; +use asap_types::ir::schema::{DataType, Field, Schema}; +use asap_types::ir::schema::{FieldDataType, SketchAlgorithm, SketchKind, SketchParams}; use asap_types::types::AccuracyTarget; use asap_types::workload::{ AccuracyRequirement, DataArrival, DataWorkload, DurationMs, Evidence, LatencyRequirement, @@ -38,16 +36,18 @@ use asap_types::workload::{ }; const NOW_MS: u64 = 1_700_000_000_000; -const HORIZON_S: f64 = 3_600.0; -/// A state costs `build` once however often it is read; raw recomputation -/// costs `raw_per_read` per read. -struct FixedCosts { - build: f64, - raw_per_read: f64, -} +/// Stand-in for the workload-level amortization Stage 2 materialization will +/// price: a sketch candidate costs `preference(kind)` per sketch state, any +/// other candidate more than every sketch. Ranking is otherwise built-in. +struct PreferSketch(fn(&SketchKind) -> f64); + +impl CostModel for PreferSketch { + // Selection takes the cheapest candidate by `estimate_cost`. + fn candidate_cost_covers_complete_plan(&self) -> bool { + true + } -impl CostModel for FixedCosts { fn rank_candidates( &self, intent: &AggIntent, @@ -56,41 +56,42 @@ impl CostModel for FixedCosts { DefaultCostModel.rank_candidates(intent, candidates) } - fn summary_maintenance_lifecycle_cost_inputs( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceLifecycleCostInputs { - SummaryMaintenanceLifecycleCostInputs { - build_cost: Some(Cost(self.build)), - maintenance_cost_per_update: Some(Cost::ZERO), - summary_read_cost: Some(Cost::ZERO), - retention_cost_rate: Some(CostRate(0.0)), - retirement_cost: Some(Cost::ZERO), - } - } - - fn summary_maintenance_capabilities( - &self, - _summary: &SummaryNode, - ) -> SummaryMaintenanceCapabilities { - SummaryMaintenanceCapabilities { - incremental_update: true, - merge: true, - delete: true, + fn estimate_cost(&self, candidate: &ReplacementSubDAG, _: &TargetSubDAG<'_>) -> f64 { + let Replacement::SubDAG(root) = &candidate.replacement else { + return 1e9; + }; + let kinds: Vec<_> = OperatorNode::reachable(root) + .into_iter() + .filter_map(|node| match &node.operator { + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { + family: FieldDataType::Sketch(kind, _), + .. + }) => Some(kind.clone()), + _ => None, + }) + .collect(); + if kinds.is_empty() { + 1e9 + } else { + kinds.iter().map(self.0).sum() } } - - fn raw_query_recompute_cost(&self, _target: &QueryExpr) -> Option { - Some(Cost(self.raw_per_read)) - } } -/// Summaries are far cheaper than raw recomputation, so every query selects -/// one independently and only sharing is under test. -const CHEAP_SUMMARY: FixedCosts = FixedCosts { - build: 1.0, - raw_per_read: 1_000.0, -}; +/// Prefers the largest KLL, i.e. one sized for the strictest consumer. +const PREFER_LARGE_KLL: PreferSketch = PreferSketch(|kind| match kind.params() { + SketchParams::Kll { k } => 1.0 / f64::from(*k), + _ => 1.0, +}); + +/// Prefers UnivMon, which can serve every frequency moment from one state. +const PREFER_UNIVMON: PreferSketch = PreferSketch(|kind| { + if kind.algorithm() == &SketchAlgorithm::UnivMon { + 0.0 + } else { + 1.0 + } +}); fn requirements(epsilon: f64) -> QueryRequirements { QueryRequirements { @@ -110,11 +111,6 @@ fn repeating(query: &str, epsilon: f64) -> RepeatingEntry { } } -fn lifecycle() -> LifecycleInput { - LifecycleInput::new(NOW_MS, SummaryMaintenanceLifecycleCapabilities::default()) - .with_horizon(Horizon(HORIZON_S)) -} - fn promql_workload(queries: &[(&str, f64)]) -> PlanningWorkload { PlanningWorkload { query_workload: QueryWorkload { @@ -142,7 +138,11 @@ fn promql_workload(queries: &[(&str, f64)]) -> PlanningWorkload { } } -async fn plan_promql(queries: &[(&str, f64)], costs: &FixedCosts) -> PlanOutput { +async fn plan_promql(queries: &[(&str, f64)]) -> PlanOutput { + plan_promql_with(queries, &DefaultCostModel).await +} + +async fn plan_promql_with(queries: &[(&str, f64)], cost: &dyn CostModel) -> PlanOutput { let workload = promql_workload(queries); let input = UserInput::new( &workload, @@ -150,13 +150,12 @@ async fn plan_promql(queries: &[(&str, f64)], costs: &FixedCosts) -> PlanOutput now_ms: NOW_MS, histograms: None, }, - PlanningModels::builtin().with_cost(costs), - lifecycle(), + PlanningModels::builtin().with_cost(cost), ); e2e_plan(input).await.expect("workload plans") } -async fn plan_sql(queries: &[&str], costs: &FixedCosts) -> PlanOutput { +async fn plan_sql(queries: &[&str]) -> PlanOutput { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::SQL(SqlDialect::DataFusionSQL), @@ -178,31 +177,39 @@ async fn plan_sql(queries: &[&str], costs: &FixedCosts) -> PlanOutput { let input = UserInput::new( &workload, FrontendInput::Sql { catalog: &catalog }, - PlanningModels::builtin().with_cost(costs), - lifecycle(), + PlanningModels::builtin(), ); e2e_plan(input).await.expect("workload plans") } -/// Every summary state each plan deploys. -fn states(output: &PlanOutput) -> Vec>> { +/// Every summary state (`SummaryAgg`) each plan reaches, in traversal order. +fn plan_states(plan: &QueryPlan) -> Vec> { + OperatorNode::reachable(&plan.root) + .into_iter() + .filter(|node| { + matches!( + node.operator, + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { .. }) + ) + }) + .collect() +} + +/// Every summary state each plan reaches; each plan selects at least one. +fn states(output: &PlanOutput) -> Vec>> { output .plans .iter() .map(|plan| { - assert!(!plan.plan.selected_raw_recompute, "{:?}", plan.plan.root); - assert!(!plan.plan.deployments.is_empty()); - plan.plan - .deployments - .iter() - .map(|deployment| Rc::clone(&deployment.summary)) - .collect() + let states = plan_states(plan); + assert!(!states.is_empty(), "{:?}", plan.root); + states }) .collect() } /// Whether the two plans deploy exactly the same states, by pointer. -fn same_states(states: &[Vec>]) -> bool { +fn same_states(states: &[Vec>]) -> bool { states[0].len() == states[1].len() && states[0] .iter() @@ -210,12 +217,12 @@ fn same_states(states: &[Vec>]) -> bool { .all(|(left, right)| Rc::ptr_eq(left, right)) } -/// The deployments a consumer would run, deduplicated by pointer. +/// The summary states a consumer would run, deduplicated by pointer. fn unique_deployments(output: &PlanOutput) -> usize { - let mut seen: Vec<*const SummaryNode> = Vec::new(); + let mut seen: Vec<*const OperatorNode> = Vec::new(); for plan in &output.plans { - for deployment in &plan.plan.deployments { - let ptr = Rc::as_ptr(&deployment.summary); + for state in plan_states(plan) { + let ptr = Rc::as_ptr(&state); if !seen.contains(&ptr) { seen.push(ptr); } @@ -225,44 +232,25 @@ fn unique_deployments(output: &PlanOutput) -> usize { } /// p50 and p99 over the same window and accuracy read one KLL: the -/// equal-params subset of summary capability. Both plans hold the same `Rc` -/// with the same lifecycle, so a consumer maintains it once. +/// equal-params subset of summary capability. Both plans hold the same `Rc`, +/// so a consumer maintains it once. #[tokio::test] +#[ignore = "Stage 3 selects the raw plan; query-time summaries never cost less until Stage 2 plans materialization: #580"] async fn quantiles_with_equal_params_share_one_producer() { - let output = plan_promql( - &[ - ("quantile_over_time(0.5, lat[5m])", 0.01), - ("quantile_over_time(0.99, lat[5m])", 0.01), - ], - &CHEAP_SUMMARY, - ) + let output = plan_promql(&[ + ("quantile_over_time(0.5, lat[5m])", 0.01), + ("quantile_over_time(0.99, lat[5m])", 0.01), + ]) .await; assert!(same_states(&states(&output))); - assert!(!Rc::ptr_eq( - &output.plans[0].plan.root, - &output.plans[1].plan.root - )); + assert!(!Rc::ptr_eq(&output.plans[0].root, &output.plans[1].root)); assert_eq!(unique_deployments(&output), 1); - let lifecycles: Vec<_> = output - .plans - .iter() - .map(|plan| { - plan.plan.deployments[0] - .summary_maintenance_lifecycle_guarantee - .clone() - }) - .collect(); - assert_eq!(lifecycles[0], lifecycles[1]); - assert!(lifecycles[0].is_some()); - // Each plan is planned against both queries' reads. - for plan in &output.plans { - assert_eq!(plan.plan.expected_reads, Some(12.0)); - } } /// A different window or label selector is a different producer, even when /// one query asks for a stricter accuracy than the other. #[tokio::test] +#[ignore = "Stage 3 selects the raw plan; query-time summaries never cost less until Stage 2 plans materialization: #580"] async fn different_producers_are_not_shared() { for queries in [ [ @@ -282,27 +270,27 @@ async fn different_producers_are_not_shared() { ("quantile_over_time(0.99, lat{job=\"b\"}[5m])", 0.01), ], ] { - let output = plan_promql(&queries, &CHEAP_SUMMARY).await; + let output = plan_promql(&queries).await; assert!(!same_states(&states(&output)), "{queries:?}"); assert_eq!(unique_deployments(&output), 2, "{queries:?}"); for (plan, (_, epsilon)) in output.plans.iter().zip(queries) { - assert_eq!(plan.plan.expected_reads, Some(6.0), "{queries:?}"); assert_eq!(kll_k(plan), kll_k_for(epsilon), "{queries:?}"); } } } /// The KLL `k` of the one state a plan deploys. -fn kll_k(plan: &asap_aware_mapping::pass::QueryLifecyclePlan) -> u32 { - let [deployment] = plan.plan.deployments.as_slice() else { - panic!("one state: {:?}", plan.plan.deployments.len()); +fn kll_k(plan: &QueryPlan) -> u32 { + let states = plan_states(plan); + let [deployment] = states.as_slice() else { + panic!("one state: {:?}", states.len()); }; - let SummaryExpr::SummaryAgg { + let asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. - } = &deployment.summary.expr + }) = &deployment.operator else { - panic!("sketch state: {:?}", deployment.summary.expr); + panic!("sketch state: {:?}", deployment.operator); }; let SketchParams::Kll { k } = kind.params() else { panic!("KLL state: {kind:?}"); @@ -324,19 +312,21 @@ fn kll_k_for(epsilon: f64) -> u32 { } /// p50 at ε=0.01 and p99 at ε=0.001 over the same input share one KLL sized -/// for the strictest consumer; each reader's guarantee meets its own target. +/// for the strictest consumer when the cost model prefers that candidate; each +/// reader's guarantee meets its own target. #[tokio::test] +#[ignore = "Pass 2 cross-query sharing is not planned by the stage pipeline: #580"] async fn quantiles_share_one_producer_sized_for_the_strictest_consumer() { let p50 = ("quantile_over_time(0.5, lat[5m])", 0.01); let p99 = ("quantile_over_time(0.99, lat[5m])", 0.001); assert!(kll_k_for(0.001) > kll_k_for(0.01)); - let output = plan_promql(&[p50, p99], &CHEAP_SUMMARY).await; + let output = plan_promql_with(&[p50, p99], &PREFER_LARGE_KLL).await; assert!(same_states(&states(&output))); assert_eq!(unique_deployments(&output), 1); for (plan, (_, epsilon)) in output.plans.iter().zip([p50, p99]) { assert_eq!(kll_k(plan), kll_k_for(0.001)); - let guarantee = plan.plan.root.guarantee.as_ref().expect("certified"); + let guarantee = plan.root.guarantee.as_ref().expect("certified"); assert!( guarantee.bound.evaluate().unwrap() <= epsilon, "{guarantee:?}" @@ -344,19 +334,16 @@ async fn quantiles_share_one_producer_sized_for_the_strictest_consumer() { } // Alone, the looser query keeps its own, smaller KLL. - let alone = plan_promql(&[p50], &CHEAP_SUMMARY).await; + let alone = plan_promql_with(&[p50], &PREFER_LARGE_KLL).await; assert_eq!(kll_k(&alone.plans[0]), kll_k_for(0.01)); } /// Cross-series quantiles name their KLL state after the input column, not the /// quantile, so p50 and p99 over one selector share it. #[tokio::test] +#[ignore = "Pass 2 cross-query sharing is not planned by the stage pipeline: #580"] async fn cross_series_p50_and_p99_share_one_producer() { - let output = plan_promql( - &[("quantile(0.5, lat)", 0.01), ("quantile(0.99, lat)", 0.01)], - &CHEAP_SUMMARY, - ) - .await; + let output = plan_promql(&[("quantile(0.5, lat)", 0.01), ("quantile(0.99, lat)", 0.01)]).await; assert!(same_states(&states(&output))); assert_eq!(unique_deployments(&output), 1); } @@ -364,9 +351,10 @@ async fn cross_series_p50_and_p99_share_one_producer() { /// An ungrouped aggregate has no unique key, so pre-ASAP CSE keeps the two /// copies apart; their identical producers (rate, then sum) are shared here. #[tokio::test] +#[ignore = "Stage 3 selects the raw plan; query-time summaries never cost less until Stage 2 plans materialization: #580"] async fn identical_ungrouped_queries_share_their_producers() { let query = ("sum(rate(x[5m]))", 0.01); - let output = plan_promql(&[query, query], &CHEAP_SUMMARY).await; + let output = plan_promql(&[query, query]).await; assert!(same_states(&states(&output))); assert_eq!(unique_deployments(&output), 2); } @@ -374,32 +362,33 @@ async fn identical_ungrouped_queries_share_their_producers() { /// The SQL frontend reaches the same sharing for two copies of one filtered /// percentile. #[tokio::test] +#[ignore = "Stage 3 selects the raw plan; query-time summaries never cost less until Stage 2 plans materialization: #580"] async fn identical_sql_percentiles_share_one_producer() { let query = "SELECT approx_percentile_cont(l_extendedprice, 0.5) FROM lineitem WHERE l_orderkey > 10"; - let output = plan_sql(&[query, query], &CHEAP_SUMMARY).await; + let output = plan_sql(&[query, query]).await; assert!(same_states(&states(&output))); assert_eq!(unique_deployments(&output), 1); } -/// The quantile is a readout parameter: SQL p50 and p99 over one filtered +/// The quantile is a evaluation parameter: SQL p50 and p99 over one filtered /// column build one KLL, named after its input, while each query keeps its /// own output column. #[tokio::test] +#[ignore = "Pass 2 cross-query sharing is not planned by the stage pipeline: #580"] async fn sql_p50_and_p99_share_one_producer() { let p50 = "SELECT approx_percentile_cont(l_extendedprice, 0.5) FROM lineitem WHERE l_orderkey > 10"; let p99 = "SELECT approx_percentile_cont(l_extendedprice, 0.99) FROM lineitem WHERE l_orderkey > 10"; - let output = plan_sql(&[p50, p99], &CHEAP_SUMMARY).await; + let output = plan_sql(&[p50, p99]).await; assert!(same_states(&states(&output))); assert_eq!(unique_deployments(&output), 1); let names: Vec<_> = output .plans .iter() .map(|plan| { - plan.plan - .root + plan.root .schema .fields .iter() @@ -425,33 +414,13 @@ async fn sql_p50_and_p99_share_one_producer() { "SELECT approx_percentile_cont(l_orderkey, 0.99) FROM lineitem WHERE l_orderkey > 10", ], ] { - let output = plan_sql(&queries, &CHEAP_SUMMARY).await; + let output = plan_sql(&queries).await; assert!(!same_states(&states(&output)), "{queries:?}"); assert_eq!(unique_deployments(&output), 2, "{queries:?}"); } } -/// A state costs 100 and recomputing a query costs 60 over its six reads: -/// alone, the query recomputes raw. Shared by p50 and p99, the state costs 50 -/// per query, so both keep it. -#[tokio::test] -async fn shared_amortization_alone_can_beat_raw_recompute() { - let costs = FixedCosts { - build: 100.0, - raw_per_read: 10.0, - }; - let p50 = ("quantile_over_time(0.5, lat[5m])", 0.01); - let p99 = ("quantile_over_time(0.99, lat[5m])", 0.01); - - let alone = plan_promql(&[p50], &costs).await; - assert!(alone.plans[0].plan.selected_raw_recompute); - - let output = plan_promql(&[p50, p99], &costs).await; - assert!(same_states(&states(&output))); - assert_eq!(unique_deployments(&output), 1); -} - -/// Synthetic evidence certifying UnivMon readouts; it exercises sharing, never +/// Synthetic evidence certifying UnivMon evaluations; it exercises sharing, never /// runtime accuracy. struct UnivMonEvidence; @@ -489,11 +458,11 @@ impl AccuracyModel for UnivMonEvidence { } /// Distinct count, entropy and L2 over one input, certified by an accuracy -/// model, read one UnivMon state: #515 sharing is the summary-capability rule -/// when the states are identical. `MajorPass` builds candidates with the -/// built-in accuracy model, so this runs its pipeline with the test model. +/// model and selected by a cost model preferring UnivMon, read one UnivMon state: #515 sharing is the summary-capability rule +/// when the states are identical. The facade's stage pipeline does not plan +/// UnivMon sharing, so this runs the legacy search with the test model. #[test] -fn certified_frequency_readouts_share_one_univmon_state() { +fn certified_frequency_evaluations_share_one_univmon_state() { let queries = [ ("distinct_over_time(m[5m])", 0.02), ("entropy_over_time(m[5m])", 0.02), @@ -505,31 +474,13 @@ fn certified_frequency_readouts_share_one_univmon_state() { .into_iter() .zip(queries) .enumerate() - .map(|(index, (expr, (_, epsilon)))| { - (index, Rc::new(expr), Some(AccuracyTarget::Epsilon(epsilon))) - }) + .map(|(index, (expr, (_, epsilon)))| (index, expr, Some(AccuracyTarget::Epsilon(epsilon)))) .collect(); - let strategies: Vec> = - vec![Box::new(SketchAlgorithmStrategy::new_with_planning_inputs( - &CHEAP_SUMMARY, - &UnivMonEvidence, - &EqualSplitAllocator, - ))]; + let strategies: Vec> = vec![Box::new( + ASAPStrategies::new_with_planning_inputs(&UnivMonEvidence, &EqualSplitAllocator), + )]; let space = search_workload_with_targets(roots, &strategies, &UnivMonEvidence); - let entry_indices: Vec = (0..queries.len()).collect(); - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &entry_indices, - }, - NOW_MS, - Some(Horizon(HORIZON_S)), - SummaryMaintenanceLifecycleCapabilities::default(), - &CHEAP_SUMMARY, - ) - .expect("selects"); + let selection = global_selection(&space, &PREFER_UNIVMON); let assembled = space .roots .iter() @@ -541,15 +492,17 @@ fn certified_frequency_readouts_share_one_univmon_state() { (*index, dag) }) .collect(); - let mut states: Vec> = Vec::new(); - for (_, root) in share_common_summary_sub_dags(assembled) { - assert!(root.guarantee.is_some(), "{:?}", root.expr); - let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { - panic!("summary readout: {:?}", root.expr); + let mut states: Vec> = Vec::new(); + for (_, root) in share_common_sub_dags(assembled) { + assert!(root.guarantee.is_some(), "{:?}", root.operator); + let asap_types::ir::Operator::ASAP(ASAPOp::SummaryEstimate { summary_input, .. }) = + &root.operator + else { + panic!("summary evaluation: {:?}", root.operator); }; assert!(matches!( - &summary_input.expr, - SummaryExpr::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. } + &summary_input.operator, + asap_types::ir::Operator::ASAP(ASAPOp::SummaryAgg { family: FieldDataType::Sketch(kind, _), .. }) if kind.algorithm() == &SketchAlgorithm::UnivMon )); states.push(Rc::clone(summary_input)); diff --git a/crates/sql-function-catalog/src/lib.rs b/crates/sql-function-catalog/src/lib.rs index a95a2a632..5ca56862d 100644 --- a/crates/sql-function-catalog/src/lib.rs +++ b/crates/sql-function-catalog/src/lib.rs @@ -74,7 +74,7 @@ pub enum Arity { /// classification [`lower_agg_intent`](../asap_frontend_sql/index.html) /// switches on to build the real `AggIntent`. /// -/// Deliberately *not* `asap_types::pre_asap::agg_intent::AggIntent` itself: +/// Deliberately *not* `asap_types::ir::operator::agg_intent::AggIntent` itself: /// most `AggIntent` variants carry call-site-only state that isn't a /// function of the name alone -- the ambient `AccuracyTarget` (thread-local, /// not catalog data), φ pulled from a call's literal 2nd argument, whether @@ -277,7 +277,7 @@ pub struct ClickHouseBuiltin { pub const CLICKHOUSE_BUILTINS: &[ClickHouseBuiltin] = &[ // Explicit time-series reducers. These deliberately survive under their // own names: the SQL frontend validates (value, timestamp, window_ms) and - // lowers the window to QueryExpr::TimeRange rather than pretending these + // lowers the window to NonASAPOp::TimeRange rather than pretending these // are ordinary tabular aggregates. ClickHouseBuiltin { name: "asap_rate", diff --git a/crates/types/Cargo.toml b/crates/types/Cargo.toml index 8572854a9..0e10ccdf4 100644 --- a/crates/types/Cargo.toml +++ b/crates/types/Cargo.toml @@ -8,7 +8,7 @@ edition = "2021" # execution logic — removed, no real implementor existed; see issue #190). # No internal deps. [dependencies] -# "rc" — QueryExpr's child fields are Rc> (issue #212, #222: +# "rc" — OperatorNode child fields are Rc (issue #212, #222: # shared sub-expressions), and Rc's Serialize/Deserialize impls live behind # this feature flag. dag_export.rs / DAGNode already flatten the DAG to a # node+edge list for JSON export, so this does not change wire format — a diff --git a/crates/types/src/cost.rs b/crates/types/src/cost.rs index 660d3bb73..1bab8ea1e 100644 --- a/crates/types/src/cost.rs +++ b/crates/types/src/cost.rs @@ -73,7 +73,7 @@ pub enum BaselineRef { /// — "do nothing" (never apply ASAP-aware replacement at all). PreAsapRecomputation, /// The best-ranked *non-selected* legal candidate for the same target - /// (`rank` into that target's own `CandidateLogicalASAPDAGs::cost_sorted` ordering, + /// (`rank` into that target's own `candidate_selection::cost_sorted` ordering, /// `0` = best; a baseline referencing this variant is always `rank >= /// 1`, since `rank 0` is what got selected). HighestRankedNonSelectedCandidate { rank: usize }, diff --git a/crates/types/src/dag_export.rs b/crates/types/src/dag_export.rs index f26673ea7..6390ac189 100644 --- a/crates/types/src/dag_export.rs +++ b/crates/types/src/dag_export.rs @@ -1,45 +1,59 @@ -//! Export the pre-ASAP [`QueryExpr`] DAG as a generic node/edge DAG, for tools -//! that need to render or diff the IR (the `dag_export` example + the -//! `tools/dag-viewer` viewer — see issue #133) rather than walk it in Rust. +//! Export an [`OperatorNode`] DAG as a generic node/edge dag, for tools +//! that need to render or diff the IR (the `dag_export` devtools binary + +//! the `tools/dag-viewer` viewer — see issue #133) rather than walk it in +//! Rust. //! -//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged DAG -//! (`Rc` children nested inside each variant's own field). This module -//! flattens that into an explicit node list + child-id edges — the shape a -//! generic DAG renderer wants — and additionally tags each node with -//! [`structural_hash`](crate::pre_asap::cse::structural_hash), so a caller -//! with several exported queries can spot identical sub-DAGs (a -//! shared `Scan`, a repeated `Aggregate` shape, …) by comparing hashes -//! rather than re-implementing `QueryExpr: PartialEq` structural comparison -//! client-side. +//! `OperatorNode` already derives `Serialize`, but as a Rust-shaped tagged +//! tree (`Rc` children nested inside each variant's own field, repeated once +//! per reference). This module flattens that into an explicit node list + +//! child-id edges — one entry per unique node, deduplicated by `Rc` pointer +//! identity, so a shared sub-DAG stays one node with several parents — and +//! additionally tags each node with +//! [`structural_hash`](crate::ir::cse::structural_hash), so a caller with +//! several exported queries can spot identical sub-DAGs (a shared `Scan`, a +//! repeated `Aggregate` shape, …) by comparing hashes rather than +//! re-implementing structural comparison client-side. //! //! This is literally the same hashing -//! [`share_common_sub_dags`](crate::pre_asap::cse::share_common_sub_dags) -//! uses to bucket candidates in its `InternTable` (issue #223 stage 3) — not -//! a parallel reimplementation. `tools/dag-viewer`'s "shared sub-DAG" +//! [`share_common_sub_dags`](crate::ir::cse::share_common_sub_dags) uses to +//! bucket candidates in its `InternTable` (issue #223 stage 3) — not a +//! parallel reimplementation. `tools/dag-viewer`'s "shared sub-DAG" //! highlighting is still a *proxy* for real CSE, though: a hash match here //! only means two nodes are legal `InternTable` bucket-mates (same coarse -//! hash), the same candidate-narrowing step `structural_hash` performs -//! inside `InternTable::intern` — it does not mean `share_common_sub_dags` -//! actually ran on this data and merged them onto one `Rc` (that also -//! requires the `PartialEq` check `InternTable::intern` performs, and the +//! hash) — it does not mean `share_common_sub_dags` actually ran on this +//! data and merged them onto one `Rc` (that also requires the structural +//! equality check `InternTable::intern` performs, and the //! `Schema::has_unique_key` legality gate, neither of which this export //! step evaluates). See `tools/dag-viewer/README.md` for the up-to-date //! caveat. //! +//! There is one IR before and after ASAP optimization, so there is one +//! exporter: an ordinary operator and an ASAP summary operator are both +//! rendered by the same per-variant [`shape`] match, whichever entry point +//! ([`export`], [`export_summary`], [`export_post_asap`]) reached them. +//! +//! ## Scalar expressions +//! +//! A [`ScalarExpr`] is owned by value by an operator field (`Filter.pred`, +//! `Project.cols`, …) and is rendered into that operator's `detail`, not as +//! a node of its own. The operator nodes a scalar expression reads +//! (`scalar(v)`, `EXISTS (subquery)`, …) *are* nodes of the dag — they are +//! in [`OperatorNode::children`] — so inside `detail` each such reference is +//! rendered as `{"scalar_ref": }` rather than inlined. +//! //! ## `DAGNode::notes` — a layering seam, not a feature this module implements //! //! [`DAGNode`] also carries `notes: Vec<`[`DAGNote`]`>`, always empty coming //! out of [`export`]. It exists so a *higher* layer — one that depends on //! `asap_types`, never the reverse — can annotate an already-exported DAG //! after the fact without this module needing to know anything about that -//! layer's concepts. Concretely: `asap-aware-mapping`'s `explanation` module -//! (issue #257) computes `structural_hash` over the same `QueryExpr` -//! sub-DAGs this module does (via the identical function). The devtools -//! exporter uses that hash to narrow candidates, then compares -//! `ReplacementExplanation::target` with [`DAGNode::source_expr`] for a -//! collision-safe match before pushing a [`DAGNote`] onto the node. -//! `asap_types` itself never constructs a `DAGNote` — see [`DAGNode::notes`] -//! for the layering rule this keeps. +//! layer's concepts. Concretely: `asap-logical-optimizer`'s `explanation` module +//! (issue #257) computes `structural_hash` over the same nodes this module +//! does (via the identical function). The devtools exporter uses that hash +//! to narrow candidates, then compares its target with +//! [`DAGNode::source_node`] for a collision-safe match before pushing a +//! [`DAGNote`] onto the node. `asap_types` itself never constructs a +//! `DAGNote` — see [`DAGNode::notes`] for the layering rule this keeps. use std::collections::HashMap; use std::rc::Rc; @@ -47,9 +61,11 @@ use std::rc::Rc; use serde::Serialize; use crate::cost::CostAnnotation; -use crate::post_asap::{AccuracyError, ResultGuarantee, SummaryExpr, SummaryNode}; -use crate::pre_asap::cse::{structural_hash, HashCache}; -use crate::pre_asap::query_expr::{QueryExpr, Source}; +use crate::ir::cse::{structural_hash, HashCache}; +use crate::ir::operator::operator_properties::Source; +use crate::ir::properties::{AccuracyError, ResultGuarantee}; +use crate::ir::schema::FieldDataType; +use crate::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode, ScalarExpr}; /// One flattened IR node. `detail` holds this node's own scalar fields /// (predicates, aggregate funcs, schema, sort keys, …) — everything except @@ -57,60 +73,52 @@ use crate::pre_asap::query_expr::{QueryExpr, Source}; #[derive(Debug, Clone, Serialize)] pub struct DAGNode { pub id: u32, - /// The `QueryExpr` variant name (e.g. `"Aggregate"`). + /// The operator variant name — [`Operator::kind_name`] (e.g. + /// `"Aggregate"`, `"SummaryAgg"`). pub kind: &'static str, /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, - /// Output schema carried by every exported node. Edge renderers use the - /// child node's schema as the schema flowing along child → consumer. + /// Output schema carried by every exported node ([`OperatorNode::schema`] + /// as JSON). Edge renderers use the child node's schema as the schema + /// flowing along child → consumer. #[serde(skip_serializing_if = "Option::is_none")] pub schema: Option, - /// Child node ids, in the variant's field order (e.g. `Join` is - /// `[left, right]`). + /// Child node ids in [`OperatorNode::children`] order: the operator's + /// own inputs in field order (e.g. `Join` is `[left, right]`), then the + /// nodes referenced from its scalar expressions. pub children: Vec, /// Explicit workload-wide identity assigned by a higher-level exporter. /// Viewers use this field to union nodes and must not reconstruct a /// structural signature client-side. #[serde(skip_serializing_if = "Option::is_none")] pub workload_node_id: Option, - /// [`structural_hash`](crate::pre_asap::cse::structural_hash) of the - /// sub-DAG rooted at this node — the exact same function `cse`'s - /// `InternTable` uses to bucket CSE candidates, so two nodes hash - /// equally here iff they would land in the same `InternTable` bucket. - /// See the module doc for what a hash match here does and doesn't - /// guarantee. - /// - /// `None` for the same reason `source_expr` is `None` — a post-ASAP- - /// originated node in an [`export_post_asap`] merged DAG has no - /// `QueryExpr` to hash. Omitted from JSON entirely (rather than, say, - /// serialized as `0`) so a consumer's shared-sub-DAG-by-hash pass can - /// tell "no hash" apart from a real hash that happens to collide with a - /// placeholder — `0` is a legal `structural_hash` output, not a safe - /// sentinel. + /// [`structural_hash`](crate::ir::cse::structural_hash) of the sub-DAG + /// rooted at this node — the exact same function `cse`'s `InternTable` + /// uses to bucket CSE candidates, so two nodes hash equally here iff they + /// would land in the same `InternTable` bucket. See the module doc for + /// what a hash match here does and doesn't guarantee. Always `Some` + /// for a node this module produces; the `Option` is retained for the + /// JSON shape (`None` is omitted rather than serialized as a sentinel, + /// since `0` is a legal hash). #[serde(skip_serializing_if = "Option::is_none")] pub hash: Option, - /// Exact source expression for in-process annotation matching. It is not - /// part of the JSON format: callers first narrow by `hash`, then compare - /// this value structurally to avoid treating a hash collision as node - /// identity. - /// - /// `None` for a node with no corresponding pre-ASAP `QueryExpr` at all — - /// only possible for a post-ASAP-originated node inside a merged - /// [`export_post_asap`] DAG (a `SummaryAgg`/`SummaryJoin`/… node has no - /// single `QueryExpr` it corresponds to). Every node [`export`] itself - /// produces is pre-ASAP by construction and always carries `Some`. + /// The exported node itself, for in-process annotation matching. Not + /// part of the JSON format: callers first narrow by `hash`, then + /// compare this value (by pointer or structurally) to avoid treating a + /// hash collision as node identity. Always `Some` for a node this + /// module produces. #[serde(skip)] - pub source_expr: Option, - /// In-process identity of the source `QueryExpr`. Unlike `source_expr`'s - /// structural value, this preserves an `Rc` child reached from multiple - /// parents so post-ASAP flattening can retain true DAG sharing. + pub source_node: Option>, + /// In-process identity of `source_node` (`Rc::as_ptr` as an address): + /// the key the builder deduplicates on, so a node reached from several + /// parents is exported once. Not part of the JSON format. #[serde(skip)] - source_ptr: Option, + pub source_ptr: Option, /// Arbitrary reporting-layer annotations for this node — e.g. why a /// replacement exists here. `asap_types` never populates this itself /// (it has no notion of a "replacement" at all — see the module doc's - /// layering note); a higher layer that does (`asap-aware-mapping`, via + /// layering note); a higher layer that does (`asap-logical-optimizer`, via /// the `dag_export` devtools binary) fills it in after the fact by /// matching [`DAGNode::hash`] and confirming structural equality. Empty /// by default, so every existing [`export`] caller and test is unaffected. @@ -126,18 +134,18 @@ pub struct DAGNode { /// One reporting-layer annotation attached to a [`DAGNode`] by a higher /// layer than `asap_types` — see [`DAGNode::notes`]. `asap_types` defines /// this shape (so the field has a concrete, serializable type) but never -/// constructs one: `asap_types` is a lower crate that `asap-aware-mapping` +/// constructs one: `asap_types` is a lower crate that `asap-logical-optimizer` /// depends on, never the reverse, so this type is deliberately generic and /// crate-agnostic rather than naming anything from that higher layer (e.g. /// its `ExplanationKind`/`ReplacementExplanation`). #[derive(Debug, Clone, Serialize)] pub struct DAGNote { /// A short tag for the kind of annotation this is (e.g. a - /// `Debug`-formatted `asap_aware_mapping::ExplanationKind`) — opaque to + /// `Debug`-formatted `asap_logical_optimizer::ExplanationKind`) — opaque to /// `asap_types`, meant for a renderer to group or color by. pub kind: String, /// Human-readable explanation text (e.g. an - /// `asap_aware_mapping::ReplacementExplanation::reason`). + /// `asap_logical_optimizer::ReplacementExplanation::reason`). pub reason: String, } @@ -187,7 +195,7 @@ pub struct EdgeCostAnnotation { pub cost: CostAnnotation, } -/// One query's exported DAG. `nodes[root as usize]` is the DAG's root. +/// One query's exported dag. `nodes[root as usize]` is the DAG's root. #[derive(Debug, Clone, Serialize)] pub struct ExportDAG { pub nodes: Vec, @@ -195,8 +203,7 @@ pub struct ExportDAG { /// See [`EdgeCostAnnotation`]. Always empty unless a higher layer /// explicitly populated it (same layering rule as [`DAGNode::notes`]); /// omitted from JSON entirely when empty, so every existing producer of - /// [`ExportDAG`] (every call to [`export`]/[`export_summary`]) is - /// unaffected. + /// [`ExportDAG`] is unaffected. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub edge_annotations: Vec, } @@ -205,17 +212,17 @@ pub struct ExportDAG { #[derive(Debug, Clone, Serialize)] pub struct NamedDAG { pub name: String, - /// The original query text (SQL or PromQL) this DAG was lowered from, - /// for display alongside the DAG — not used by `export` itself, since - /// that only sees the already-lowered `QueryExpr`. Optional because not - /// every producer of a `NamedDAG` has the source text on hand. + /// The original query text (SQL or PromQL) this dag was lowered from, + /// for display alongside the dag — not used by `export` itself, since + /// that only sees the already-lowered DAG. Optional because not every + /// producer of a `NamedDAG` has the source text on hand. #[serde(skip_serializing_if = "Option::is_none")] pub source: Option, pub dag: ExportDAG, /// Concrete post-ASAP replacement sites discovered for this query — see /// [`TargetReplacement`]. Always empty coming out of anything in this /// module (same layering rule as [`DAGNode::notes`]: `asap_types` never - /// runs `asap-aware-mapping`'s search itself); a higher layer populates + /// runs `asap-logical-optimizer`'s search itself); a higher layer populates /// this after the fact, e.g. the `dag_export` devtools binary's /// `--post-asap` flag. Omitted from the JSON entirely when empty, so /// every existing producer/consumer of `NamedDAG` (in particular every @@ -228,15 +235,13 @@ pub struct NamedDAG { /// [`TargetReplacement::before`]/`::after` (small, self-contained /// before/after pairs, one per independently-discovered replacement /// site), this is a single flattened [`ExportDAG`] spanning the whole - /// query: every node that has no winning replacement renders as an - /// ordinary pre-ASAP [`DAGNode`] (same shape [`export`] itself - /// produces), and every node that does splices in its winning - /// candidate's shape instead — a rewritten [`QueryExpr`] sub-DAG, or a - /// bound `SummaryNode` sub-DAG, rendered inline in the very same node - /// list. `None` unless a higher layer explicitly built one (e.g. the - /// `dag_export` devtools binary's `--post-asap` flag); omitted from the - /// JSON entirely when absent, so every existing producer/consumer of - /// `NamedDAG` is unaffected. + /// query: every node that has no winning replacement renders as it does + /// in [`export`], and every node that does splices in its winning + /// candidate's sub-DAG instead, in the very same node list. `None` + /// unless a higher layer explicitly built one (e.g. the `dag_export` + /// devtools binary's `--post-asap` flag); omitted from the JSON entirely + /// when absent, so every existing producer/consumer of `NamedDAG` is + /// unaffected. #[serde(default, skip_serializing_if = "Option::is_none")] pub post_dag: Option, /// This query's own selected-workload cost/benefit — one of issue @@ -282,85 +287,55 @@ pub struct WorkloadDAG { pub workload_cost: Option, } -// ── Post-ASAP replacement export — a second, layering-seam-shaped feature ── +// ── Post-ASAP replacement export — a layering-seam-shaped feature ────────── // -// Everything below this point is the post-ASAP counterpart of the pre-ASAP -// flattening above: [`export_summary`] flattens a `SummaryNode` the same way -// [`export`] flattens a `QueryExpr`, and [`TargetReplacement`] is the -// generic, crate-agnostic "one replacement site, before and after" shape a -// higher layer (`asap-aware-mapping`, via the `dag_export` devtools binary's -// `--post-asap` flag) populates after running its own search — the exact -// same layering rule [`DAGNode::notes`]'s doc above already states: this -// module never runs `asap_aware_mapping::replacement::search_workload_with` -// itself, never picks a "winning" candidate, and has no opinion on what a +// [`TargetReplacement`] is the generic, crate-agnostic "one replacement +// site, before and after" shape a higher layer (`asap-logical-optimizer`, via +// the `dag_export` devtools binary's `--post-asap` flag) populates after +// running its own search — the exact same layering rule [`DAGNode::notes`]'s +// doc above already states: this module never runs +// `asap_logical_optimizer::pass1::replacement::search_workload_with` itself, never +// picks a "winning" candidate, and has no opinion on what a // `ReplacementProvenance` or a cost model even is. It only defines shapes // concrete and serializable enough for a higher layer to fill in, and for // `tools/dag-viewer` to render without needing to know anything about -// `asap-aware-mapping`'s own vocabulary. -// -// A single whole-query "post-ASAP DAG" isn't attempted here, and isn't -// representable in the current type system either: `SummaryExpr` has no -// variant letting a `SummaryNode` be embedded back inside a plain -// `QueryExpr`'s child slot (`QueryExpr`'s own children are always -// `Rc`, never `Rc`), so there is no way to splice a -// post-ASAP binding back into its original pre-ASAP DAG in place. Inventing -// a bridge type for that is a real `asap_types`/`asap-aware-mapping` IR -// design decision, well beyond what a devtools visualization export should -// decide unilaterally. Instead, each independently-discovered replacement -// target gets its own small, self-contained `before`/`after` pair — the -// target's own pre-ASAP sub-DAG, and either the winning `SummaryNode` or the -// winning rewritten `QueryExpr`, both of which *are* fully representable -// today via [`export`]/[`export_summary`] as-is. - -/// One flattened post-ASAP node — the [`SummaryExpr`] analogue of -/// [`DAGNode`]. `detail` holds this node's own scalar fields (the summarized -/// column, the summary family, grouping strategy, sketch-query kind, …) — -/// everything except its `SummaryNode` children, which live in `children` -/// instead. -/// -/// Unlike [`DAGNode`], this carries no `hash`/`source_expr` pair: nothing in -/// this module ever needs to re-identify a particular `SummaryDAGNode` the -/// way `DAGNode::hash` lets a higher layer re-identify a pre-ASAP node (a -/// `SummaryNode` is always freshly exported for exactly one -/// [`TargetReplacementAfter::Summary`] site, never matched back against a -/// separately-exported DAG the way pre-ASAP notes are). -/// -/// Several of `SummaryExpr`'s own fields (`FieldDataType`, -/// `GroupingStrategy`, `SketchStatistic`) derive neither `Serialize` nor -/// `Deserialize` in `asap_types::post_asap` — they carry no reporting -/// obligation there, since nothing before this module ever needed to -/// serialize a post-ASAP node. Rather than adding `Serialize` impls to -/// `post_asap`'s own core types purely for this devtools-facing export (a -/// change to that module's own public API contract, out of scope for a -/// reporting concern), this module renders those particular fields into -/// `detail` via their `Debug` formatting instead — human-readable, and -/// sufficient for the display purpose `detail` exists for on every other -/// node in this file (see [`DAGNode::detail`]'s own doc), at the cost of -/// those particular fields being opaque strings rather than structured JSON -/// on the `SummaryDAGNode` side of the export. +// `asap-logical-optimizer`'s own vocabulary. + +/// One flattened node of a [`SummaryDAG`] — the same node as a +/// [`DAGNode`], in the shape the summary-maintenance consumers read: +/// snake_case `kind`, the accuracy guarantee as its own field, no +/// hash/annotation seams. #[derive(Debug, Clone, Serialize)] pub struct SummaryDAGNode { pub id: u32, - /// The `SummaryExpr` variant name (e.g. `"SummaryAgg"`). + /// The operator variant name in snake_case (e.g. `"summary_agg"`, + /// `"scan"`) — see [`snake_case_kind`]. pub kind: &'static str, /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, - /// Child node ids, in the variant's field order (e.g. `SummaryJoin` is - /// `[outer, inner]`). + /// [`OperatorNode::schema`] as JSON. + #[serde(skip_serializing_if = "Option::is_none")] + pub schema: Option, + /// Child node ids in [`OperatorNode::children`] order. pub children: Vec, /// The value's machine-readable accuracy guarantee (issue #172) — - /// [`SummaryNode::guarantee`] serialized structurally (metric, symbolic + /// [`OperatorNode::guarantee`] serialized structurally (metric, symbolic /// bound, failure probability, provenance including any budget /// allocation), not as prose. Omitted when the node carries none (raw /// summary state, or a family with no error model), so every consumer /// predating this field parses the same shape it always has. #[serde(default, skip_serializing_if = "Option::is_none")] pub guarantee: Option, + /// The exported node itself, so a caller annotating the dag can find + /// a node by `Rc` pointer identity rather than by walk order. Not part + /// of the JSON format. Always `Some`. + #[serde(skip)] + pub source_node: Option>, } /// One accuracy-illegal candidate a higher layer's search refused for a -/// target (issue #172) — `asap_aware_mapping::replacement::RejectedCandidate` +/// target (issue #172) — `asap_logical_optimizer::pass1::replacement::RejectedCandidate` /// re-shaped into this crate's own crate-agnostic vocabulary, the same /// layering rule as [`TargetReplacement`]. Carried on /// [`NamedDAG::rejections`] so a renderer can explain *why* a target kept @@ -378,242 +353,17 @@ pub struct TargetRejection { pub error: AccuracyError, } -/// One post-ASAP `SummaryNode` DAG, flattened the same way [`ExportDAG`] -/// flattens a pre-ASAP `QueryExpr` DAG. +/// A DAG flattened into [`SummaryDAGNode`]s — the same dag [`ExportDAG`] +/// holds, in the summary-maintenance consumers' node shape. #[derive(Debug, Clone, Serialize)] pub struct SummaryDAG { pub nodes: Vec, pub root: u32, } -/// Flatten a [`SummaryNode`] the same way [`export`] flattens a `QueryExpr` -/// — post-order, one [`SummaryDAGNode`] per [`SummaryExpr`] variant, no -/// memoization of repeated `Rc` references (a shared -/// sub-expression reachable through two parents is flattened twice, into two -/// separate node entries — the same "this is a flattened DAG view, not a -/// pointer-identity-preserving DAG" behavior [`build`] already has for -/// `QueryExpr`). -/// -/// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP sub-DAG beneath it -/// as a nested [`ExportDAG`] (via [`export(inner)`](export)) inside its own -/// `detail` field (`{"pre_asap_sub_dag": }`) rather than trying to -/// flatten it into this same node list — [`DAGNode`] and [`SummaryDAGNode`] -/// are different types with different id spaces, so mixing them into one -/// `Vec` isn't type-safe; nesting is. `label` for a `KeepPreAsap` node is -/// `format!("KeepPreAsap({kind})")`, where `kind` is the inner sub-DAG's own -/// top-level `DAGNode::kind`. -pub fn export_summary(node: &SummaryNode) -> SummaryDAG { - let mut nodes = Vec::new(); - let root = build_summary(node, &mut nodes); - SummaryDAG { nodes, root } -} - -fn push_summary_node( - nodes: &mut Vec, - kind: &'static str, - label: String, - detail: serde_json::Value, - children: Vec, - guarantee: Option, -) -> u32 { - let id = nodes.len() as u32; - nodes.push(SummaryDAGNode { - id, - kind, - label, - detail, - children, - guarantee, - }); - id -} - -/// A short, human-readable label for a [`crate::post_asap::FieldDataType`] -/// (e.g. `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for -/// [`SummaryDAGNode::label`] text on a `SummaryAgg`/`SummaryJoin` node. Not -/// exhaustive prose (mirrors `asap_aware_mapping::replacement::describe_intent`'s -/// own "this is a label, not a decision" stance) — every variant is covered, -/// but via `Debug` for the inner kind rather than hand-written prose per -/// algorithm. -fn family_label(family: &crate::post_asap::FieldDataType) -> String { - use crate::post_asap::FieldDataType; - match family { - FieldDataType::Plain(dtype) => format!("Plain({dtype:?})"), - FieldDataType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), - FieldDataType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), - FieldDataType::Sample(kind, _) => format!("Sample({kind:?})"), - FieldDataType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), - FieldDataType::StatModel(kind, _) => format!("StatModel({kind:?})"), - } -} - -/// `(kind, label, detail)` for every [`SummaryExpr`] variant *except* -/// [`SummaryExpr::KeepPreAsap`] — that variant has no `SummaryDAGNode`/ -/// `DAGNode` of its own (see [`build_summary`]/[`build_summary_hybrid`], its -/// only two callers, both of which special-case it before ever reaching -/// this function). Factored out so [`build_summary`] (nests a `KeepPreAsap` -/// leaf's pre-ASAP sub-DAG as its own [`SummaryDAG`]) and -/// [`build_summary_hybrid`] (splices that same sub-DAG directly into a -/// shared [`ExportDAG`] node list — see [`export_post_asap`]) can't drift -/// apart on how every *other* variant's own shape is described, since -/// nothing about that description differs between the two. -macro_rules! define_summary_kind_tags { - ($($pattern:pat => $tag:literal),+ $(,)?) => { - #[cfg(test)] - const SUMMARY_KIND_TAGS: &[&str] = &[$($tag),+]; - - fn summary_kind_tag(expr: &SummaryExpr) -> &'static str { - match expr { - SummaryExpr::KeepPreAsap(_) => unreachable!( - "summary_kind_tag's callers special-case KeepPreAsap" - ), - $($pattern => $tag),+ - } - } - }; -} - -define_summary_kind_tags! { - SummaryExpr::BinaryOp { .. } => "SummaryBinaryOp", - - SummaryExpr::ValueOperation { .. } => "ValueOperation", - SummaryExpr::RelationalJoin { .. } => "RelationalJoin", - SummaryExpr::SummaryAgg { .. } => "SummaryAgg", - SummaryExpr::SummaryJoin { .. } => "SummaryJoin", - SummaryExpr::SummarySubtract { .. } => "SummarySubtract", - SummaryExpr::SummaryDelete { .. } => "SummaryDelete", - SummaryExpr::SummaryEstimate { .. } => "SummaryEstimate", - SummaryExpr::SummaryMerge { .. } => "SummaryMerge", -} - -fn summary_shape(expr: &SummaryExpr) -> (&'static str, String, serde_json::Value) { - let kind = summary_kind_tag(expr); - match expr { - SummaryExpr::KeepPreAsap(_) => { - unreachable!("summary_shape's callers special-case KeepPreAsap before calling it") - } - SummaryExpr::BinaryOp { operator, .. } => { - let label = format!("BinaryOp({:?})", operator.kind); - let detail = serde_json::json!({ - "kind": format!("{:?}", operator.kind), - "vector_match": operator.vector_match, - }); - (kind, label, detail) - } - - SummaryExpr::ValueOperation { - operation, timing, .. - } => ( - kind, - format!("ValueOperation({operation:?})"), - serde_json::json!({ - "operation": format!("{operation:?}"), - "timing": timing.as_str(), - }), - ), - SummaryExpr::RelationalJoin { - kind: join_kind, - pred, - .. - } => ( - kind, - format!("RelationalJoin({join_kind:?})"), - serde_json::json!({ "join_kind": join_kind, "predicate": pred }), - ), - SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - .. - } => { - let label = format!("SummaryAgg({})", family_label(family)); - let detail = serde_json::json!({ - "family": format!("{family:?}"), - "input": input, - "reduction": reduction, - "grouping": format!("{grouping:?}"), - }); - (kind, label, detail) - } - SummaryExpr::SummaryJoin { key, family, .. } => { - let label = format!("SummaryJoin({})", family_label(family)); - let detail = serde_json::json!({ - "key": key, - "family": format!("{family:?}"), - }); - (kind, label, detail) - } - SummaryExpr::SummarySubtract { .. } => { - (kind, "SummarySubtract".into(), serde_json::json!({})) - } - SummaryExpr::SummaryDelete { key, .. } => { - let detail = serde_json::json!({ "key": key }); - (kind, "SummaryDelete".into(), detail) - } - SummaryExpr::SummaryEstimate { query, .. } => { - let label = format!("SummaryEstimate({query:?})"); - let detail = serde_json::json!({ "query": format!("{query:?}") }); - (kind, label, detail) - } - SummaryExpr::SummaryMerge { children, .. } => { - let label = format!("SummaryMerge({} children)", children.len()); - (kind, label, serde_json::json!({})) - } - } -} - -/// `expr`'s own `Rc` children, in the variant's field order -/// (e.g. `SummaryJoin` is `[outer, inner]`) — empty for -/// [`SummaryExpr::KeepPreAsap`], which has no `SummaryNode` children at all -/// (only a boxed pre-ASAP `QueryExpr`). Shared by [`build_summary`] and -/// [`build_summary_hybrid`] for the same reason [`summary_shape`] is. -fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { - match expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => vec![lhs, rhs], - - SummaryExpr::ValueOperation { child, .. } => vec![child], - SummaryExpr::RelationalJoin { left, right, .. } => vec![left, right], - SummaryExpr::SummaryAgg { child, .. } => vec![child], - SummaryExpr::SummaryJoin { outer, inner, .. } => vec![outer, inner], - SummaryExpr::SummarySubtract { left, right } => vec![left, right], - SummaryExpr::SummaryDelete { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryEstimate { summary_input, .. } => vec![summary_input], - SummaryExpr::SummaryMerge { children, .. } => children.iter().collect(), - } -} - -/// Recursively flatten `node`, appending [`SummaryDAGNode`]s to `nodes` in -/// post-order (children pushed before their parent), and return the pushed -/// root's id. Exhaustive over every [`SummaryExpr`] variant, matching this -/// file's own exhaustive style for `QueryExpr` in [`build`]. -fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { - if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - let pre_asap_sub_dag = export(inner); - let inner_kind = pre_asap_sub_dag.nodes[pre_asap_sub_dag.root as usize].kind; - let label = format!("KeepPreAsap({inner_kind})"); - let detail = serde_json::json!({ "pre_asap_sub_dag": pre_asap_sub_dag }); - return push_summary_node( - nodes, - "KeepPreAsap", - label, - detail, - vec![], - node.guarantee.clone(), - ); - } - let children: Vec = summary_children(&node.expr) - .into_iter() - .map(|child| build_summary(child, nodes)) - .collect(); - let (kind, label, detail) = summary_shape(&node.expr); - push_summary_node(nodes, kind, label, detail, children, node.guarantee.clone()) -} - /// One replacement site a higher layer (the `dag_export` binary) found by -/// running `asap_aware_mapping::replacement::search_workload_with` + -/// `CandidateLogicalASAPDAGs::cost_sorted` and picking the best-ranked candidate for one +/// running `asap_logical_optimizer::pass1::replacement::search_workload_with` + +/// `candidate_selection::cost_sorted` and picking the best-ranked candidate for one /// `TargetSubDAGCandidates` — `asap_types` never runs that search itself (same layering /// rule as [`DAGNote`]: this crate defines the shape, a higher crate /// populates it). @@ -624,7 +374,7 @@ pub struct TargetReplacement { /// so renderers can explain a clicked post-ASAP node without guessing by /// label, hash, or DAG shape. pub decision_id: u32, - /// Id of the [`DAGNode`] (in this query's own `DAG.nodes`, i.e. the + /// Id of the [`DAGNode`] (in this query's own `dag.nodes`, i.e. the /// [`NamedDAG`] this `TargetReplacement` is attached to) this /// replacement's `before` sub-DAG is rooted at. pub target_pre_id: u32, @@ -640,7 +390,7 @@ pub struct TargetReplacement { /// not re-derived here). pub rationale: String, /// This candidate's rank among its `TargetSubDAGCandidates`'s alternatives after - /// `CandidateLogicalASAPDAGs::cost_sorted` (`0` = best). Exposed so a renderer can show + /// `candidate_selection::cost_sorted` (`0` = best). Exposed so a renderer can show /// "this was the best of N candidates" without re-deriving the ranking. pub rank: usize, /// This candidate's own estimated cost, straight off @@ -648,7 +398,7 @@ pub struct TargetReplacement { /// doesn't estimate a numeric cost for this candidate shape (see that /// field's own doc upstream). pub cost: f64, - /// The target's own pre-ASAP sub-DAG, before replacement — literally + /// The target's own sub-DAG, before replacement — literally /// `export(target)` for the `TargetSubDAGCandidates`'s own `target`, reused as-is. pub before: ExportDAG, pub after: TargetReplacementAfter, @@ -668,8 +418,10 @@ pub struct TargetReplacement { } /// What a [`TargetReplacement`] became — either a genuine post-ASAP binding -/// or a still-pre-ASAP-shaped structural rewrite, mirroring -/// `asap_aware_mapping::replacement::Replacement`'s own two variants. +/// or a still-relational structural rewrite, mirroring +/// `asap_logical_optimizer::pass1::replacement::Replacement`'s own two variants. Both +/// carry an ordinary [`ExportDAG`]: the unified IR renders a summary sub-DAG +/// and a rewritten relational sub-DAG through the same [`export`]. /// /// Serializes as `{"kind": "Summary"|"Rewrite", "DAG": {...}}` (serde's /// adjacently-tagged representation for a `#[serde(tag = "kind", content = @@ -680,77 +432,87 @@ pub struct TargetReplacement { #[serde(tag = "kind", content = "dag")] pub enum TargetReplacementAfter { /// A `Replacement::Summary` candidate — a genuine post-ASAP binding. - Summary(SummaryDAG), - /// A `Replacement::Rewrite` candidate — still pre-ASAP shaped (CSE + Summary(ExportDAG), + /// A `Replacement::Rewrite` candidate — still relational (CSE /// share/recompute, `AvgToSumOverCountStrategy`, and `RollupStrategy` - /// all produce this kind), so this reuses [`ExportDAG`]/[`export`] too, - /// not a new type. + /// all produce this kind). Rewrite(ExportDAG), } -/// Flatten `expr` into a [`ExportDAG`]. -pub fn export(expr: &QueryExpr) -> ExportDAG { - let mut nodes = Vec::new(); - // One cache for the whole export — persisted across every `build`/ - // `push_node` call, not reset per node, so `structural_hash` memoizes - // real work across this pass instead of re-walking an already-hashed - // shared descendant once per node that references it. - let mut cache = HashCache::new(); - // No substitution: an ordinary pre-ASAP export never splices anything - // in — see `build`'s own doc for why it always takes a `find_winner` - // callback regardless (so `export_post_asap` can share this exact - // per-variant traversal instead of duplicating it). - let root = build(expr, &mut nodes, &mut cache, &mut |_| None); - ExportDAG { - nodes, - root, - edge_annotations: Vec::new(), - } -} - -/// What a higher layer found for one specific pre-ASAP node when building a -/// merged post-ASAP DAG via [`export_post_asap`] — see that function's own -/// doc for the full design. `asap_types` has no opinion on *how* this is -/// decided (that's `asap_aware_mapping::replacement::search_workload_with` + -/// `CandidateLogicalASAPDAGs::cost_sorted`'s job, a higher layer, exactly the layering rule +/// What a higher layer found for one specific node when building a merged +/// post-ASAP dag via [`export_post_asap`] — see that function's own doc +/// for the full design. `asap_types` has no opinion on *how* this is +/// decided (that's `asap_logical_optimizer::pass1::replacement::search_workload_with` + +/// `candidate_selection::cost_sorted`'s job, a higher layer, exactly the layering rule /// [`DAGNode::notes`] already states); it only defines the shape a decision -/// comes back in. +/// comes back in. Both variants render identically (one IR, one builder); +/// they are kept apart so the caller's `Replacement` maps one-to-one. #[derive(Debug, Clone)] pub enum PostAsapSubstitution { /// This exact node has a winning `Replacement::Rewrite` — keep building - /// from `.0` instead of the original node. Still pre-ASAP shaped, so - /// [`build`] renders it via the same ordinary `DAGNode` path — see - /// [`build`]'s own doc for why `.0`'s own top level is rendered without - /// re-querying `find_winner` on it (its descendants still are). + /// from `replacement` instead of the original node. Rewrite { - replacement: Rc, + replacement: Rc, decision: DAGDecision, }, - /// This exact node has a winning `Replacement::Summary` — switch to - /// rendering `.0`'s bound `SummaryNode` shape from here down, via - /// [`build_summary_hybrid`]. + /// This exact node has a winning `Replacement::Summary` — keep building + /// from `replacement` (a summary-bound sub-DAG) instead of the original + /// node. Summary { - replacement: Rc, + replacement: Rc, decision: DAGDecision, }, } +/// Flatten the DAG rooted at `root` into a [`ExportDAG`]: one [`DAGNode`] +/// per unique reachable node, children pushed before their parents. +pub fn export(root: &Rc) -> ExportDAG { + let mut no_substitution = |_: &Rc| None; + let mut builder = Builder::new(&mut no_substitution); + let root = builder.build(root); + builder.finish(root) +} + +/// Flatten the DAG rooted at `node` into a [`SummaryDAG`] — the same +/// nodes [`export`] produces, in the [`SummaryDAGNode`] shape (snake_case +/// `kind`, `guarantee` as its own field). +pub fn export_summary(node: &Rc) -> SummaryDAG { + let dag = export(node); + let nodes = dag + .nodes + .into_iter() + .map(|node| { + let source = node + .source_node + .expect("every exported node carries its source"); + SummaryDAGNode { + id: node.id, + kind: snake_case_kind(&source.operator), + label: node.label, + detail: node.detail, + schema: node.schema, + children: node.children, + guarantee: source.guarantee.clone(), + source_node: Some(source), + } + }) + .collect(); + SummaryDAG { + nodes, + root: dag.root, + } +} + /// Build one merged "whole query, but post-ASAP" [`ExportDAG`] by walking -/// `root`'s ordinary pre-ASAP shape and, at every node, asking `find_winner` -/// whether *that exact node* has a winning replacement — if so, splicing -/// the replacement's own shape in at that position instead, in the very -/// same flattened node list (not a nested sub-DAG the way -/// [`TargetReplacement::before`]/`::after` — small, independent, per-site -/// before/after pairs — already do; see this file's "Post-ASAP replacement -/// export" section doc for why *that* design doesn't attempt a single -/// whole-query composite, and why this one can: this is a synthetic -/// id/edge list, the same kind of thing [`ExportDAG`] already is for the -/// pre-ASAP side, not a real `QueryExpr`/`SummaryNode` value with a type -/// system to satisfy). +/// `root` and, at every node, asking `find_winner` whether *that exact +/// node* has a winning replacement — if so, splicing the replacement's own +/// sub-DAG in at that position instead, in the very same flattened node +/// list (not a nested sub-dag the way [`TargetReplacement::before`]/ +/// `::after` — small, independent, per-site before/after pairs — do). /// /// `find_winner` is the whole layering seam: `asap_types` never runs -/// `asap_aware_mapping::replacement::search_workload_with` or -/// `CandidateLogicalASAPDAGs::cost_sorted` itself, and has no idea what a `TargetSubDAGCandidates` or a +/// `asap_logical_optimizer::pass1::replacement::search_workload_with` or +/// `candidate_selection::cost_sorted` itself, and has no idea what a `TargetSubDAGCandidates` or a /// `ReplacementProvenance` is — it only asks, for one node at a time, "did a /// higher layer already decide something for you?" A caller (e.g. the /// `dag_export` devtools binary) builds this closure once per workload @@ -758,7 +520,7 @@ pub enum PostAsapSubstitution { /// for [`TargetReplacement`] discovery, and passes it in here unchanged. /// /// `find_winner` is deliberately consulted only once per node, at the -/// moment [`build`] first reaches it — **not** re-consulted on a +/// moment the builder first reaches it — **not** re-consulted on a /// substitution's own immediate top level (only on that substitution's /// *descendants*, which get an ordinary fresh call same as any other node). /// This matters for correctness, not just efficiency: @@ -770,225 +532,189 @@ pub enum PostAsapSubstitution { /// re-query at exactly that one level is what makes this termination-safe /// for every registered strategy, not just the ones that happen not to /// return the target itself as a candidate. +/// +/// Every node a substitution introduced carries the substitution's +/// [`DAGDecision`] (`role = "replacement_root"` on the spliced-in root, +/// `"replacement_region"` on its newly exported descendants); a descendant +/// that was already exported before the splice (a shared input the +/// replacement reuses) keeps whatever it already had. pub fn export_post_asap( - root: &QueryExpr, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, + root: &Rc, + find_winner: &mut dyn FnMut(&Rc) -> Option, ) -> ExportDAG { - let mut nodes = Vec::new(); - let mut cache = HashCache::new(); - let root_id = build(root, &mut nodes, &mut cache, find_winner); - deduplicate_pointer_shared_nodes(nodes, root_id) + let mut builder = Builder::new(find_winner); + let root = builder.build(root); + builder.finish(root) } -fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> ExportDAG { - let mut by_source_ptr = HashMap::::new(); - let mut old_to_new = vec![0_u32; nodes.len()]; - let mut deduplicated = Vec::with_capacity(nodes.len()); - for mut node in nodes { - node.children = node - .children - .into_iter() - .map(|child| old_to_new[child as usize]) - .collect(); - if let Some(existing) = node - .source_ptr - .and_then(|source_ptr| by_source_ptr.get(&source_ptr).copied()) - { - old_to_new[node.id as usize] = existing; - continue; - } - let old_id = node.id; - let new_id = deduplicated.len() as u32; - node.id = new_id; - if let Some(source_ptr) = node.source_ptr { - by_source_ptr.insert(source_ptr, new_id); +/// The one flattening pass behind every entry point. Nodes are memoized by +/// `Rc` pointer identity: a node reached from several parents (an operator +/// input shared with a scalar reference, say) is exported once. +struct Builder<'a> { + nodes: Vec, + /// `Rc::as_ptr` of every node already exported (or substituted) → its id. + ids: HashMap<*const OperatorNode, u32>, + /// One cache for the whole export — persisted across every node, not + /// reset per node, so `structural_hash` memoizes real work across this + /// pass instead of re-walking an already-hashed shared descendant once + /// per node that references it. + cache: HashCache, + find_winner: &'a mut dyn FnMut(&Rc) -> Option, +} + +impl<'a> Builder<'a> { + fn new( + find_winner: &'a mut dyn FnMut(&Rc) -> Option, + ) -> Self { + Self { + nodes: Vec::new(), + ids: HashMap::new(), + cache: HashCache::new(), + find_winner, } - old_to_new[old_id as usize] = new_id; - deduplicated.push(node); } - ExportDAG { - nodes: deduplicated, - root: old_to_new[root as usize], - edge_annotations: Vec::new(), + fn finish(self, root: u32) -> ExportDAG { + ExportDAG { + nodes: self.nodes, + root, + edge_annotations: Vec::new(), + } } -} -macro_rules! define_query_kind_tags { - ($($pattern:pat => $tag:literal),+ $(,)?) => { - #[cfg(test)] - const QUERY_KIND_TAGS: &[&str] = &[$($tag),+]; - - fn kind_tag(expr: &QueryExpr) -> &'static str { - match expr { - $($pattern => $tag),+, - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => unreachable!( - "kind_tag reached a scalar QueryExpr variant directly: {other:?}" - ), + /// Export `node` (or, when `find_winner` has a substitution for it, the + /// substitution's sub-DAG in its place) and return its id. + fn build(&mut self, node: &Rc) -> u32 { + let ptr = Rc::as_ptr(node); + if let Some(&id) = self.ids.get(&ptr) { + return id; + } + let (replacement, decision) = match (self.find_winner)(node) { + None => return self.build_node(node), + Some(PostAsapSubstitution::Rewrite { + replacement, + decision, + }) + | Some(PostAsapSubstitution::Summary { + replacement, + decision, + }) => (replacement, decision), + }; + let first = self.nodes.len(); + let root = self.build_node(&replacement); + for exported in &mut self.nodes[first..] { + if exported.decision.is_none() { + let mut node_decision = decision.clone(); + node_decision.role = if exported.id == root { + "replacement_root" + } else { + "replacement_region" + }; + exported.decision = Some(node_decision); } } - }; -} - -define_query_kind_tags! { - QueryExpr::Scan { .. } => "Scan", - QueryExpr::PromqlScalarBridge(_) => "PromqlScalarBridge", - QueryExpr::EvalTimestamp => "EvalTimestamp", - QueryExpr::CurrentTimestamp => "CurrentTimestamp", - QueryExpr::PromqlVectorFromScalar(_) => "PromqlVectorFromScalar", - QueryExpr::PromqlScalarFromVector(_) => "PromqlScalarFromVector", - QueryExpr::PromqlRelabel { .. } => "PromqlRelabel", - QueryExpr::PromqlInfoEnrich { .. } => "PromqlInfoEnrich", - QueryExpr::PromqlSeriesSample { .. } => "PromqlSeriesSample", - QueryExpr::Filter { .. } => "Filter", - QueryExpr::Project { .. } => "Project", - QueryExpr::Aggregate { .. } => "Aggregate", - QueryExpr::Dedup { .. } => "Dedup", - QueryExpr::Concat { .. } => "Concat", - QueryExpr::Join { .. } => "Join", - QueryExpr::SetOp { .. } => "SetOp", - QueryExpr::Sort { .. } => "Sort", - QueryExpr::Limit { .. } => "Limit", - QueryExpr::PromqlSubquery { .. } => "PromqlSubquery", - QueryExpr::TimeRange { .. } => "TimeRange", - QueryExpr::TimeShift { .. } => "TimeShift", - QueryExpr::SQLWindowFunc { .. } => "SQLWindowFunc", - QueryExpr::BinaryOp { .. } => "BinaryOp", -} - -/// Push one flattened node for `expr`. `expr` is the *whole* sub-DAG this -/// node represents (not just its own fields) — `hash` is -/// [`structural_hash(expr)`](structural_hash), the identical function and -/// the identical input `InternTable::intern` would hash for this same -/// sub-DAG, so this node's `hash` matches what `cse::share_common_sub_dags` -/// would bucket it under. `kind` is [`kind_tag(expr)`](kind_tag), not a -/// caller-supplied argument — see that function's doc for why. -fn push_node( - nodes: &mut Vec, - expr: &QueryExpr, - cache: &mut HashCache, - label: String, - detail: serde_json::Value, - children: Vec, -) -> u32 { - let id = nodes.len() as u32; - let hash = Some(structural_hash(expr, cache)); - nodes.push(DAGNode { - id, - kind: kind_tag(expr), - label, - detail, - schema: expr - .output_schema() - .ok() - .and_then(|schema| serde_json::to_value(schema).ok()), - children, - workload_node_id: None, - hash, - source_expr: Some(expr.clone()), - source_ptr: Some(expr as *const QueryExpr as usize), - notes: Vec::new(), - decision: None, - }); - id -} + // The original node now resolves to the substitution: another + // parent of the same `Rc` reuses the spliced-in sub-DAG. + self.ids.insert(ptr, root); + root + } -/// Push one flattened node with no corresponding pre-ASAP `QueryExpr` at -/// all — a post-ASAP-originated node inside [`export_post_asap`]'s merged -/// DAG (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). -/// `hash`/`source_expr`-based re-identification (see [`DAGNode::hash`]'s own -/// doc) has no meaning for a node with no `QueryExpr` behind it, so this -/// pushes a fixed placeholder hash (`0`) and `source_expr: None` rather than -/// inventing a hash over `SummaryExpr` (which, unlike `QueryExpr`, has no -/// [`structural_hash`]-equivalent function at all — see [`SummaryDAGNode`]'s -/// own doc on why `SummaryExpr`'s fields don't even derive `Hash`/`PartialEq` -/// consistently enough to build one). -fn push_summary_originated_node( - nodes: &mut Vec, - kind: &'static str, - label: String, - detail: serde_json::Value, - children: Vec, -) -> u32 { - let id = nodes.len() as u32; - nodes.push(DAGNode { - id, - kind, - label, - detail, - schema: None, - children, - workload_node_id: None, - hash: None, - source_expr: None, - source_ptr: None, - notes: Vec::new(), - decision: None, - }); - id + /// Export `node` itself (no substitution check at this level; children + /// still go through [`Self::build`]) and return its id. + fn build_node(&mut self, node: &Rc) -> u32 { + let ptr = Rc::as_ptr(node); + if let Some(&id) = self.ids.get(&ptr) { + return id; + } + let children: Vec = node.children().into_iter().map(|c| self.build(c)).collect(); + let (label, mut detail) = shape(node, &self.ids); + if let serde_json::Value::Object(map) = &mut detail { + if let Some(timing) = node.timing { + map.insert("timing".into(), serde_json::json!(timing.as_str())); + } + if let Some(guarantee) = &node.guarantee { + if let Ok(value) = serde_json::to_value(guarantee) { + map.insert("guarantee".into(), value); + } + } + } + let hash = structural_hash(node, &mut self.cache); + self.cache.insert(ptr, hash); + let id = self.nodes.len() as u32; + self.nodes.push(DAGNode { + id, + kind: node.operator.kind_name(), + label, + detail, + schema: serde_json::to_value(&node.schema).ok(), + children, + workload_node_id: None, + hash: Some(hash), + source_node: Some(Rc::clone(node)), + source_ptr: Some(ptr as usize), + notes: Vec::new(), + decision: None, + }); + self.ids.insert(ptr, id); + id + } } -/// The [`build_summary`]/[`build_summary_hybrid`] counterpart of [`build`] -/// for a bound [`SummaryNode`] reached while building -/// [`export_post_asap`]'s merged DAG: appends into the *same* `nodes: -/// Vec` list `build` itself is filling, instead of a separate -/// [`SummaryDAG`]. A `KeepPreAsap(inner)` leaf recurses back into -/// [`build`] on `inner` (the general pre-ASAP entry, `find_winner` included) -/// rather than nesting a `{"pre_asap_sub_dag": ...}` blob the way -/// [`build_summary`] does — so the merged DAG reads as one seamless DAG -/// with no dead ends, and so a target reachable underneath a `KeepPreAsap` -/// wrapper (a nested aggregate a strategy independently found a -/// replacement for, say) still gets spliced in correctly. -fn build_summary_hybrid( - node: &SummaryNode, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { - if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - return build(inner, nodes, cache, find_winner); +/// [`Operator::kind_name`] in snake_case, for [`SummaryDAGNode::kind`]. +/// Exhaustive so a new operator variant fails to compile here until it is +/// named. +fn snake_case_kind(operator: &Operator) -> &'static str { + match operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Scan { .. } => "scan", + NonASAPOp::Values { .. } => "values", + NonASAPOp::Filter { .. } => "filter", + NonASAPOp::Project { .. } => "project", + NonASAPOp::Aggregate { .. } => "aggregate", + NonASAPOp::Join { .. } => "join", + NonASAPOp::SetOp { .. } => "set_op", + NonASAPOp::Concat { .. } => "concat", + NonASAPOp::Dedup { .. } => "dedup", + NonASAPOp::Sort { .. } => "sort", + NonASAPOp::Limit { .. } => "limit", + NonASAPOp::BinaryOp { .. } => "binary_op", + NonASAPOp::SQLWindowFunc { .. } => "sql_window_func", + NonASAPOp::TimeRange { .. } => "time_range", + NonASAPOp::TimeShift { .. } => "time_shift", + NonASAPOp::PromqlVectorFromScalar(_) => "promql_vector_from_scalar", + NonASAPOp::PromqlRelabel { .. } => "promql_relabel", + NonASAPOp::PromqlInfoEnrich { .. } => "promql_info_enrich", + NonASAPOp::PromqlSeriesSample { .. } => "promql_series_sample", + NonASAPOp::PromqlSubquery { .. } => "promql_subquery", + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { .. } => "summary_agg", + ASAPOp::SummaryEstimate { .. } => "summary_estimate", + ASAPOp::FinalizeExactAccumulator { .. } => "finalize_exact_accumulator", + ASAPOp::MaintainPopulation { .. } => "maintain_population", + ASAPOp::EvaluatePopulation { .. } => "read_population", + ASAPOp::SummaryMerge { .. } => "summary_merge", + ASAPOp::SummarySubtract { .. } => "summary_subtract", + ASAPOp::SummaryDelete { .. } => "summary_delete", + ASAPOp::SummaryJoin { .. } => "summary_join", + ASAPOp::Extension { .. } => "extension", + }, } - let children: Vec = summary_children(&node.expr) - .into_iter() - .map(|child| build_summary_hybrid(child, nodes, cache, find_winner)) - .collect(); - let (kind, label, mut detail) = summary_shape(&node.expr); - // The merged DAG's `DAGNode` has no dedicated guarantee field (it is - // the pre-ASAP node shape); the guarantee rides in `detail` under the - // same key/shape `SummaryDAGNode::guarantee` uses, additively. - if let Some(guarantee) = &node.guarantee { - if let (serde_json::Value::Object(map), Ok(value)) = - (&mut detail, serde_json::to_value(guarantee)) - { - map.insert("guarantee".into(), value); - } - } - let id = push_summary_originated_node(nodes, kind, label, detail, children); - nodes[id as usize].schema = Some(summary_schema_json(&node.schema)); - id } -fn summary_schema_json(schema: &crate::post_asap::Schema) -> serde_json::Value { - serde_json::json!({ - "fields": schema.fields.iter().map(|field| serde_json::json!({ - "name": field.name, - "dtype": format!("{:?}", field.dtype), - "nullable": field.nullable, - })).collect::>(), - "time_index": schema.time_index, - }) +/// A short, human-readable label for a [`FieldDataType`] (e.g. +/// `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for the label text on a +/// `SummaryAgg`/`SummaryJoin` node. Every variant is covered, via `Debug` +/// for the inner kind rather than hand-written prose per algorithm. +fn family_label(family: &FieldDataType) -> String { + match family { + FieldDataType::Plain(dtype) => format!("Plain({dtype:?})"), + FieldDataType::ExactAggregate(kind, _) => format!("ExactAggregate({kind:?})"), + FieldDataType::Sketch(kind, _grouping) => format!("Sketch({:?})", kind.algorithm()), + FieldDataType::Sample(kind, _) => format!("Sample({kind:?})"), + FieldDataType::Wavelet(kind, _) => format!("Wavelet({kind:?})"), + FieldDataType::StatModel(kind, _) => format!("StatModel({kind:?})"), + } } fn source_label(source: &Source) -> String { @@ -998,407 +724,336 @@ fn source_label(source: &Source) -> String { } } -/// Recursively flatten `expr`, appending nodes to `nodes` in post-order -/// (children pushed before their parent), and return the id of the pushed -/// root node. Exhaustive over every **operator** `QueryExpr` variant — a new -/// one fails to compile here until this match is extended, matching the rest -/// of the IR's exhaustive-match style (e.g. `output_schema`). The scalar -/// variants (issue #205) are never passed to `build` directly: every operator -/// arm that carries one (`Filter.pred`, `Project.cols`, `Aggregate.having`, …) -/// serializes it as opaque `detail` JSON via `Predicate`/`ProjectItem`/ -/// `AggIntent`'s own `Serialize` impl, same as before the merge — a scalar -/// sub-DAG was never a separate DAG node, so this doesn't change that. -/// -/// `find_winner` is [`export_post_asap`]'s substitution seam, threaded -/// through every recursive call (including [`export`]'s own, which always -/// passes a closure that returns `None`) so both entry points share this -/// exact traversal instead of maintaining two copies of it. `build` itself -/// only ever calls `find_winner` once, right here at the top, before -/// dispatching into the ordinary per-variant match below — see -/// [`export_post_asap`]'s own doc for why a substitution's own immediate -/// result is rendered via that match directly (recursing into its children -/// through `build` again, so *they* still get a fresh `find_winner` call) -/// rather than by looping back through this check a second time. -fn build( - expr: &QueryExpr, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { - match find_winner(expr) { - Some(PostAsapSubstitution::Rewrite { - replacement, - decision, - }) => { - let first = nodes.len(); - let root = build_no_recheck(&replacement, nodes, cache, find_winner); - for node in &mut nodes[first..] { - if node.decision.is_none() { - let mut node_decision = decision.clone(); - node_decision.role = if node.id == root { - "replacement_root" - } else { - "replacement_region" - }; - node.decision = Some(node_decision); - } +/// `(label, detail)` for one node: its own fields, never its children. +/// Exhaustive over every operator variant — a new one fails to compile +/// here until this match is extended, matching the rest of the IR's +/// exhaustive-match style. Scalar expressions are rendered through +/// [`scalar_json`] with `ids` resolving their operator references. +fn shape( + node: &OperatorNode, + ids: &HashMap<*const OperatorNode, u32>, +) -> (String, serde_json::Value) { + let scalar = |expr: &ScalarExpr| scalar_json(expr, ids); + let scalars = + |exprs: &[ScalarExpr]| -> Vec { exprs.iter().map(scalar).collect() }; + let predicate = |pred: &crate::ir::Predicate| scalar(&pred.0); + let sort_keys = |keys: &[crate::ir::SortKey]| -> Vec { + keys.iter() + .map(|key| { + serde_json::json!({ + "expr": scalar(&key.expr), + "ascending": key.ascending, + "nulls_first": key.nulls_first, + }) + }) + .collect() + }; + match &node.operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Scan { + source, + predicates, + schema, + } => ( + format!("Scan({})", source_label(source)), + serde_json::json!({ + "source": source, + "predicates": predicates.iter().map(predicate).collect::>(), + "schema": schema, + }), + ), + NonASAPOp::Values { rows, schema } => ( + format!("Values({} rows)", rows.len()), + serde_json::json!({ + "rows": rows.iter().map(|row| scalars(row)).collect::>(), + "schema": schema, + }), + ), + NonASAPOp::Filter { pred, .. } => ( + "Filter".into(), + serde_json::json!({ "pred": predicate(pred) }), + ), + NonASAPOp::Project { + cols, qualifier, .. + } => ( + format!("Project({} cols)", cols.len()), + serde_json::json!({ + "cols": cols.iter().map(|item| serde_json::json!({ + "alias": item.alias, + "expr": scalar(&item.expr), + })).collect::>(), + "qualifier": qualifier, + }), + ), + NonASAPOp::Aggregate { + reduction, + measures, + output_names, + having, + .. + } => ( + format!("Aggregate({} measures)", measures.len()), + serde_json::json!({ + "reduction": reduction, + "measures": measures, + "output_names": output_names, + "having": having.as_ref().map(predicate), + }), + ), + NonASAPOp::Join { kind, pred, .. } => ( + format!("Join({kind:?})"), + serde_json::json!({ "kind": kind, "pred": predicate(pred) }), + ), + NonASAPOp::SetOp { kind, all, .. } => ( + format!("SetOp({kind:?})"), + serde_json::json!({ "kind": kind, "all": all }), + ), + NonASAPOp::Concat { + children, + discriminator_unique_key, + } => ( + format!("Concat({} branches)", children.len()), + serde_json::json!({ "discriminator_unique_key": discriminator_unique_key }), + ), + NonASAPOp::Dedup { cols, .. } => ( + format!("Dedup({} cols)", cols.len()), + serde_json::json!({ "cols": cols }), + ), + NonASAPOp::Sort { + keys, partition_by, .. + } => ( + format!("Sort({} keys)", keys.len()), + serde_json::json!({ "keys": sort_keys(keys), "partition_by": partition_by }), + ), + NonASAPOp::Limit { + n, + offset, + partition_by, + .. + } => ( + match n { + Some(n) => format!("Limit({n})"), + None => format!("Limit(offset {offset})"), + }, + serde_json::json!({ "n": n, "offset": offset, "partition_by": partition_by }), + ), + NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } => ( + format!("BinaryOp({})", operator.kind), + serde_json::json!({ + "op": operator.kind.to_string(), + "vector_match": operator.vector_match, + "checked_relative_division": operator.checked_relative_division, + "checked_finite_division": operator.checked_finite_division, + "return_bool": return_bool, + }), + ), + NonASAPOp::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + .. + } => ( + format!("SQLWindowFunc({func:?})"), + serde_json::json!({ + "func": func, + "args": scalars(args), + "partition_by": partition_by, + "order_by": sort_keys(order_by), + "frame": frame, + "output_name": output_name, + }), + ), + NonASAPOp::TimeRange { range, kind, .. } => ( + format!("TimeRange({kind:?}, {range:?})"), + serde_json::json!({ "range": range, "kind": kind }), + ), + NonASAPOp::TimeShift { shift, .. } => { + ("TimeShift".into(), serde_json::json!({ "shift": shift })) } - return root; - } - Some(PostAsapSubstitution::Summary { - replacement, - decision, - }) => { - let first = nodes.len(); - let root = build_summary_hybrid(&replacement, nodes, cache, find_winner); - for node in &mut nodes[first..] { - if node.decision.is_none() { - let mut node_decision = decision.clone(); - node_decision.role = if node.id == root { - "replacement_root" - } else { - "replacement_region" - }; - node.decision = Some(node_decision); - } + NonASAPOp::PromqlVectorFromScalar(value) => ( + "vector()".into(), + serde_json::json!({ "value": scalar(value) }), + ), + NonASAPOp::PromqlRelabel { dst, value, .. } => ( + format!("PromqlRelabel(dst={dst})"), + serde_json::json!({ "dst": dst, "value": scalar(value) }), + ), + NonASAPOp::PromqlInfoEnrich { selector, .. } => ( + "PromqlInfoEnrich".into(), + serde_json::json!({ "selector": selector }), + ), + NonASAPOp::PromqlSeriesSample { by, kind, .. } => ( + format!("PromqlSeriesSample({kind:?})"), + serde_json::json!({ "by": by, "kind": kind }), + ), + NonASAPOp::PromqlSubquery { + range, resolution, .. + } => ( + "PromqlSubquery".into(), + serde_json::json!({ "range": range, "resolution": resolution }), + ), + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { + family, + input, + reduction, + grouping, + .. + } => ( + format!("SummaryAgg({})", family_label(family)), + serde_json::json!({ + "family": format!("{family:?}"), + "input": input, + "reduction": reduction, + "grouping": format!("{grouping:?}"), + }), + ), + ASAPOp::SummaryEstimate { query, .. } => ( + format!("SummaryEstimate({query:?})"), + serde_json::json!({ "query": format!("{query:?}") }), + ), + ASAPOp::FinalizeExactAccumulator { .. } => { + ("FinalizeExactAccumulator".into(), serde_json::json!({})) } - return root; - } - None => {} + ASAPOp::MaintainPopulation { population, .. } => ( + format!("MaintainPopulation(max_k={})", population.max_k), + serde_json::json!({ "population": population }), + ), + ASAPOp::EvaluatePopulation { evaluation, .. } => ( + format!("EvaluatePopulation({evaluation:?})"), + serde_json::json!({ "evaluation": evaluation }), + ), + ASAPOp::SummaryMerge { children } => ( + format!("SummaryMerge({} children)", children.len()), + serde_json::json!({}), + ), + ASAPOp::SummarySubtract { .. } => ("SummarySubtract".into(), serde_json::json!({})), + ASAPOp::SummaryDelete { key, .. } => { + ("SummaryDelete".into(), serde_json::json!({ "key": key })) + } + ASAPOp::SummaryJoin { key, family, .. } => ( + format!("SummaryJoin({})", family_label(family)), + serde_json::json!({ "key": key, "family": format!("{family:?}") }), + ), + ASAPOp::Extension { name, .. } => ( + format!("Extension({name})"), + serde_json::json!({ "name": name }), + ), + }, } - build_no_recheck(expr, nodes, cache, find_winner) } -/// The actual per-variant match [`build`] dispatches to once it has decided -/// (by consulting `find_winner` exactly once) which `QueryExpr` value to -/// render at this position — either `expr` itself (unchanged), or a winning -/// `Replacement::Rewrite`'s own target. Every recursive call here goes back -/// through [`build`] (not this function), so every child gets its own fresh -/// `find_winner` query. -fn build_no_recheck( - expr: &QueryExpr, - nodes: &mut Vec, - cache: &mut HashCache, - find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> u32 { +/// `{"scalar_ref": }` for an operator node a scalar expression reads. +/// The node is one of the owning operator's children, so it has already +/// been exported by the time its parent's `detail` is built. +fn scalar_ref( + node: &Rc, + ids: &HashMap<*const OperatorNode, u32>, +) -> serde_json::Value { + serde_json::json!({ "scalar_ref": ids.get(&Rc::as_ptr(node)).copied() }) +} + +/// `expr` as JSON in `ScalarExpr`'s own serde shape (externally tagged +/// variants), except that every operator reference is rendered via +/// [`scalar_ref`] instead of inlining the referenced sub-DAG. Exhaustive so +/// a new variant fails to compile here until it is rendered. +fn scalar_json(expr: &ScalarExpr, ids: &HashMap<*const OperatorNode, u32>) -> serde_json::Value { + let sub = |e: &ScalarExpr| scalar_json(e, ids); + let list = |es: &[ScalarExpr]| -> Vec { es.iter().map(sub).collect() }; match expr { - QueryExpr::Scan { - source, - predicates, - schema, - } => { - let label = format!("Scan({})", source_label(source)); - let detail = serde_json::json!({ - "source": source, - "predicates": predicates, - "schema": schema, - }); - push_node(nodes, expr, cache, label, detail, vec![]) - } - // The bridged child is a scalar-sub-language node (issue #220), not - // an operator node `build` can recurse into — serialize it as opaque - // `detail` JSON, same as every other scalar-typed field - // (`Filter.pred`, `Project.cols`, …) rather than pushing it as a - // separate DAG node. - QueryExpr::PromqlScalarBridge(inner) => { - let detail = serde_json::json!({ "value": inner }); - push_node( - nodes, - expr, - cache, - format!("PromqlScalarBridge({inner:?})"), - detail, - vec![], - ) - } - QueryExpr::EvalTimestamp => push_node( - nodes, - expr, - cache, - "EvalTimestamp".into(), - serde_json::json!({}), - vec![], - ), - QueryExpr::CurrentTimestamp => push_node( - nodes, - expr, - cache, - "CurrentTimestamp".into(), - serde_json::json!({}), - vec![], - ), - QueryExpr::PromqlVectorFromScalar(child) => { - let c = build(child, nodes, cache, find_winner); - push_node( - nodes, - expr, - cache, - "vector()".into(), - serde_json::json!({}), - vec![c], - ) - } - QueryExpr::PromqlScalarFromVector(child) => { - let c = build(child, nodes, cache, find_winner); - push_node( - nodes, - expr, - cache, - "scalar()".into(), - serde_json::json!({}), - vec![c], - ) - } - QueryExpr::PromqlRelabel { dst, value, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "dst": dst, "value": value }); - push_node( - nodes, - expr, - cache, - format!("PromqlRelabel(dst={dst})"), - detail, - vec![c], - ) - } - QueryExpr::PromqlInfoEnrich { selector, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "selector": selector }); - push_node( - nodes, - expr, - cache, - "PromqlInfoEnrich".into(), - detail, - vec![c], - ) - } - QueryExpr::PromqlSeriesSample { by, kind, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "by": by, "kind": kind }); - push_node( - nodes, - expr, - cache, - format!("PromqlSeriesSample({kind:?})"), - detail, - vec![c], - ) - } - QueryExpr::Filter { pred, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "pred": pred }); - push_node(nodes, expr, cache, "Filter".into(), detail, vec![c]) - } - QueryExpr::Project { - cols, - qualifier, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "cols": cols, "qualifier": qualifier }); - push_node( - nodes, - expr, - cache, - format!("Project({} cols)", cols.len()), - detail, - vec![c], - ) - } - QueryExpr::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ - "reduction": reduction, - "measures": measures, - "output_names": output_names, - "filters": filters, - "having": having, - }); - push_node( - nodes, - expr, - cache, - format!("Aggregate({} measures)", measures.len()), - detail, - vec![c], - ) - } - QueryExpr::Dedup { cols, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "cols": cols }); - push_node( - nodes, - expr, - cache, - format!("Dedup({} cols)", cols.len()), - detail, - vec![c], - ) - } - QueryExpr::Concat { - children, - discriminator_unique_key, - } => { - let ids: Vec = children - .iter() - .map(|c| build(c, nodes, cache, find_winner)) - .collect(); - let label = format!("Concat({} branches)", ids.len()); - let detail = - serde_json::json!({ "discriminator_unique_key": discriminator_unique_key }); - push_node(nodes, expr, cache, label, detail, ids) - } - QueryExpr::Join { - kind, - pred, + ScalarExpr::Column(id) => serde_json::json!({ "Column": id }), + ScalarExpr::Literal(value) => serde_json::json!({ "Literal": value }), + ScalarExpr::Negative { expr, semantics } => serde_json::json!({ + "Negative": { "expr": sub(expr), "semantics": semantics } + }), + ScalarExpr::Compare { left, + op, right, - } => { - let l = build(left, nodes, cache, find_winner); - let r = build(right, nodes, cache, find_winner); - let detail = serde_json::json!({ "kind": kind, "pred": pred }); - push_node( - nodes, - expr, - cache, - format!("Join({kind:?})"), - detail, - vec![l, r], - ) - } - QueryExpr::SetOp { - kind, - all, + semantics, + } => serde_json::json!({ + "Compare": { + "left": sub(left), + "op": op, + "right": sub(right), + "semantics": semantics, + } + }), + ScalarExpr::BoolAnd(parts) => serde_json::json!({ "BoolAnd": list(parts) }), + ScalarExpr::BoolOr(parts) => serde_json::json!({ "BoolOr": list(parts) }), + ScalarExpr::Not(e) => serde_json::json!({ "Not": sub(e) }), + ScalarExpr::IsNull(e) => serde_json::json!({ "IsNull": sub(e) }), + ScalarExpr::IsNotNull(e) => serde_json::json!({ "IsNotNull": sub(e) }), + ScalarExpr::Cast { expr, to, try_cast } => serde_json::json!({ + "Cast": { "expr": sub(expr), "to": to, "try_cast": try_cast } + }), + ScalarExpr::InList { + expr, + list: items, + negated, + } => serde_json::json!({ + "InList": { "expr": sub(expr), "list": list(items), "negated": negated } + }), + ScalarExpr::FunctionCall { name, args } => serde_json::json!({ + "FunctionCall": { "name": name, "args": list(args) } + }), + ScalarExpr::Arithmetic { + op, left, right, - } => { - let l = build(left, nodes, cache, find_winner); - let r = build(right, nodes, cache, find_winner); - let detail = serde_json::json!({ "kind": kind, "all": all }); - push_node( - nodes, - expr, - cache, - format!("SetOp({kind:?})"), - detail, - vec![l, r], - ) - } - QueryExpr::Sort { - keys, - partition_by, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "keys": keys, "partition_by": partition_by }); - push_node( - nodes, - expr, - cache, - format!("Sort({} keys)", keys.len()), - detail, - vec![c], - ) - } - QueryExpr::Limit { n, offset, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "n": n, "offset": offset }); - push_node(nodes, expr, cache, format!("Limit({n})"), detail, vec![c]) - } - QueryExpr::PromqlSubquery { - range, - resolution, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "range": range, "resolution": resolution }); - push_node(nodes, expr, cache, "PromqlSubquery".into(), detail, vec![c]) - } - QueryExpr::TimeRange { range, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "range": range }); - push_node( - nodes, - expr, - cache, - format!("TimeRange({range:?})"), - detail, - vec![c], - ) - } - QueryExpr::TimeShift { shift, child } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ "shift": shift }); - push_node(nodes, expr, cache, "TimeShift".into(), detail, vec![c]) - } - QueryExpr::SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => { - let c = build(child, nodes, cache, find_winner); - let detail = serde_json::json!({ - "func": func, - "args": args, - "partition_by": partition_by, - "order_by": order_by, - "frame": frame, - "output_name": output_name, - }); - push_node( - nodes, - expr, - cache, - format!("SQLWindowFunc({func:?})"), - detail, - vec![c], - ) - } - QueryExpr::BinaryOp { - op, - lhs, - rhs, - vector_match, - } => { - let l = build(lhs, nodes, cache, find_winner); - let r = build(rhs, nodes, cache, find_winner); - let detail = serde_json::json!({ "op": op.to_string(), "vector_match": vector_match }); - push_node( - nodes, - expr, - cache, - format!("BinaryOp({op})"), - detail, - vec![l, r], - ) - } - other @ (QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. }) => { - unreachable!("dag_export::build reached a scalar QueryExpr variant directly: {other:?}") - } + semantics, + } => serde_json::json!({ + "Arithmetic": { + "op": op, + "left": sub(left), + "right": sub(right), + "semantics": semantics, + } + }), + ScalarExpr::Case { + operand, + branches, + else_expr, + } => serde_json::json!({ + "Case": { + "operand": operand.as_deref().map(sub), + "branches": branches + .iter() + .map(|(when, then)| serde_json::json!([sub(when), sub(then)])) + .collect::>(), + "else_expr": else_expr.as_deref().map(sub), + } + }), + ScalarExpr::CurrentTimestamp => serde_json::json!("CurrentTimestamp"), + ScalarExpr::EvalTimestamp => serde_json::json!("EvalTimestamp"), + ScalarExpr::PromqlScalarFromVector(node) => serde_json::json!({ + "PromqlScalarFromVector": scalar_ref(node, ids) + }), + ScalarExpr::ScalarSubquery(node) => serde_json::json!({ + "ScalarSubquery": scalar_ref(node, ids) + }), + ScalarExpr::Exists { subquery, negated } => serde_json::json!({ + "Exists": { "subquery": scalar_ref(subquery, ids), "negated": negated } + }), + ScalarExpr::InSubquery { + expr, + subquery, + negated, + } => serde_json::json!({ + "InSubquery": { + "expr": sub(expr), + "subquery": scalar_ref(subquery, ids), + "negated": negated, + } + }), } } @@ -1407,14 +1062,22 @@ mod tests { use std::rc::Rc; use super::*; - use crate::pre_asap::agg_intent::AggIntent; - use crate::pre_asap::expr_ir::ScalarValue; - use crate::pre_asap::query_expr::{GroupKeys, Predicate, Reduction}; - use crate::pre_asap::schema::{DataType, Field, Schema}; + use crate::ir::operator::agg_intent::AggIntent; + use crate::ir::operator::operator_properties::{GroupKeys, JoinKind, Reduction}; + use crate::ir::properties::{ + BoundExpr, CompositionOperator, ErrorMetric, GuaranteeSource, ProbabilityExpr, + }; + use crate::ir::scalar::{ColumnRef, ScalarValue}; + use crate::ir::schema::{DataType, Field, Schema}; + use crate::ir::schema::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, SummaryUpdate, + }; + use crate::ir::Predicate; + use crate::types::AccuracyTarget; - fn scan(table: &str, columns: Vec) -> QueryExpr { - QueryExpr::Scan { + fn scan(table: &str, columns: Vec) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { table_ref: table.into(), }, @@ -1425,25 +1088,95 @@ mod tests { unique_keys: vec![], closed: true, }, - } + })) + .unwrap() } fn value_col() -> Vec { vec![Field::plain("value", DataType::Float64, false)] } + fn true_pred() -> Predicate { + Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))) + } + + fn count_agg(child: Rc) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(GroupKeys::none()), + measures: vec![AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child, + })) + .unwrap() + } + + fn join(left: Rc, right: Rc) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Join { + kind: JoinKind::Inner, + pred: true_pred(), + left, + right, + })) + .unwrap() + } + + /// A KLL `SummaryAgg` over `leaf`'s `v` column, read out as a quantile. + fn quantile_evaluation( + leaf: Rc, + guarantee: Option, + ) -> (Rc, Rc) { + let family = FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 40 }), + GroupingStrategy::default(), + ); + let agg = std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child: leaf, + family: family.clone(), + input: SummaryUpdate::column(ColumnRef::Named("v".into())), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }), + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(None), + ); + let evaluation = std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: Rc::clone(&agg), + query: SketchStatistic::Quantile { q: 0.99 }, + }), + Schema::lifted( + vec![Field::plain("quantile", DataType::Float64, false)], + None, + ), + ) + .with_guarantee(guarantee), + ); + (agg, evaluation) + } + #[test] fn leaf_scan_is_a_single_node() { let dag = export(&scan("metrics", value_col())); assert_eq!(dag.nodes.len(), 1); assert_eq!(dag.root, 0); assert_eq!(dag.nodes[0].kind, "Scan"); + assert_eq!(dag.nodes[0].label, "Scan(metrics)"); assert!(dag.nodes[0].children.is_empty()); + assert!(dag.nodes[0].source_node.is_some()); } /// `export` itself never populates higher-layer annotations. Empty - /// annotations must not appear in serialized JSON, so ordinary (non-ASAP) - /// exports retain their existing shape. + /// annotations must not appear in serialized JSON, so ordinary exports + /// retain their existing shape. #[test] fn export_omits_empty_higher_layer_annotations() { let dag = export(&scan("metrics", value_col())); @@ -1460,6 +1193,10 @@ mod tests { !json.contains("decision"), "empty `decision` must be skipped, not serialized as `null`: {json}" ); + assert!( + !json.contains("source_node"), + "`source_node` is in-process only: {json}" + ); let dag_json = serde_json::to_string(&dag).unwrap(); assert!( !dag_json.contains("edge_annotations"), @@ -1473,22 +1210,29 @@ mod tests { // single child slot) share the exact same `Rc` Scan — // `export_post_asap` must merge them onto one node id. Sharing alone // is not physical cost evidence, so no edge cost may be fabricated. - let shared_scan = Rc::new(scan("metrics", value_col())); - let left_branch = QueryExpr::Dedup { - cols: vec![0], - child: Rc::clone(&shared_scan), - }; - let right_branch = QueryExpr::Limit { - n: 5, - offset: 0, - child: Rc::clone(&shared_scan), - }; - let root = QueryExpr::Concat { + let shared_scan = scan("metrics", value_col()); + let left_branch = + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Dedup { + cols: vec![0], + child: Rc::clone(&shared_scan), + })) + .unwrap(); + let right_branch = + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(5), + offset: 0, + partition_by: GroupKeys::none(), + child: Rc::clone(&shared_scan), + })) + .unwrap(); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Concat { children: vec![left_branch, right_branch], discriminator_unique_key: None, - }; + })) + .unwrap(); let dag = export_post_asap(&root, &mut |_| None); + assert_eq!(dag.nodes.len(), 4, "Scan, Dedup, Limit, Concat"); assert_eq!( dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1, @@ -1497,20 +1241,15 @@ mod tests { assert!(dag.edge_annotations.is_empty()); } - /// Regression test: a single parent referencing the same shared child - /// from two of its own operand slots at once (a `Join` whose left and - /// right sides are the exact same `Rc`, post pointer-dedup) is *one* - /// downstream consumer, not two — this must not inflate - /// produce an edge-cost annotation without explicit physical evidence. + /// A single parent referencing the same shared child from two of its + /// own operand slots at once (a `Join` whose left and right sides are + /// the exact same `Rc`) is *one* downstream consumer, not two — this + /// must not produce an edge-cost annotation without explicit physical + /// evidence. #[test] fn a_single_parent_referencing_a_shared_child_twice_is_one_consumer_not_two() { - let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { - kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::clone(&shared_scan), - right: Rc::clone(&shared_scan), - }; + let shared_scan = scan("metrics", value_col()); + let root = join(Rc::clone(&shared_scan), Rc::clone(&shared_scan)); let dag = export_post_asap(&root, &mut |_| None); assert_eq!( @@ -1518,6 +1257,7 @@ mod tests { 1, "the shared Scan must be merged onto one node, not duplicated" ); + assert_eq!(dag.nodes[dag.root as usize].children, vec![0, 0]); assert!( dag.edge_annotations.is_empty(), "a single parent referencing the same child twice is one consumer, not a genuine \ @@ -1527,38 +1267,33 @@ mod tests { } #[test] - fn export_never_produces_edge_annotations_since_it_never_shares_nodes() { - // Plain `export` (no `export_post_asap`) never deduplicates by `Rc` - // pointer identity — even a workload-level shared sub-DAG renders as - // two independent DAG nodes here, so there is nothing to annotate. - let shared_scan = Rc::new(scan("metrics", value_col())); - let root = QueryExpr::Join { - kind: crate::pre_asap::query_expr::JoinKind::Inner, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::clone(&shared_scan), - right: Rc::clone(&shared_scan), - }; - let dag = export(&root); - assert_eq!(dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 2); + fn export_merges_pointer_shared_nodes_but_not_equal_copies() { + // Plain `export` deduplicates by `Rc` pointer identity: the same + // `Rc` reached twice is one node ... + let shared_scan = scan("metrics", value_col()); + let dag = export(&join(Rc::clone(&shared_scan), Rc::clone(&shared_scan))); + assert_eq!(dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1); assert!(dag.edge_annotations.is_empty()); + + // ... while two structurally equal but distinct `Rc`s stay two + // nodes (with equal hashes — that is CSE's job, not the export's). + let dag = export(&join( + scan("metrics", value_col()), + scan("metrics", value_col()), + )); + let scans: Vec<_> = dag.nodes.iter().filter(|n| n.kind == "Scan").collect(); + assert_eq!(scans.len(), 2); + assert_eq!(scans[0].hash, scans[1].hash); } #[test] fn chain_preserves_shape_and_child_links() { - let expr = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::none()), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan("metrics", value_col())), - }), - }; - let dag = export(&expr); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: true_pred(), + child: count_agg(scan("metrics", value_col())), + })) + .unwrap(); + let dag = export(&root); assert_eq!(dag.nodes.len(), 3, "Filter -> Aggregate -> Scan"); let filter = &dag.nodes[dag.root as usize]; @@ -1567,6 +1302,7 @@ mod tests { let agg = &dag.nodes[filter.children[0] as usize]; assert_eq!(agg.kind, "Aggregate"); + assert_eq!(agg.label, "Aggregate(1 measures)"); assert_eq!(agg.children.len(), 1); let leaf = &dag.nodes[agg.children[0] as usize]; @@ -1576,18 +1312,76 @@ mod tests { #[test] fn merge_keeps_every_branch_as_a_child() { - let expr = QueryExpr::concat(vec![ - scan("a", value_col()), - scan("b", value_col()), - scan("c", value_col()), - ]); - let dag = export(&expr); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Concat { + children: vec![ + scan("a", value_col()), + scan("b", value_col()), + scan("c", value_col()), + ], + discriminator_unique_key: None, + })) + .unwrap(); + let dag = export(&root); assert_eq!(dag.nodes.len(), 4, "3 branches + the Concat node"); let merge = &dag.nodes[dag.root as usize]; assert_eq!(merge.kind, "Concat"); assert_eq!(merge.children.len(), 3); } + /// An operator node read from a scalar expression is a child of the + /// owning operator (after its operator inputs), and the expression's + /// `detail` points at it by id instead of inlining it. + #[test] + fn scalar_operator_references_are_children_rendered_as_scalar_refs() { + let subquery = scan("other", value_col()); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Exists { + subquery: Rc::clone(&subquery), + negated: false, + }), + child: scan("metrics", value_col()), + })) + .unwrap(); + let dag = export(&root); + assert_eq!(dag.nodes.len(), 3); + let filter = &dag.nodes[dag.root as usize]; + assert_eq!( + filter.children.len(), + 2, + "operator input, then the scalar reference" + ); + let input = &dag.nodes[filter.children[0] as usize]; + let referenced = &dag.nodes[filter.children[1] as usize]; + assert_eq!(input.label, "Scan(metrics)"); + assert_eq!(referenced.label, "Scan(other)"); + assert_eq!( + filter.detail["pred"]["Exists"]["subquery"]["scalar_ref"], + serde_json::json!(referenced.id) + ); + assert_eq!(filter.detail["pred"]["Exists"]["negated"], false); + let json = serde_json::to_string(&filter.detail).unwrap(); + assert!( + !json.contains("other"), + "the referenced sub_dag must not be inlined into detail: {json}" + ); + } + + #[test] + fn limit_without_n_is_offset_only() { + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Limit { + n: None, + offset: 3, + partition_by: GroupKeys::none(), + child: scan("metrics", value_col()), + })) + .unwrap(); + let dag = export(&root); + let limit = &dag.nodes[dag.root as usize]; + assert_eq!(limit.label, "Limit(offset 3)"); + assert_eq!(limit.detail["n"], serde_json::Value::Null); + assert_eq!(limit.detail["offset"], 3); + } + #[test] fn identical_sub_dags_hash_equal_and_differing_ones_dont() { let left = scan("metrics", value_col()); @@ -1615,17 +1409,18 @@ mod tests { // Two roots that each wrap the *same* Scan shape in a different outer // node — the exported hash should still flag the shared Scan even // though it's embedded at different depths / under different parents. - let shared_shape = || scan("metrics", value_col()); - - let q1 = QueryExpr::Limit { - n: 10, + let q1 = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Limit { + n: Some(10), offset: 0, - child: Rc::new(shared_shape()), - }; - let q2 = QueryExpr::Dedup { + partition_by: GroupKeys::none(), + child: scan("metrics", value_col()), + })) + .unwrap(); + let q2 = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Dedup { cols: vec![0], - child: Rc::new(shared_shape()), - }; + child: scan("metrics", value_col()), + })) + .unwrap(); let g1 = export(&q1); let g2 = export(&q2); @@ -1647,9 +1442,9 @@ mod tests { fn root_hash_matches_cse_structural_hash_for_the_same_node() { // Not just "hashes equal for equal inputs" (any two consistent hash // functions would do that) — the exported root's `hash` must be the - // literal `u64` `crate::pre_asap::cse::structural_hash` produces for - // this exact node, because it's the same function call, not a - // parallel reimplementation that happens to agree. + // literal `u64` `crate::ir::cse::structural_hash` produces for this + // exact node, because it's the same function call, not a parallel + // reimplementation that happens to agree. let leaf = scan("metrics", value_col()); let dag = export(&leaf); assert_eq!( @@ -1661,24 +1456,15 @@ mod tests { #[test] fn every_node_hash_matches_cse_structural_hash_on_its_own_sub_dag() { - // A multi-level DAG: check the parity holds at every depth, not - // just the root — each `DAGNode::hash` must equal - // `structural_hash` applied to the actual `QueryExpr` sub-DAG that - // node represents. - let agg = QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::none()), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan("metrics", value_col())), - }; - let root = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - child: Rc::new(agg.clone()), - }; + // A multi-level tree: check the parity holds at every depth, not + // just the root — each `DAGNode::hash` must equal `structural_hash` + // applied to the actual node it represents. + let agg = count_agg(scan("metrics", value_col())); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: true_pred(), + child: Rc::clone(&agg), + })) + .unwrap(); let dag = export(&root); assert_eq!( @@ -1697,39 +1483,23 @@ mod tests { ); } - /// Issue #172: a readout's guarantee is exported structurally — metric, + /// Issue #172: a evaluation's guarantee is exported structurally — metric, /// symbolic bound, failure probability, provenance (allocation - /// included) — and a rejection carries its typed reason. + /// included) — and a rejection carries its typed reason. A relational + /// node below a summary is its own node, in the same dag. #[test] fn export_carries_guarantee_allocation_and_rejection_reason() { - use crate::post_asap::{ - BoundExpr, CompositionOperator, ErrorMetric, FieldDataType, GroupingStrategy, - GuaranteeSource, ProbabilityExpr, Schema, SketchAlgorithm, SketchKind, SketchParams, - SketchStatistic, - }; - let leaf = Rc::new(scan("t", vec![Field::plain("v", DataType::Float64, false)])); - let kept = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::clone(&leaf)), - schema: Schema::lifted(vec![], None), - guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), - }); - let agg = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: kept, - family: FieldDataType::Sketch( - SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 40 }), - GroupingStrategy::default(), - ), - input: crate::post_asap::SummaryUpdate::column( - crate::pre_asap::expr_ir::ColumnRef::Named("v".into()), - ), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![], None), - guarantee: None, - }); + let leaf = Rc::new( + OperatorNode::new(Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Schema::lifted(vec![Field::plain("v", DataType::Float64, false)], None), + })) + .unwrap() + .with_guarantee(Some(ResultGuarantee::exact("Scan"))), + ); let guarantee = ResultGuarantee { metric: ErrorMetric::Rank, bound: BoundExpr::Sum { @@ -1755,15 +1525,15 @@ mod tests { }, ], }; - let root = SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: agg, - query: SketchStatistic::Quantile { q: 0.99 }, - }, - schema: Schema::lifted(vec![], None), - guarantee: Some(guarantee), - }; + let (_, root) = quantile_evaluation(Rc::clone(&leaf), Some(guarantee)); let dag = export_summary(&root); + assert_eq!( + dag.nodes.iter().map(|n| n.kind).collect::>(), + ["scan", "summary_agg", "summary_estimate"] + ); + assert_eq!(dag.nodes[1].label, "SummaryAgg(Sketch(Kll))"); + assert!(dag.nodes[2].label.starts_with("SummaryEstimate(Quantile")); + assert!(dag.nodes.iter().all(|n| n.schema.is_some())); let json = serde_json::to_value(&dag).unwrap(); let root_json = &json["nodes"][dag.root as usize]; assert_eq!(root_json["guarantee"]["metric"], "rank"); @@ -1779,10 +1549,18 @@ mod tests { assert!(provenance.iter().any(|s| s["kind"] == "composition_step")); // Raw sketch state carries none; the exact leaf carries zero error. let state = &json["nodes"][1]; - assert_eq!(state["kind"], "SummaryAgg"); assert!(state.get("guarantee").is_none()); assert_eq!(json["nodes"][0]["guarantee"]["bound"]["op"], "zero"); + // The `DAGNode` shape carries the same guarantee inside `detail`. + let dag = export(&root); + assert_eq!( + dag.nodes.iter().map(|n| n.kind).collect::>(), + ["Scan", "SummaryAgg", "SummaryEstimate"] + ); + assert_eq!(dag.nodes[2].detail["guarantee"]["metric"], "rank"); + assert!(dag.nodes[1].detail.get("guarantee").is_none()); + let named = NamedDAG { name: "q".into(), source: None, @@ -1792,7 +1570,7 @@ mod tests { workload_cost: None, rejections: vec![TargetRejection { target_pre_id: 0, - strategy: "SketchAlgorithmStrategy".into(), + strategy: "ASAPStrategies".into(), description: "quantile over quantile".into(), error: AccuracyError::UnsupportedComposition { operator: CompositionOperator::ApproximateAggregate, @@ -1819,38 +1597,166 @@ mod tests { .is_none()); } - fn viewer_kind_categories() -> std::collections::BTreeMap { - const START: &str = "const KIND_CATEGORY_JSON = `"; - let source = include_str!(concat!( - env!("CARGO_MANIFEST_DIR"), - "/../../tools/dag-viewer/node-style.js" - )); - let json = source - .split_once(START) - .expect("node-style.js must declare KIND_CATEGORY_JSON") - .1 - .split_once("`;") - .expect("KIND_CATEGORY_JSON must be a template literal") - .0; - serde_json::from_str(json).expect("KIND_CATEGORY_JSON must be valid JSON") + /// `export_post_asap` splices a winning summary in place of its target, + /// tags every node the splice introduced with the decision, and leaves + /// the rest of the query — including an input the summary reuses that + /// was already exported — untagged and shared. + #[test] + fn export_post_asap_splices_a_summary_substitution_in_place() { + let leaf = scan("t", vec![Field::plain("v", DataType::Float64, false)]); + let target = count_agg(Rc::clone(&leaf)); + // `leaf` is exported through the Join's left side before the target + // (its right side) is reached and substituted. + let root = join(Rc::clone(&leaf), Rc::clone(&target)); + let (_, evaluation) = quantile_evaluation(Rc::clone(&leaf), None); + let decision = DAGDecision { + id: 7, + strategy: "Sketch".into(), + rationale: "quantile via KLL".into(), + rank: 0, + cost: 1.0, + role: "", + baseline_cost: None, + selected_cost: None, + benefit: None, + }; + let mut calls = Vec::new(); + let dag = export_post_asap(&root, &mut |node| { + calls.push(node.operator.kind_name()); + Rc::ptr_eq(node, &target).then(|| PostAsapSubstitution::Summary { + replacement: Rc::clone(&evaluation), + decision: decision.clone(), + }) + }); + + let kinds: Vec<_> = dag.nodes.iter().map(|n| n.kind).collect(); + assert_eq!(kinds, ["Scan", "SummaryAgg", "SummaryEstimate", "Join"]); + assert!(!kinds.contains(&"Aggregate"), "the target itself is gone"); + let join_node = &dag.nodes[dag.root as usize]; + assert_eq!(join_node.children, vec![0, 2]); + assert!(join_node.decision.is_none()); + let estimate = &dag.nodes[2]; + assert_eq!(estimate.kind, "SummaryEstimate"); + assert_eq!( + estimate.decision.as_ref().map(|d| (d.id, d.role)), + Some((7, "replacement_root")) + ); + let agg = &dag.nodes[estimate.children[0] as usize]; + assert_eq!( + agg.decision.as_ref().map(|d| (d.id, d.role)), + Some((7, "replacement_region")) + ); + let scan_node = &dag.nodes[agg.children[0] as usize]; + assert_eq!( + scan_node.id, 0, + "the summary reuses the already-exported input" + ); + assert!( + scan_node.decision.is_none(), + "a node exported before the splice is not tagged by it" + ); + assert_eq!( + calls, + ["Join", "Scan", "Aggregate", "SummaryAgg"], + "the substitution's own top level (SummaryEstimate) is never re-queried; its \ + descendants are, except the input already exported" + ); } + /// A `SharedSubDAGStrategy`-shaped substitution returns the target + /// itself as its replacement; the walk must still terminate and render + /// the target once. #[test] - fn viewer_categorizes_exactly_the_exported_node_kinds() { - let expected: std::collections::BTreeSet<_> = QUERY_KIND_TAGS - .iter() - .chain(SUMMARY_KIND_TAGS) - .copied() - .chain(std::iter::once("KeepPreAsap")) - .collect(); + fn export_post_asap_terminates_when_the_replacement_is_the_target() { + let target = count_agg(scan("t", value_col())); + let decision = DAGDecision { + id: 1, + strategy: "SharedSubDAG".into(), + rationale: "share".into(), + rank: 0, + cost: f64::NAN, + role: "", + baseline_cost: None, + selected_cost: None, + benefit: None, + }; + let dag = export_post_asap(&target, &mut |node| { + Rc::ptr_eq(node, &target).then(|| PostAsapSubstitution::Rewrite { + replacement: Rc::clone(&target), + decision: decision.clone(), + }) + }); + assert_eq!(dag.nodes.len(), 2); + assert_eq!(dag.nodes[dag.root as usize].kind, "Aggregate"); assert_eq!( - expected.len(), - QUERY_KIND_TAGS.len() + SUMMARY_KIND_TAGS.len() + 1, - "exported kind tags must be unique" + dag.nodes[dag.root as usize] + .decision + .as_ref() + .map(|d| d.role), + Some("replacement_root") ); - let categories = viewer_kind_categories(); - let actual: std::collections::BTreeSet<_> = categories.keys().map(String::as_str).collect(); + } - assert_eq!(actual, expected); + /// The snake_case `kind` table is exactly `kind_name` re-cased, for + /// every variant: a `SummaryDAGNode` and a `DAGNode` for the same node + /// never disagree on what it is. + #[test] + fn summary_kind_is_the_operator_kind_name_in_snake_case() { + fn to_snake(name: &str) -> String { + let mut out = String::new(); + let chars: Vec = name.chars().collect(); + for (i, &c) in chars.iter().enumerate() { + if c.is_ascii_uppercase() { + let prev_lower = i > 0 && !chars[i - 1].is_ascii_uppercase(); + let next_lower = chars.get(i + 1).is_some_and(|n| n.is_ascii_lowercase()); + if i > 0 && (prev_lower || next_lower) { + out.push('_'); + } + out.push(c.to_ascii_lowercase()); + } else { + out.push(c); + } + } + out + } + let leaf = scan("t", vec![Field::plain("v", DataType::Float64, false)]); + let (_, evaluation) = quantile_evaluation(Rc::clone(&leaf), None); + let finalize = std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child: evaluation }), + Schema::lifted(vec![], None), + ) + .with_guarantee(None), + ); + let root = + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::SQLWindowFunc { + func: crate::ir::operator::operator_properties::WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::none(), + order_by: vec![], + frame: None, + output_name: "rn".into(), + child: finalize, + })) + .unwrap(); + let dag = export(&root); + let summary = export_summary(&root); + assert_eq!(dag.nodes.len(), summary.nodes.len()); + for (a, b) in dag.nodes.iter().zip(&summary.nodes) { + assert_eq!(a.id, b.id); + assert_eq!(b.kind, to_snake(a.kind), "{}", a.kind); + assert_eq!(a.children, b.children); + assert_eq!(a.label, b.label); + } + assert_eq!( + summary.nodes.iter().map(|n| n.kind).collect::>(), + [ + "scan", + "summary_agg", + "summary_estimate", + "finalize_exact_accumulator", + "sql_window_func", + ] + ); } } diff --git a/crates/types/src/ir/canonicalize.rs b/crates/types/src/ir/canonicalize.rs new file mode 100644 index 000000000..daf1d45e1 --- /dev/null +++ b/crates/types/src/ir/canonicalize.rs @@ -0,0 +1,1077 @@ +//! Post-lowering canonicalization of the operator DAG. +//! +//! Erases *structural* differences between semantically identical queries so +//! a post-ASAP binding rule matching on the intent algebra sees one canonical +//! spelling regardless of source language (issue #34). +//! +//! ## Heavy-hitter promotion +//! +//! An additive-ranked "order by the aggregate, take the top k" is a +//! heavy-hitter represented by [`AggIntent::TopK`]. Front ends may emit it as +//! an ordinary `Limit { Sort { … Aggregate } }`; this pass promotes that shape +//! to the canonical +//! +//! ```text +//! Aggregate { reduction: Reduce(), measures: [TopK{k}], +//! child: Aggregate { measures: [Count | Sum], … } } +//! ``` +//! +//! Count supplies unit weights and Sum supplies value weights. Because the +//! match is positional, aliases do not affect it. Other ranked expressions +//! retain Sort + Limit. +//! +//! ## Subquery lowering +//! +//! EXISTS/NOT EXISTS and positive IN filter conjuncts may use semi/anti joins. +//! Scalar subqueries remain explicit: a cross join does not preserve their +//! zero-row NULL or multiple-row error semantics. All scalar plan references +//! participate in DAG traversal and canonicalization. + +use std::collections::HashMap; +use std::rc::Rc; + +use crate::ir::operator::agg_intent::{topk, AggIntent}; +use crate::ir::operator::node::{Operator, OperatorNode}; +use crate::ir::operator::non_asap::NonASAPOp; +use crate::ir::operator::operator_properties::{JoinKind, Reduction}; +use crate::ir::scalar::{CompareOpKind, ScalarValue}; +use crate::ir::scalar::{ExprSemantics, Predicate, ProjectItem, ScalarExpr, SortKey}; +use crate::ir::SchemaDerivationError; +use crate::types::AccuracyTarget; + +/// Rewrite the DAG under `root` into its canonical form (bottom-up). +/// Idempotent: an already-canonical DAG comes back as the same `Rc`. Only +/// nodes that change (or whose inputs change) are rebuilt; every untouched +/// sub-DAG keeps its pointer identity, and a shared sub-DAG that is rewritten +/// stays shared. +pub fn canonicalize(root: Rc) -> Result, SchemaDerivationError> { + canon(&root, &mut HashMap::new()) +} + +fn canon( + node: &Rc, + memo: &mut HashMap<*const OperatorNode, Rc>, +) -> Result, SchemaDerivationError> { + if let Some(done) = memo.get(&Rc::as_ptr(node)) { + return Ok(Rc::clone(done)); + } + + // A `Concat` asserting a caller-proven `discriminator_unique_key` (issue + // #228) had that key's `ColumnId`s resolved against exactly the first + // branch's output schema *as it stood before this pass ran*. The rewrites + // below can restructure that branch (anywhere within it) into a shape + // with a different output schema, which would leave those `ColumnId`s + // pointing at the wrong column, or out of bounds. Snapshot the schema the + // key was resolved against before recursing into the children. + let discriminator_branch_schema_before = match &node.operator { + Operator::NonASAP(NonASAPOp::Concat { + children, + discriminator_unique_key: Some(_), + }) => children.first().map(|c| c.schema.clone()), + _ => None, + }; + + // Bottom-up: canonicalize every operator input before matching at this + // node, so an inner heavy-hitter is promoted before an enclosing rewrite + // inspects it. + let mut rebuilt: Vec<(*const OperatorNode, Rc)> = Vec::new(); + let mut changed = false; + for child in operator_children(&node.operator) { + let new = canon(child, memo)?; + changed |= !Rc::ptr_eq(&new, child); + rebuilt.push((Rc::as_ptr(child), new)); + } + let mut current = if changed { + // `map_children` also visits the operator nodes referenced from + // scalar expressions; those are not in `rebuilt` and pass through + // unchanged. (A node that is both an operator input and a scalar + // reference is one shared node, so it takes its canonical form in + // both places.) + let rebuilt_child = |c: &Rc| { + rebuilt + .iter() + .find(|(ptr, _)| *ptr == Rc::as_ptr(c)) + .map_or_else(|| Rc::clone(c), |(_, new)| Rc::clone(new)) + }; + Rc::new(node.map_children(rebuilt_child)?) + } else { + Rc::clone(node) + }; + + // If the first branch's output schema moved out from under the asserted + // key, the key can no longer be trusted — drop it (never re-derive it by + // guessing at name/position). A wrong `unique_keys` claim is a wrong + // query answer, not a missed optimization, so any difference at all + // drops the key. + if let Operator::NonASAP(NonASAPOp::Concat { + children, + discriminator_unique_key: Some(_), + }) = ¤t.operator + { + let after = children.first().map(|c| &c.schema); + if discriminator_branch_schema_before.as_ref() != after { + current = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Concat { + children: children.clone(), + discriminator_unique_key: None, + }))?; + } + } + + let current = apply_local_rules(current, memo)?; + + memo.insert(Rc::as_ptr(node), Rc::clone(¤t)); + Ok(current) +} + +type Memo = HashMap<*const OperatorNode, Rc>; + +/// Apply the local rewrite rules at `node` (whose inputs are already +/// canonical) until none matches. The rules chain: a `ROW_NUMBER()`- +/// partitioned top-k rewrites to a `Limit{Sort}`, which the heavy-hitter +/// rule may then promote to an `Aggregate([TopK])`; a `Filter` with several +/// subquery conjuncts sheds one per round. Each rule strictly simplifies the +/// node (one fewer idiom, or one fewer subquery reference), so the loop +/// terminates. +fn apply_local_rules( + mut current: Rc, + memo: &mut Memo, +) -> Result, SchemaDerivationError> { + loop { + let next = if let Some(next) = try_promote_additive_top_ranking(¤t)? { + next + } else if let Some(next) = try_lower_subquery_conjunct(¤t, memo)? { + next + } else { + break; + }; + current = next; + } + Ok(current) +} + +/// The direct **operator** inputs of a node — the relational skeleton only. +/// Operator nodes referenced from a scalar position (`ScalarSubquery`, +/// `Exists`, …) are not visited here: a subquery that the lowering rules +/// lift into a join is canonicalized at that point, and one they leave in +/// place (`NOT IN`, an `EXISTS` outside a `Filter` conjunct) stays as the +/// front end emitted it. +fn operator_children(op: &Operator) -> Vec<&Rc> { + op.children() +} + +/// Recognise an additive-ranked +/// `Limit { Sort { [Project] Aggregate([Count | Sum]) } }` and rewrite it to +/// the canonical heavy-hitter `Aggregate([TopK])` over the explicit inner +/// aggregate. Returns `None` when the shape does not match. +fn try_promote_additive_top_ranking( + node: &OperatorNode, +) -> Result>, SchemaDerivationError> { + // Limit k, no offset (an OFFSET means "not the top k"). + let Some(NonASAPOp::Limit { + n: Some(k), + offset: 0, + partition_by: limit_partition, + child, + }) = node.non_asap() + else { + return Ok(None); + }; + // A single ordering key on a column. + let Some(NonASAPOp::Sort { + keys, + partition_by, + child: sort_child, + }) = child.non_asap() + else { + return Ok(None); + }; + // A per-group `Limit` must agree with its `Sort`'s partition: the + // ranking's partition is what the outer `TopK` groups by. + if !limit_partition.is_empty() && limit_partition != partition_by { + return Ok(None); + } + let [SortKey { + expr: ScalarExpr::Column(sort_col), + ascending, + .. + }] = keys.as_slice() + else { + return Ok(None); + }; + + // The ordered relation is an `Aggregate`, optionally behind a passthrough + // projection (a bare-column SELECT list). Map the sort key through the + // projection to the aggregate's own output column. + let (agg_node, ranked_col) = match sort_child.non_asap() { + Some(NonASAPOp::Project { cols, child, .. }) => { + let Some(ProjectItem { + expr: ScalarExpr::Column(underlying), + .. + }) = cols.get(*sort_col) + else { + return Ok(None); + }; + (child, *underlying) + } + _ => (sort_child, *sort_col), + }; + + // Exactly one aggregate, ranked by *its* output column — the measure sits + // at index `by.len()` (after the group keys). A `PerEntity` reduction has + // no `by` to rank a measure against, so it is a non-match. + let Some(NonASAPOp::Aggregate { + reduction, + measures, + child: aggregate_child, + .. + }) = agg_node.non_asap() + else { + return Ok(None); + }; + let Reduction::Reduce(by) = reduction else { + return Ok(None); + }; + let [ranked_agg] = measures.as_slice() else { + return Ok(None); + }; + if ranked_col != by.len() { + return Ok(None); + } + // The heavy-hitter decision — descending, over a measure with a realised + // heavy-hitter sketch — is the shared rule both front ends consult (issue + // #38). An ascending additive-ranked limit (bottom-k) stays generic. + if !topk::Ranking::from_aggregate(ranked_agg).is_supported(!ascending) { + return Ok(None); + } + // A direct Sum is a stream of additive observation weights. A Sum over a + // derived child such as Rate/Increase still needs exact reset-aware + // values to rerank sketch candidates, and the post-ASAP IR has no + // candidate-sidecar + exact-rerank node, so that shape keeps Sort + Limit. + if matches!(ranked_agg, AggIntent::Sum { .. }) + && matches!( + aggregate_child.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + ) + { + return Ok(None); + } + // Count ranks unit updates; a direct Sum ranks weighted updates. + let accuracy = match ranked_agg { + AggIntent::Count { accuracy } => accuracy.clone(), + AggIntent::Sum { .. } => AccuracyTarget::Exact, + _ => unreachable!("additive ranking gate admitted a non-additive measure"), + }; + + // Outer heavy-hitter `TopK`, grouped by the ranking's partition (empty for + // a global `ORDER BY … LIMIT k`; the `by` labels for a partitioned `topk + // by`), over the unchanged inner additive aggregate. + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(partition_by.to_vec()), + measures: vec![AggIntent::TopK { k: *k, accuracy }], + output_names: Vec::new(), + filters: vec![], + having: None, + child: Rc::clone(agg_node), + })) + .map(Some) +} + +// ROW_NUMBER filters retain the window output. Eliminating it without a +// consumer-aware rewrite drops a visible column and invalidates outer scopes. + +/// `Predicate(true)`: the unconditional join predicate the SQL front end +/// emits for an uncorrelated `EXISTS` and for a `CROSS JOIN`. +fn always_true() -> Predicate { + Predicate(ScalarExpr::Literal(ScalarValue::Boolean(true))) +} + +/// Whether `conjunct` is one this pass lowers to a semi/anti join. +fn is_join_conjunct(conjunct: &ScalarExpr) -> bool { + match conjunct { + ScalarExpr::Exists { .. } => true, + // An `IN` whose probe expression itself reads a scalar subquery is + // lowered only after that subquery has been joined in by + // `try_lower_scalar_subquery` (a `Join` predicate is not a place that + // rule looks). `NOT IN` is never lowered — see the module docs. + ScalarExpr::InSubquery { + expr, + negated: false, + .. + } => find_scalar_subquery(expr).is_none(), + _ => false, + } +} + +/// Lower one `[NOT] EXISTS (s)` / `x IN (s)` conjunct of a `Filter` to the +/// semi-/anti-join the SQL front end used to emit directly. The remaining +/// conjuncts stay in an outer `Filter` over the join: a semi/anti join's +/// output schema is the left's, so their column ids are unchanged. One +/// conjunct per call; the fixpoint loop picks up the next. +fn try_lower_subquery_conjunct( + node: &OperatorNode, + memo: &mut Memo, +) -> Result>, SchemaDerivationError> { + let Some(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) = node.non_asap() + else { + return Ok(None); + }; + let conjuncts = pred.conjuncts(); + let Some(idx) = conjuncts.iter().position(is_join_conjunct) else { + return Ok(None); + }; + let left_width = child.schema.fields.len(); + let (kind, subquery, join_pred) = match &conjuncts[idx] { + // Uncorrelated by construction (the IR's `Exists` carries no outer + // column references), so the join condition is unconditionally true. + ScalarExpr::Exists { subquery, negated } => { + let kind = if *negated { + JoinKind::Anti + } else { + JoinKind::Semi + }; + (kind, subquery, always_true()) + } + // `x = `, which sits right after the + // left's columns in the `left ++ right` scope the predicate resolves + // against. + ScalarExpr::InSubquery { expr, subquery, .. } => ( + JoinKind::Semi, + subquery, + Predicate(ScalarExpr::Compare { + left: expr.clone(), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(left_width)), + semantics: ExprSemantics::Sql, + }), + ), + _ => unreachable!("`is_join_conjunct` admitted a non-subquery conjunct"), + }; + let join = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Join { + kind, + pred: join_pred, + left: Rc::clone(child), + right: canon(subquery, memo)?, + }))?; + let mut rest: Vec = conjuncts + .iter() + .enumerate() + .filter(|(i, _)| *i != idx) + .map(|(_, c)| c.clone()) + .collect(); + let out = match rest.len() { + 0 => join, + 1 => filter(rest.remove(0), join)?, + _ => filter(ScalarExpr::BoolAnd(rest), join)?, + }; + Ok(Some(out)) +} + +/// Lower one scalar subquery read by a `Project` item or a `Filter` +/// predicate: the owner reads it through a cross join against the subquery, +/// whose single column is appended after the left's (`Column(|left|)`), and +/// every occurrence of that subquery node in the owner is replaced by that +/// column reference. One subquery node per call; the fixpoint loop handles +/// the rest, each getting its own cross join further out (so earlier column +/// ids are never shifted). For a `Filter` the output schema is restored to +/// the left's columns by a positional `Project` over the result. +/// +/// Not representable in the IR, and therefore not checked here: SQL raises +/// an error when a scalar subquery yields more than one row (the cross join +/// would duplicate the left's rows instead), and yields NULL when it yields +/// none (the cross join yields no rows instead). +fn filter( + pred: ScalarExpr, + child: Rc, +) -> Result, SchemaDerivationError> { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Filter { + pred: Predicate(pred), + child, + })) +} + +/// The first `ScalarSubquery` node read by `expr` (pre-order over its scalar +/// children; referenced operator subgraphs are their own scope and are not +/// entered). +fn find_scalar_subquery(expr: &ScalarExpr) -> Option<&Rc> { + if let ScalarExpr::ScalarSubquery(node) = expr { + return Some(node); + } + expr.children().into_iter().find_map(find_scalar_subquery) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::operator::operator_properties::WindowFuncKind; + use crate::ir::operator::operator_properties::{ + ConcatDiscriminatorKey, GroupKeys, Source, WindowFrame, WindowFrameBound, + WindowFrameOffset, WindowFrameUnits, + }; + use crate::ir::schema::{DataType, Field, Schema}; + + fn node(op: NonASAPOp) -> Rc { + Rc::new(OperatorNode::new(Operator::NonASAP(op)).expect("fixture derives a schema")) + } + + fn scan() -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + ], + 0, + vec![], + ), + }) + } + + fn aggregate( + reduction: Reduction, + agg: AggIntent, + child: Rc, + ) -> Rc { + node(NonASAPOp::Aggregate { + reduction, + measures: vec![agg], + output_names: vec![], + filters: vec![], + having: None, + child, + }) + } + + fn count() -> AggIntent { + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + } + } + + /// `Aggregate{ by: [1], [Count] }` over the scan — output cols `[service, count]`. + fn count_by_service() -> Rc { + aggregate(Reduction::by(vec![1]), count(), scan()) + } + + fn key(col: usize, ascending: bool) -> Vec { + vec![SortKey { + expr: ScalarExpr::Column(col), + ascending, + nulls_first: false, + }] + } + + fn desc(col: usize) -> Vec { + key(col, false) + } + + fn limit(n: usize, offset: usize, child: Rc) -> Rc { + node(NonASAPOp::Limit { + n: Some(n), + offset, + partition_by: GroupKeys::none(), + child, + }) + } + + fn sort(keys: Vec, child: Rc) -> Rc { + node(NonASAPOp::Sort { + keys, + partition_by: GroupKeys::none(), + child, + }) + } + + fn passthrough_project(child: Rc) -> Rc { + node(NonASAPOp::Project { + cols: vec![ + ProjectItem { + alias: None, + expr: ScalarExpr::Column(0), + }, + ProjectItem { + alias: Some("c".into()), + expr: ScalarExpr::Column(1), + }, + ], + qualifier: None, + child, + }) + } + + fn concat( + children: Vec>, + key: Option, + ) -> Rc { + node(NonASAPOp::Concat { + children, + discriminator_unique_key: key, + }) + } + + fn measures(n: &OperatorNode) -> &[AggIntent] { + match n.non_asap() { + Some(NonASAPOp::Aggregate { measures, .. }) => measures, + _ => &[], + } + } + + fn is_topk_over_count(n: &OperatorNode) -> bool { + let Some(NonASAPOp::Aggregate { + measures, child, .. + }) = n.non_asap() + else { + return false; + }; + matches!(measures.as_slice(), [AggIntent::TopK { k: 5, .. }]) + && matches!(self::measures(child), [AggIntent::Count { .. }]) + } + + #[test] + fn promotes_count_ranked_limit_sort() { + // Limit 5 { Sort DESC by count-col (1) { Aggregate[Count] by [1] } }. + let q = limit(5, 0, sort(desc(1), count_by_service())); + assert!(is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn promotes_through_a_passthrough_projection() { + // …with a `SELECT service, count` projection between the Sort and the Agg. + let q = limit(5, 0, sort(desc(1), passthrough_project(count_by_service()))); + assert!(is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn promoted_topk_reuses_the_inner_aggregate_node() { + // The inner aggregate is untouched, so the rewrite shares it rather + // than copying it. + let agg = count_by_service(); + let out = canonicalize(limit(5, 0, sort(desc(1), Rc::clone(&agg)))).unwrap(); + let Some(NonASAPOp::Aggregate { child, .. }) = out.non_asap() else { + panic!("expected TopK aggregate"); + }; + assert!(Rc::ptr_eq(child, &agg)); + } + + #[test] + fn is_idempotent() { + let q = limit(5, 0, sort(desc(1), count_by_service())); + let once = canonicalize(q).unwrap(); + let twice = canonicalize(Rc::clone(&once)).unwrap(); + assert!(Rc::ptr_eq(&once, &twice), "canonicalize must be idempotent"); + } + + #[test] + fn untouched_dag_is_returned_pointer_equal() { + // Nothing here matches a rewrite: a Concat of two projections over + // one shared aggregate. The root (and everything under it) must come + // back as the same `Rc`. + let agg = count_by_service(); + let q = concat( + vec![ + passthrough_project(Rc::clone(&agg)), + passthrough_project(Rc::clone(&agg)), + ], + None, + ); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!(Rc::ptr_eq(&out, &q)); + } + + #[test] + fn rewritten_shared_subtree_stays_shared() { + // One promotable sub-DAG referenced twice is rewritten once. + let branch = limit(5, 0, sort(desc(1), count_by_service())); + let q = concat(vec![Rc::clone(&branch), Rc::clone(&branch)], None); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Concat { children, .. }) = out.non_asap() else { + panic!("expected Concat"); + }; + assert!(is_topk_over_count(&children[0])); + assert!(Rc::ptr_eq(&children[0], &children[1])); + } + + // ── Concat's discriminator_unique_key vs. canonicalize (issue #228) ── + // + // `discriminator_unique_key`'s `ColumnId`s were resolved against the + // first branch's *pre-canonicalize* output schema. The key is dropped + // whenever that branch's schema actually changed, and survives untouched + // otherwise. Never guessed at. + + fn discriminator_key(n: &OperatorNode) -> &Option { + match n.non_asap() { + Some(NonASAPOp::Concat { + discriminator_unique_key, + .. + }) => discriminator_unique_key, + _ => panic!("expected Concat"), + } + } + + #[test] + fn concat_discriminator_key_survives_canonicalize_when_first_branch_is_unaffected() { + // A plain `Aggregate` first branch matches neither rewrite trigger, + // so its schema is identical before and after canonicalize. + let q = concat( + vec![count_by_service(), count_by_service()], + Some(ConcatDiscriminatorKey::new(0, vec![1])), + ); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!( + discriminator_key(&out).is_some(), + "an untouched first branch's discriminator key must survive canonicalize" + ); + assert!(Rc::ptr_eq(&out, &q)); + } + + #[test] + fn concat_discriminator_key_is_dropped_when_first_branch_gets_rewritten() { + // The first branch is exactly the heavy-hitter promotion trigger, so + // canonicalize rewrites it to `Aggregate{TopK}`, whose own output is + // a single column, not the original two (`[service, count]`). A key + // resolved against the 2-column shape must not survive pointing at + // the new 1-column schema. + let promotable_branch = limit(5, 0, sort(desc(1), count_by_service())); + let q = concat( + vec![promotable_branch, count_by_service()], + Some(ConcatDiscriminatorKey::new(0, vec![1])), + ); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Concat { + children, + discriminator_unique_key, + }) = out.non_asap() + else { + panic!("expected Concat"); + }; + assert!( + is_topk_over_count(&children[0]), + "the first branch is still promoted normally" + ); + assert!( + discriminator_unique_key.is_none(), + "a stale discriminator key must be dropped, never silently kept wrong" + ); + assert!( + out.schema.unique_keys.is_empty(), + "the dropped key leaves the schema" + ); + } + + #[test] + fn does_not_promote_ascending_sort() { + // Ascending = bottom-k: the Top-K ranking rule rejects it (needs + // descending), so it stays a generic Sort+Limit (issue #38). + let q = limit(5, 0, sort(key(1, true), count_by_service())); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn does_not_promote_with_offset() { + let q = limit(5, 2, sort(desc(1), count_by_service())); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn does_not_promote_ranking_by_a_group_key() { + // DESC by col 0 (the `service` group key), not the count → not a + // frequency heavy-hitter. + let q = limit(5, 0, sort(desc(0), count_by_service())); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn does_not_promote_when_limit_partition_disagrees_with_sort() { + // A per-group Limit partitioned differently from its Sort is not the + // top-k shape. + let q = node(NonASAPOp::Limit { + n: Some(5), + offset: 0, + partition_by: GroupKeys::by(vec![0]), + child: sort(desc(1), count_by_service()), + }); + assert!(!is_topk_over_count(&canonicalize(q).unwrap())); + } + + #[test] + fn promotes_sum_ranked_limit_sort_as_weighted_heavy_hitter() { + let sum = aggregate(Reduction::by(vec![1]), AggIntent::Sum { col: None }, scan()); + let out = canonicalize(limit(5, 0, sort(desc(1), sum))).unwrap(); + let Some(NonASAPOp::Aggregate { + measures, child, .. + }) = out.non_asap() + else { + panic!("expected weighted TopK aggregate"); + }; + assert!(matches!( + measures.as_slice(), + [AggIntent::TopK { k: 5, .. }] + )); + assert!(matches!(self::measures(child), [AggIntent::Sum { .. }])); + } + + #[test] + fn keeps_sum_over_counter_reduction_as_exact_value_ranking() { + for counter in [AggIntent::Rate, AggIntent::Increase] { + let derived = aggregate(Reduction::PerEntity, counter, scan()); + let sum = aggregate( + Reduction::by(vec![1]), + AggIntent::Sum { col: None }, + derived, + ); + let out = canonicalize(limit(5, 0, sort(desc(1), sum))).unwrap(); + let Some(NonASAPOp::Limit { child, .. }) = out.non_asap() else { + panic!("expected Limit, got {out:?}"); + }; + let Some(NonASAPOp::Sort { child, .. }) = child.non_asap() else { + panic!("expected Sort under the Limit"); + }; + let Some(NonASAPOp::Aggregate { + measures, child, .. + }) = child.non_asap() + else { + panic!("expected Aggregate under the Sort"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!( + child.non_asap(), + Some(NonASAPOp::Aggregate { .. }) + )); + } + } + + // ── ROW_NUMBER() partitioned top-k (issue #24) ────────────────────────── + + /// A scan with `[ts, service, region, value]`. + fn scan4() -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + ], + 0, + vec![], + ), + }) + } + + /// `Aggregate{ by: [1,2] (service, region), [agg] }` — output `[service, + /// region, ]` (3 cols), so a ROW_NUMBER over it appends `rn` at index 3. + fn grouped(agg: AggIntent) -> Rc { + aggregate(Reduction::by(vec![1, 2]), agg, scan4()) + } + + /// `ROW_NUMBER` ignores its frame clause; any concrete frame works. + fn rownumber_frame() -> WindowFrame { + WindowFrame { + units: WindowFrameUnits::Rows, + start_bound: WindowFrameBound::Preceding(WindowFrameOffset::Scalar(ScalarValue::Null)), + end_bound: WindowFrameBound::Following(WindowFrameOffset::Scalar(ScalarValue::Null)), + } + } + + /// `SQLWindowFunc{ RowNumber, PARTITION BY region(2), ORDER BY col(2) DESC } { agg }`. + fn rownumber_window(agg: Rc) -> Rc { + node(NonASAPOp::SQLWindowFunc { + func: WindowFuncKind::RowNumber, + args: vec![], + partition_by: GroupKeys::by(vec![2]), // region + order_by: vec![SortKey { + expr: ScalarExpr::Column(2), // the aggregate output column + ascending: false, + nulls_first: true, + }], + frame: Some(rownumber_frame()), + output_name: "rn".into(), + child: agg, + }) + } + + /// `Filter{ col <= 5 } { child }`. + fn filter_le_5(col: usize, child: Rc) -> Rc { + node(NonASAPOp::Filter { + pred: Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(col)), + op: CompareOpKind::Le, + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(5))), + semantics: ExprSemantics::Sql, + }), + child, + }) + } + + /// `Filter{ rn(3) <= 5 } { ROW_NUMBER window { agg } }`. + fn rownumber_topk(agg: Rc) -> Rc { + filter_le_5(3, rownumber_window(agg)) + } + + #[test] + fn rownumber_count_topk_becomes_a_partitioned_heavy_hitter() { + let original = rownumber_topk(grouped(count())); + let out = canonicalize(Rc::clone(&original)).unwrap(); + assert_eq!(out.schema, original.schema); + assert!( + matches!(out.non_asap(),Some(NonASAPOp::Filter { child,.. }) if matches!(child.non_asap(),Some(NonASAPOp::SQLWindowFunc { .. }))) + ); + assert_idempotent(&out); + } + + #[test] + fn rownumber_avg_topk_becomes_a_partitioned_sort_limit() { + let original = rownumber_topk(grouped(AggIntent::Avg { col: None })); + let out = canonicalize(Rc::clone(&original)).unwrap(); + assert_eq!(out.schema, original.schema); + assert!( + matches!(out.non_asap(),Some(NonASAPOp::Filter { child,.. }) if matches!(child.non_asap(),Some(NonASAPOp::SQLWindowFunc { .. }))) + ); + assert_idempotent(&out); + } + + #[test] + fn filter_on_a_non_rownumber_column_is_left_alone() { + // `WHERE service_len <= 5` (col 0, not the rn window column) must not + // be mistaken for a top-k. + let q = filter_le_5(0, rownumber_window(grouped(count()))); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!(Rc::ptr_eq(&out, &q), "left as the same Filter"); + } + + // ── Subquery lowering ─────────────────────────────────────────────────── + + fn filter_of(pred: ScalarExpr, child: Rc) -> Rc { + node(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) + } + + /// `SELECT service FROM scan` — a one-column subquery. + /// Scalar reads retain cardinality/null semantics and a shared producer. + #[test] + fn scalar_subqueries_remain_explicit_and_shared() { + let sub = one_column_subquery(); + let root = node(NonASAPOp::Project { + child: scan(), + qualifier: None, + cols: vec![ + ProjectItem { + alias: Some("a".into()), + expr: ScalarExpr::ScalarSubquery(Rc::clone(&sub)), + }, + ProjectItem { + alias: Some("b".into()), + expr: ScalarExpr::ScalarSubquery(Rc::clone(&sub)), + }, + ], + }); + let out = canonicalize(root).unwrap(); + let NonASAPOp::Project { cols, child, .. } = out.expect_non_asap() else { + panic!() + }; + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); + for col in cols { + assert!(matches!(&col.expr,ScalarExpr::ScalarSubquery(node) if Rc::ptr_eq(node,&sub))); + } + assert!(out.schema.fields.iter().all(|f| f.nullable)); + assert_idempotent(&out); + } + + fn one_column_subquery() -> Rc { + node(NonASAPOp::Scan { + source: Source::Table { + table_ref: "sub".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + }) + } + + fn exists(subquery: Rc, negated: bool) -> ScalarExpr { + ScalarExpr::Exists { subquery, negated } + } + + fn in_subquery(expr: ScalarExpr, subquery: Rc, negated: bool) -> ScalarExpr { + ScalarExpr::InSubquery { + expr: Box::new(expr), + subquery, + negated, + } + } + + /// `value(2) > 1`. + fn value_gt_1() -> ScalarExpr { + ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(2)), + op: CompareOpKind::Gt, + right: Box::new(ScalarExpr::Literal(ScalarValue::Int64(1))), + semantics: ExprSemantics::Sql, + } + } + + fn literal_true() -> ScalarExpr { + ScalarExpr::Literal(ScalarValue::Boolean(true)) + } + + fn join_parts( + n: &OperatorNode, + ) -> (JoinKind, &ScalarExpr, &Rc, &Rc) { + match n.non_asap() { + Some(NonASAPOp::Join { + kind, + pred: Predicate(pred), + left, + right, + }) => (kind.clone(), pred, left, right), + _ => panic!("expected a Join, got {n:?}"), + } + } + + fn assert_idempotent(once: &Rc) { + let twice = canonicalize(Rc::clone(once)).unwrap(); + assert!(Rc::ptr_eq(once, &twice), "canonicalize must be idempotent"); + } + + #[test] + fn exists_filter_becomes_semi_join() { + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of(exists(Rc::clone(&sub), false), Rc::clone(&left)); + let out = canonicalize(q).unwrap(); + let (kind, pred, l, r) = join_parts(&out); + assert_eq!(kind, JoinKind::Semi); + assert_eq!(*pred, literal_true()); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_eq!( + out.schema.fields, left.schema.fields, + "a semi join outputs the left's columns" + ); + assert_idempotent(&out); + } + + #[test] + fn not_exists_becomes_anti_join() { + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of(exists(Rc::clone(&sub), true), Rc::clone(&left)); + let out = canonicalize(q).unwrap(); + let (kind, pred, l, r) = join_parts(&out); + assert_eq!(kind, JoinKind::Anti); + assert_eq!(*pred, literal_true()); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_idempotent(&out); + } + + #[test] + fn in_subquery_becomes_semi_join_on_the_subquery_column() { + // `WHERE service IN (SELECT service …)` over a 3-column left: the + // subquery's column is `Column(3)` in the `left ++ right` scope. + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of( + in_subquery(ScalarExpr::Column(1), Rc::clone(&sub), false), + Rc::clone(&left), + ); + let out = canonicalize(q).unwrap(); + let (kind, pred, l, r) = join_parts(&out); + assert_eq!(kind, JoinKind::Semi); + assert_eq!( + *pred, + ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(1)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Column(3)), + semantics: ExprSemantics::Sql, + } + ); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_eq!( + out.schema.fields, left.schema.fields, + "a semi join outputs the left's columns" + ); + assert_idempotent(&out); + } + + #[test] + fn exists_with_other_conjuncts_keeps_an_outer_filter() { + // `WHERE value > 1 AND EXISTS (…)` → Filter{ value > 1 }{ Semi }. + let (left, sub) = (scan(), one_column_subquery()); + let q = filter_of( + ScalarExpr::BoolAnd(vec![value_gt_1(), exists(Rc::clone(&sub), false)]), + Rc::clone(&left), + ); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) = out.non_asap() + else { + panic!("expected an outer Filter, got {out:?}"); + }; + assert_eq!(*pred, value_gt_1()); + let (kind, _, l, r) = join_parts(child); + assert_eq!(kind, JoinKind::Semi); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &sub)); + assert_idempotent(&out); + } + + #[test] + fn two_subquery_conjuncts_become_nested_joins() { + // `WHERE EXISTS (a) AND service NOT EXISTS (b) AND value > 1` sheds + // one conjunct per round: Filter{ value > 1 }{ Anti{ Semi{ l, a }, b } }. + let (left, a, b) = (scan(), one_column_subquery(), one_column_subquery()); + let q = filter_of( + ScalarExpr::BoolAnd(vec![ + exists(Rc::clone(&a), false), + exists(Rc::clone(&b), true), + value_gt_1(), + ]), + Rc::clone(&left), + ); + let out = canonicalize(q).unwrap(); + let Some(NonASAPOp::Filter { + pred: Predicate(pred), + child, + }) = out.non_asap() + else { + panic!("expected an outer Filter, got {out:?}"); + }; + assert_eq!(*pred, value_gt_1()); + let (kind, _, inner, r) = join_parts(child); + assert_eq!(kind, JoinKind::Anti); + assert!(Rc::ptr_eq(r, &b)); + let (kind, _, l, r) = join_parts(inner); + assert_eq!(kind, JoinKind::Semi); + assert!(Rc::ptr_eq(l, &left) && Rc::ptr_eq(r, &a)); + assert_idempotent(&out); + } + + #[test] + fn not_in_subquery_is_left_alone() { + let q = filter_of( + in_subquery(ScalarExpr::Column(1), one_column_subquery(), true), + scan(), + ); + let out = canonicalize(Rc::clone(&q)).unwrap(); + assert!(Rc::ptr_eq(&out, &q), "NOT IN keeps its Filter"); + assert_idempotent(&out); + } + + #[test] + fn lifted_subquery_is_canonicalized() { + // The subquery is itself a promotable heavy-hitter; once lifted into + // the join it is canonical, so a second pass finds nothing to do. + let sub = limit(5, 0, sort(desc(1), count_by_service())); + let q = filter_of(exists(sub, false), scan()); + let out = canonicalize(q).unwrap(); + let (_, _, _, r) = join_parts(&out); + assert!(is_topk_over_count(r)); + assert_idempotent(&out); + } +} diff --git a/crates/types/src/ir/cse.rs b/crates/types/src/ir/cse.rs new file mode 100644 index 000000000..15715b55b --- /dev/null +++ b/crates/types/src/ir/cse.rs @@ -0,0 +1,791 @@ +//! Structural common-subexpression elimination over the unified operator IR: +//! bottom-up hash-consing of [`OperatorNode`] DAGs across a workload's roots. +//! +//! CSE only runs on already-bound, already-canonicalized plans — structural +//! matching is meaningless before canonicalization has converged +//! semantically-equivalent queries onto one shape. [`share_common_sub_dags`] +//! is the single entry point, run once per workload batch (a batch of one +//! still deduplicates a query's own repeated sub-DAGs, see below). +//! +//! ## Algorithm: classic hash-consing / value-numbering +//! +//! Bottom-up: every child is interned before its parent, so two parents whose +//! children were independently deduplicated down to the same `Rc`s are +//! structurally identical iff their own fields also match, without re-walking +//! the sub-DAGs. "Child" means everything [`OperatorNode::children`] returns: +//! the operator inputs *and* the operator nodes a scalar expression reads +//! (`PromqlScalarFromVector`, `ScalarSubquery`, `Exists`, `InSubquery`), so a +//! vector read by `scalar(v)` in two queries is shared like any other input. +//! The scalar expressions themselves stay opaque data on their owning node. +//! +//! ## Correctness: hash is a filter, `PartialEq` is the decision +//! +//! This is the one non-negotiable rule. A **false positive** here — two +//! sub-DAGs wrongly judged shareable — is a wrong query answer, not a missed +//! optimization: two different queries would read each other's data. +//! [`structural_hash`] (SipHash over a canonical serialization, no +//! collision-freedom guarantee) may only narrow the candidate set within one +//! bucket; the typed equality check on that bucket ([`same_node`]) is what +//! actually decides sharing, every time, no exceptions for "the hash probably +//! didn't collide." Equality is intentionally conservative: it recognizes +//! *exact* structural matches only, never "a stricter-accuracy summary could +//! also answer a looser request" (that subsumption question belongs to the +//! ASAP matcher, not here). +//! +//! ## Legality +//! +//! Structural equality is necessary but not sufficient. A non-ASAP node is +//! only ever *returned* as a match for another when its output has a provable +//! unique key (`Schema::has_unique_key()`): a producer's output can only be +//! shared across consumers when its row identity is stable across reads, so +//! an ungrouped aggregate, a `without(..)` grouping, a `Concat`/`SetOp` that +//! drops its keys, … is always inserted fresh even when it is structurally +//! identical to something already interned. An ASAP node (summary state and +//! its evaluations) has no such gate: equal operator, schema and guarantee make +//! it shareable, exactly as post-ASAP sharing decided before this IR. +//! +//! ## Single-query CSE falls out for free +//! +//! A repeated sub-expression within *one* query (the same grouped aggregate on +//! both `BinaryOp` branches) is deduplicated by the same bottom-up interning — +//! a workload of size one still interns bottom-up within that one DAG. + +use std::collections::hash_map::DefaultHasher; +use std::collections::HashMap; +use std::hash::{Hash, Hasher}; +use std::rc::Rc; + +use crate::ir::operator::node::{Operator, OperatorNode}; +use crate::ir::operator::non_asap::NonASAPOp; +use crate::ir::schema::Schema; + +/// [`structural_hash`]'s memoization cache: an already-hashed node's `Rc` +/// pointer to its hash. A fresh cache is always correct; what matters is +/// letting it persist across every node of one bottom-up pass rather than +/// starting a new one per call. The caller must keep every cached node alive +/// for the cache's lifetime, or a reused address would alias a stale entry. +pub type HashCache = HashMap<*const OperatorNode, u64>; + +/// A constant stand-in for every child position. Substituting it before +/// serializing or comparing a node leaves exactly the node's own fields. +fn placeholder() -> Rc { + Rc::new(OperatorNode::with_schema( + Operator::NonASAP(NonASAPOp::Values { + rows: vec![], + schema: Schema::lifted(vec![], None), + }), + Schema::lifted(vec![], None), + )) +} + +/// The operator with every child — operator inputs and the operator nodes +/// referenced from its scalar expressions alike — replaced by +/// [`placeholder`]. What remains is the node's own data: variant tag, scalar +/// expressions (with their operator references blanked), parameters. +fn own_fields(node: &OperatorNode) -> Operator { + let placeholder = placeholder(); + node.operator.map_children(|_| Rc::clone(&placeholder)) +} + +/// Coarse structural hash used only to bucket [`InternTable::intern`]'s +/// candidate search — never the sharing decision ([`same_node`] is). +/// +/// `OperatorNode` carries `f64`s (`ScalarValue::Float64`, quantile targets, +/// `ResultGuarantee` bounds, …), so it cannot derive `std::hash::Hash`. The +/// hash is SipHash over two parts: +/// +/// 1. the canonical JSON of [`own_fields`] plus `result_kind`, `schema`, +/// `guarantee` and `timing` — every field `PartialEq` compares except the +/// children. A scalar expression is serialized as data with each operator +/// node it reads replaced by a constant placeholder, so a reference to an +/// interned sub-DAG contributes nothing of its own here; +/// 2. for every child in [`OperatorNode::children`] order (operator inputs, +/// then scalar-referenced nodes), the child's own `structural_hash`, +/// memoized in `cache` by `Rc` pointer identity. +/// +/// Part 2 is what makes equal sub-DAGs hash equal whether they are reached +/// through an operator input or through a `scalar(v)`, and what keeps the +/// pass linear: a node is generally a DAG, and re-serializing a shared +/// descendant once per parent would cost `O(sub-DAG)` per node instead of +/// `O(1)` beyond the children's already-known hashes. A non-finite `f64` +/// serializes as `null`, merely widening one (still equality-checked) bucket. +pub fn structural_hash(node: &OperatorNode, cache: &mut HashCache) -> u64 { + fn child_hash(child: &Rc, cache: &mut HashCache) -> u64 { + let ptr = Rc::as_ptr(child); + if let Some(&h) = cache.get(&ptr) { + return h; + } + let h = structural_hash(child, cache); + cache.insert(ptr, h); + h + } + + let mut hasher = DefaultHasher::new(); + let own = ( + own_fields(node), + node.result_kind, + &node.schema, + &node.guarantee, + node.timing, + &node.coverage, + ); + serde_json::to_string(&own) + .unwrap_or_default() + .hash(&mut hasher); + for child in node.children() { + child_hash(child, cache).hash(&mut hasher); + } + hasher.finish() +} + +/// Numeric `PartialEq` alone conflates signed zeros. The serialized check is +/// additional evidence, never a replacement for typed equality (JSON maps +/// non-finite floats to `null`). Used for the guarantee, whose bounds are +/// floats a shared node must preserve bit-for-bit. +fn same_value(left: &T, right: &T) -> bool { + left == right + && match (serde_json::to_string(left), serde_json::to_string(right)) { + (Ok(left), Ok(right)) => left == right, + _ => false, + } +} + +/// Memo of child-pair comparisons already decided by [`same_node`], keyed by +/// pointer pair. Only interned (table-owned, hence alive) nodes are keys. +type EqMemo = HashMap<(*const OperatorNode, *const OperatorNode), bool>; + +/// The sharing decision: typed equality of two nodes. +/// +/// `OperatorNode`'s derived `PartialEq` would recurse into children by value +/// even when both sides hold the same `Rc` (`OperatorNode` is not `Eq`, so +/// `Rc` gets no pointer shortcut), expanding a shared diamond once per path. +/// Children are therefore compared by pointer first; only when the pointers +/// differ (an equal child that was not legal to share) are the values +/// compared, memoized per pair so a diamond is still walked once. +fn same_node(left: &OperatorNode, right: &OperatorNode, memo: &mut EqMemo) -> bool { + let (lc, rc) = (left.children(), right.children()); + if lc.len() != rc.len() { + return false; + } + let children_equal = lc.iter().zip(&rc).all(|(a, b)| { + if Rc::ptr_eq(a, b) { + return true; + } + let key = (Rc::as_ptr(a), Rc::as_ptr(b)); + if let Some(&eq) = memo.get(&key) { + return eq; + } + let eq = same_node(a, b, memo); + memo.insert(key, eq); + eq + }); + children_equal + && left.result_kind == right.result_kind + && left.schema == right.schema + && left.timing == right.timing + && left.coverage == right.coverage + && same_value(&left.guarantee, &right.guarantee) + && same_value(&own_fields(left), &own_fields(right)) +} + +/// Bottom-up hash-consing table: structurally-equal, sharing-legal nodes +/// collapse onto one `Rc`. +/// +/// `buckets` is keyed by [`structural_hash`] — a coarse candidate filter +/// only. Every entry within one bucket is a full node kept around for the +/// [`same_node`] comparison that actually decides a match; a hash collision +/// between structurally different nodes just means a harmless linear scan of +/// a few extra candidates. +struct InternTable { + buckets: HashMap>>, + /// Persisted for the table's whole lifetime so hashing is `O(1)` per node + /// beyond its children; every cached node is owned by `buckets`. + hash_cache: HashCache, + eq_memo: EqMemo, +} + +impl InternTable { + fn new() -> Self { + Self { + buckets: HashMap::new(), + hash_cache: HashMap::new(), + eq_memo: HashMap::new(), + } + } + + /// Intern one node whose children are already interned: look it up by + /// [`structural_hash`], confirm with [`same_node`], and — only when + /// sharing is legal (module doc, "Legality") — return the existing `Rc` + /// instead of allocating a new one. + fn intern(&mut self, node: OperatorNode) -> Rc { + let hash = structural_hash(&node, &mut self.hash_cache); + // A node that is not legal to share is never *returned* as a match + // for something else; it still occupies a fresh slot in the bucket + // (harmless: later scans require legality of the new node too). + let reusable = node.is_asap() || node.schema.has_unique_key(); + let bucket = self.buckets.entry(hash).or_default(); + if reusable { + if let Some(existing) = bucket + .iter() + .find(|candidate| same_node(candidate, &node, &mut self.eq_memo)) + { + return Rc::clone(existing); + } + } + let rc = Rc::new(node); + bucket.push(Rc::clone(&rc)); + rc + } +} + +/// Count of *unique* nodes reachable from `root` (pointer identity, +/// following [`OperatorNode::children`]): the real size of the DAG, not a +/// tree-walk count that re-counts a shared descendant once per parent. +pub fn dag_node_count(root: &Rc) -> usize { + OperatorNode::reachable(root).len() +} + +/// Input pointer → (input `Rc`, interned result). The input `Rc` is retained +/// so its address cannot be freed and reused by a fresh allocation while the +/// memo still maps it. +type Visited = HashMap<*const OperatorNode, (Rc, Rc)>; + +/// Intern `node`'s children (recursively), then `node` itself. The rebuilt +/// node keeps `node`'s retained schema, result kind, guarantee and timing: +/// every child is replaced by an equal node, so each derived property stays +/// valid, and the result is `PartialEq`-equal to the input. +fn intern_bottom_up( + table: &mut InternTable, + visited: &mut Visited, + node: &Rc, +) -> Rc { + if let Some((_, interned)) = visited.get(&Rc::as_ptr(node)) { + return Rc::clone(interned); + } + let operator = node + .operator + .map_children(|child| intern_bottom_up(table, visited, child)); + let rebuilt = OperatorNode { + operator, + result_kind: node.result_kind, + schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + timing: node.timing, + coverage: node.coverage.clone(), + }; + let interned = table.intern(rebuilt); + visited.insert(Rc::as_ptr(node), (Rc::clone(node), Rc::clone(&interned))); + interned +} + +/// Share structurally-identical, sharing-legal sub-DAGs across a workload's +/// roots (or within one root). Every root's *value* is unchanged +/// (`PartialEq`-equal to its input) — only its internal `Rc` structure may +/// now alias another root's, or another part of its own DAG. A node already +/// reached through two paths is visited once. +/// +/// `Id` is caller-chosen — a workload entry's key, an index, a query name. +pub fn share_common_sub_dags( + roots: Vec<(Id, Rc)>, +) -> Vec<(Id, Rc)> { + let mut table = InternTable::new(); + let mut visited = Visited::new(); + roots + .into_iter() + .map(|(id, root)| (id, intern_bottom_up(&mut table, &mut visited, &root))) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::operator::agg_intent::AggIntent; + use crate::ir::operator::asap::ASAPOp; + use crate::ir::operator::operator_properties::{BinaryOpKind, GroupKeys, Reduction, Source}; + use crate::ir::properties::guarantee::ResultGuarantee; + use crate::ir::scalar::{ColumnRef, CompareOpKind}; + use crate::ir::schema::state_type::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate, + }; + use crate::ir::schema::{DataType, Field, FieldDataType, Schema}; + use crate::ir::BinaryOperator; + use crate::ir::ScalarExpr; + + use crate::types::AccuracyTarget; + + fn node(op: NonASAPOp) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(op)).unwrap() + } + + /// `[ts, service, value, latency]`, no unique key. + fn scan() -> Rc { + node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + Field::plain("latency", DataType::Float64, false), + ], + 0, + vec![], + ), + }) + } + + fn quantile_agg(by: Vec, col: Option, q: f64) -> Rc { + node(NonASAPOp::Aggregate { + reduction: Reduction::by(by), + measures: vec![AggIntent::Quantile { + col, + q, + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + }) + } + + fn compare(lhs: Rc, rhs: Rc) -> Rc { + node(NonASAPOp::BinaryOp { + operator: BinaryOperator { + checked_relative_division: false, + checked_finite_division: false, + kind: BinaryOpKind::Compare(CompareOpKind::Eq), + vector_match: None, + }, + return_bool: false, + lhs, + rhs, + }) + } + + fn two_roots(a: Rc, b: Rc) -> (Rc, Rc) { + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); + let [(_, ra), (_, rb)] = shared.as_slice() else { + panic!("expected 2 roots"); + }; + (Rc::clone(ra), Rc::clone(rb)) + } + + #[test] + fn distinct_column_quantiles_do_not_merge() { + // Grouped (unique key present) so only the differing `col` blocks it. + let (ra, rb) = two_roots( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(3), 0.5), + ); + assert!(!Rc::ptr_eq(&ra, &rb)); + assert_ne!(ra, rb); + } + + #[test] + fn no_unique_keys_means_no_merge_even_when_structurally_identical() { + let a = quantile_agg(vec![], Some(2), 0.9); + let b = quantile_agg(vec![], Some(2), 0.9); + assert_eq!(a, b, "fixture sanity: structurally equal"); + assert!( + !a.schema.has_unique_key(), + "fixture sanity: a global aggregate has no provable unique key" + ); + let (ra, rb) = two_roots(a, b); + assert!( + !Rc::ptr_eq(&ra, &rb), + "no unique key ⇒ never hoisted, even for an identical structural match" + ); + } + + #[test] + fn median_and_explicit_half_percentile_merge() { + // Two spellings that lower to the identical grouped `Quantile { q: 0.5 }`. + let (m, p) = two_roots( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert!(Rc::ptr_eq(&m, &p)); + } + + #[test] + fn single_query_shares_its_own_repeated_sub_dag() { + // One root with the same grouped aggregate on both branches, built as + // two separately-allocated sub-DAGs (no sharing yet). + let root = compare( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + let shared = share_common_sub_dags(vec![("q", root)]); + let [(_, root)] = shared.as_slice() else { + panic!("expected 1 root"); + }; + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = root.non_asap() else { + panic!("expected BinaryOp root, got {root:?}"); + }; + assert!(Rc::ptr_eq(lhs, rhs)); + } + + #[test] + fn shared_root_value_is_unchanged() { + let a = quantile_agg(vec![1], Some(2), 0.5) + .as_ref() + .clone() + .with_guarantee(Some(ResultGuarantee::exact("fixture"))); + let before = Rc::new(a); + let (ra, _) = two_roots(Rc::clone(&before), Rc::clone(&before)); + assert_eq!(ra.as_ref(), before.as_ref()); + assert!( + ra.guarantee.is_some(), + "retained properties survive the rebuild" + ); + } + + // ── scalar-referenced sub-DAGs ────────────────────────────────────── + + /// `vector(scalar(sum by (service) (up)))`. + fn scalar_of_vector() -> Rc { + let sum_up = node(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![1]), + measures: vec![AggIntent::Sum { col: Some(2) }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + }); + assert!(sum_up.schema.has_unique_key(), "fixture sanity"); + node(NonASAPOp::PromqlVectorFromScalar( + ScalarExpr::PromqlScalarFromVector(sum_up), + )) + } + + fn bridged_vector(root: &Rc) -> &Rc { + match root.non_asap() { + Some(NonASAPOp::PromqlVectorFromScalar(ScalarExpr::PromqlScalarFromVector(v))) => v, + other => panic!("expected vector(scalar(v)), got {other:?}"), + } + } + + #[test] + fn scalar_referenced_vector_is_shared_across_queries() { + let (ra, rb) = two_roots(scalar_of_vector(), scalar_of_vector()); + assert!( + Rc::ptr_eq(bridged_vector(&ra), bridged_vector(&rb)), + "the vector read by scalar(v) is a child and must be interned" + ); + assert!( + !Rc::ptr_eq(&ra, &rb), + "the scalar bridge itself has no unique key and stays separate" + ); + } + + #[test] + fn structural_hash_sees_through_a_scalar_reference() { + // Two equal bridges must hash equal whether or not their referenced + // vector is the same Rc — the reference contributes the vector's + // memoized hash, not its identity. + let a = scalar_of_vector(); + let b = scalar_of_vector(); + let mut cache = HashMap::new(); + assert_eq!( + structural_hash(&a, &mut cache), + structural_hash(&b, &mut cache) + ); + assert_eq!( + cache.len(), + 4, + "aggregate + scan cached once per root: {cache:?}" + ); + let other = node(NonASAPOp::PromqlVectorFromScalar( + ScalarExpr::PromqlScalarFromVector(quantile_agg(vec![1], Some(2), 0.5)), + )); + assert_ne!( + structural_hash(&a, &mut cache), + structural_hash(&other, &mut cache) + ); + } + + // ── ASAP nodes ────────────────────────────────────────────────────── + + fn summary_agg(alpha: f64, guarantee: Option) -> Rc { + let family = FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha }), + GroupingStrategy::default(), + ); + let schema = Schema::lifted(vec![Field::new("state", family.clone(), false)], None); + assert!(!schema.has_unique_key(), "fixture sanity"); + Rc::new( + OperatorNode::with_schema( + Operator::ASAP(ASAPOp::SummaryAgg { + child: scan(), + family, + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::PerEntity, + grouping: GroupingStrategy::default(), + filter: None, + }), + schema, + ) + .with_guarantee(guarantee), + ) + } + + #[test] + fn asap_nodes_share_without_a_unique_key() { + let exact = || Some(ResultGuarantee::exact("fixture")); + let (ra, rb) = two_roots(summary_agg(0.01, exact()), summary_agg(0.01, exact())); + assert!(Rc::ptr_eq(&ra, &rb)); + assert!(ra.guarantee.is_some()); + } + + #[test] + fn asap_nodes_with_distinct_parameters_or_guarantees_are_not_shared() { + let exact = || Some(ResultGuarantee::exact("fixture")); + let (ra, rb) = two_roots(summary_agg(0.01, exact()), summary_agg(0.001, exact())); + assert!(!Rc::ptr_eq(&ra, &rb), "different sketch parameters"); + let (ra, rb) = two_roots(summary_agg(0.01, exact()), summary_agg(0.01, None)); + assert!( + !Rc::ptr_eq(&ra, &rb), + "an unknown guarantee never borrows an exact one" + ); + assert!(rb.guarantee.is_none()); + } + + #[test] + fn evaluations_share_their_producer_but_not_each_other() { + use crate::ir::schema::state_type::SketchStatistic; + let evaluation = |q: f64| { + Rc::new(OperatorNode::with_schema( + Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: summary_agg(0.01, None), + query: SketchStatistic::Quantile { q }, + }), + Schema::lifted( + vec![Field::plain("quantile", DataType::Float64, false)], + None, + ), + )) + }; + let (p95, p99) = two_roots(evaluation(0.95), evaluation(0.99)); + let producer = |n: &Rc| Rc::clone(n.children()[0]); + assert!(!Rc::ptr_eq(&p95, &p99)); + assert!(Rc::ptr_eq(&producer(&p95), &producer(&p99))); + } + + // ── structural_hash (DAG-aware memoization) ───────────────────────── + + #[test] + fn structural_hash_is_stable_across_cache_states() { + let agg = quantile_agg(vec![1], Some(2), 0.5); + let mut cold = HashMap::new(); + let mut warm = HashMap::new(); + structural_hash(&scan(), &mut warm); + assert_eq!( + structural_hash(&agg, &mut cold), + structural_hash(&agg, &mut warm), + "hash must be independent of unrelated cache state" + ); + } + + #[test] + fn structural_hash_of_an_internally_shared_dag_matches_the_unshared_equivalent() { + let agg = quantile_agg(vec![1], Some(2), 0.5); + let shared_root = compare(Rc::clone(&agg), Rc::clone(&agg)); + let unshared_root = compare( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert_eq!( + structural_hash(&shared_root, &mut HashMap::new()), + structural_hash(&unshared_root, &mut HashMap::new()), + ); + } + + #[test] + fn structural_hash_memoizes_a_shared_descendant_exactly_once() { + let agg = quantile_agg(vec![1], Some(2), 0.5); + let root = compare(Rc::clone(&agg), Rc::clone(&agg)); + let mut cache = HashMap::new(); + structural_hash(&root, &mut cache); + assert_eq!( + cache.len(), + 2, + "one entry per unique node in the shared branch (Aggregate + Scan): {cache:?}" + ); + } + + // ── dag_node_count ─────────────────────────────────────────────────── + + #[test] + fn dag_node_count_is_the_naive_count_when_nothing_is_shared() { + assert_eq!(dag_node_count(&scan()), 1); + assert_eq!(dag_node_count(&quantile_agg(vec![1], Some(2), 0.5)), 2); + assert_eq!( + dag_node_count(&scalar_of_vector()), + 3, + "follows scalar references" + ); + } + + #[test] + fn dag_node_count_deduplicates_an_internally_shared_sub_dag() { + let root = compare( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert_eq!( + dag_node_count(&root), + 5, + "fixture sanity: nothing shared yet" + ); + let shared = share_common_sub_dags(vec![("q", root)]); + let [(_, root)] = shared.as_slice() else { + panic!("expected 1 root"); + }; + assert_eq!( + dag_node_count(root), + 3, + "BinaryOp + one Aggregate + its Scan" + ); + } + + #[test] + fn dag_node_count_deduplicates_across_two_workload_roots() { + let (ra, rb) = two_roots( + quantile_agg(vec![1], Some(2), 0.5), + quantile_agg(vec![1], Some(2), 0.5), + ); + assert!(Rc::ptr_eq(&ra, &rb), "fixture sanity: the two roots merged"); + assert_eq!(dag_node_count(&ra), 2); + assert_eq!(dag_node_count(&rb), 2); + } + + #[test] + fn dedup_gates_sharing_the_same_as_aggregate() { + // `Dedup { cols }` adds `cols` as a unique key, so two identical + // `Dedup`s merge even though their keyless `Scan`s could not. + let dedup = || { + node(NonASAPOp::Dedup { + cols: vec![1], + child: scan(), + }) + }; + let (ra, rb) = two_roots(dedup(), dedup()); + assert!(Rc::ptr_eq(&ra, &rb)); + } + + #[test] + fn group_keys_gate_still_prevented_when_partition_by_without_used() { + let without_agg = || { + node(NonASAPOp::Aggregate { + reduction: Reduction::Reduce(GroupKeys::without(vec![0])), + measures: vec![AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }], + output_names: vec![], + filters: vec![], + having: None, + child: scan(), + }) + }; + let a = without_agg(); + assert!(!a.schema.has_unique_key()); + let (ra, rb) = two_roots(a, without_agg()); + assert!(!Rc::ptr_eq(&ra, &rb)); + } + + #[test] + fn already_shared_nodes_are_visited_once() { + // A diamond already present in the input stays one node and is not + // re-interned per path. + let agg = quantile_agg(vec![1], Some(2), 0.5); + let root = compare(Rc::clone(&agg), Rc::clone(&agg)); + let shared = share_common_sub_dags(vec![("q", root)]); + let Some(NonASAPOp::BinaryOp { lhs, rhs, .. }) = shared[0].1.non_asap() else { + panic!("expected BinaryOp root"); + }; + assert!(Rc::ptr_eq(lhs, rhs)); + assert_eq!(dag_node_count(&shared[0].1), 3); + } + + // Comparing a shareable node whose equal-but-unshareable children form a + // deep diamond must not expand the diamond once per path. The timeout is + // a coarse runaway guard, not a performance SLA. + #[test] + fn shared_diamond_does_not_expand_during_comparison() { + let (done, completion) = std::sync::mpsc::channel(); + let worker = std::thread::spawn(move || { + fn keyed_diamond() -> Rc { + // BinaryOp over a keyless scan has no unique key at any level, + // so none of the 24 levels is shareable; the `Dedup` on top is. + let mut current = scan(); + for _ in 0..24 { + current = compare(Rc::clone(¤t), current); + } + node(NonASAPOp::Dedup { + cols: vec![1], + child: current, + }) + } + let (ra, rb) = two_roots(keyed_diamond(), keyed_diamond()); + assert!(Rc::ptr_eq(&ra, &rb)); + done.send(()).unwrap(); + }); + completion + .recv_timeout(std::time::Duration::from_secs(5)) + .expect("comparison expanded the shared DAG"); + worker.join().unwrap(); + } + + /// A keyed (hence shareable) projection emitting the literal `value`. + fn keyed_literal(value: f64) -> Rc { + let keyed = node(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index( + vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("service", DataType::Utf8, false), + ], + 0, + vec![vec![1]], + ), + }); + node(NonASAPOp::Project { + cols: vec![ + crate::ir::ProjectItem { + alias: None, + expr: ScalarExpr::Column(1), + }, + crate::ir::ProjectItem { + alias: Some("v".into()), + expr: ScalarExpr::literal_f64(value), + }, + ], + qualifier: None, + child: keyed, + }) + } + + /// Sharing preserves IEEE signed zero, and JSON's `null` encoding of + /// non-finite floats never becomes the equality decision. + #[test] + fn signed_zero_and_nonfinite_values_remain_distinct() { + assert!( + keyed_literal(0.0).schema.has_unique_key(), + "fixture is shareable" + ); + for (a, b) in [ + (0.0, -0.0), + (-0.0, 0.0), + (f64::INFINITY, f64::NEG_INFINITY), + (f64::NAN, f64::NAN), + ] { + let (ra, rb) = two_roots(keyed_literal(a), keyed_literal(b)); + assert!(!Rc::ptr_eq(&ra, &rb), "{a} and {b} must not be shared"); + } + let (ra, rb) = two_roots(keyed_literal(f64::INFINITY), keyed_literal(f64::INFINITY)); + assert!(Rc::ptr_eq(&ra, &rb)); + } +} diff --git a/crates/types/src/ir/export.rs b/crates/types/src/ir/export.rs new file mode 100644 index 000000000..b297586ac --- /dev/null +++ b/crates/types/src/ir/export.rs @@ -0,0 +1,371 @@ +//! Logical ASAP DAG transport (planner-layering stage 1), with no execution timing assigned. +//! +//! This representation preserves operator semantics and summary state types. +//! Timing is derived from materialization during physical planning; +//! physical implementation, materialization and retention remain downstream. +use std::collections::{HashMap, HashSet}; +use std::rc::Rc; + +use serde::{Deserialize, Serialize}; +use thiserror::Error; + +pub use crate::ir::physical_export::*; +use crate::ir::properties::guarantee::ResultGuarantee; +use crate::ir::schema::{FieldDataType, Schema}; +pub use crate::ir::wire::NonASAPOpKind; +use crate::ir::wire::{grouping_compatibility, input_edges, payload_of}; +pub use crate::ir::wire::{ + EdgeRole, GroupingEdgeCompatibility, LogicalASAPNodeId, LogicalASAPOperatorPayload, + WirePredicate, WireProjectItem, WireScalarExpr, WireSortKey, +}; +use crate::ir::{ + ASAPOp, Operator, OperatorNode, OperatorResultKind, QueryRoot, SchemaDerivationError, +}; + +/// Independent envelope version: this replaces the older phase-assigned format. +pub const LOGICAL_ASAP_DAG_WIRE_VERSION: u32 = 1; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LogicalASAPDAGNode { + pub id: LogicalASAPNodeId, + pub payload: LogicalASAPOperatorPayload, + pub result_kind: OperatorResultKind, + pub output_schema: Schema, + pub guarantee: Option, + #[serde(default)] + pub coverage: Option, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LogicalASAPDAGEdge { + pub producer: LogicalASAPNodeId, + pub consumer: LogicalASAPNodeId, + pub role: EdgeRole, + pub intermediate_schema: Schema, + pub grouping: GroupingEdgeCompatibility, +} + +/// Standalone scalars remain scalar roots rather than fabricated operator nodes. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub enum LogicalASAPQueryRoot { + Operator(LogicalASAPNodeId), + Scalar(WireScalarExpr), +} +impl LogicalASAPQueryRoot { + pub fn operator_refs(&self) -> Vec { + match self { + Self::Operator(id) => vec![*id], + Self::Scalar(expr) => expr.operator_refs(), + } + } +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LogicalASAPDAG { + pub nodes: Vec, + pub edges: Vec, + /// One root per query of the batch, in workload order. Queries that share + /// a sub-DAG reference the same exported nodes. + pub roots: Vec, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LogicalASAPDAGDocument { + pub schema_version: u32, + pub dag: LogicalASAPDAG, +} + +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum LogicalASAPDAGValidationError { + #[error("unsupported logical ASAP DAG version {0}")] + UnsupportedVersion(u32), + #[error("duplicate logical node {0:?}")] + DuplicateNode(LogicalASAPNodeId), + #[error("missing logical node {0:?}")] + MissingNode(LogicalASAPNodeId), + #[error("edge schema differs from producer {0:?}")] + EdgeSchemaMismatch(LogicalASAPNodeId), + #[error("summary node {0:?} schema does not contain its declared family/grouping")] + SummarySchemaMismatch(LogicalASAPNodeId), + #[error("invalid summary coverage at {0:?}")] + InvalidCoverage(LogicalASAPNodeId), + #[error("logical ASAP DAG has no query roots")] + NoRoots, + #[error("logical ASAP DAG contains a cycle")] + Cycle, + #[error("unreachable logical node {0:?}")] + UnreachableNode(LogicalASAPNodeId), +} + +impl LogicalASAPDAGDocument { + pub fn new(dag: LogicalASAPDAG) -> Self { + Self { + schema_version: LOGICAL_ASAP_DAG_WIRE_VERSION, + dag, + } + } + + pub fn validate(&self) -> Result<(), LogicalASAPDAGValidationError> { + if self.schema_version != LOGICAL_ASAP_DAG_WIRE_VERSION { + return Err(LogicalASAPDAGValidationError::UnsupportedVersion( + self.schema_version, + )); + } + self.dag.validate() + } +} + +impl LogicalASAPDAG { + /// Transport integrity checks; full operator/scalar typing is checked on + /// the in-memory IR before compilation. + pub fn validate(&self) -> Result<(), LogicalASAPDAGValidationError> { + let mut nodes = HashMap::new(); + for node in &self.nodes { + if nodes.insert(node.id, node).is_some() { + return Err(LogicalASAPDAGValidationError::DuplicateNode(node.id)); + } + if let Some(coverage) = &node.coverage { + if node.result_kind != OperatorResultKind::State || coverage.validate().is_err() { + return Err(LogicalASAPDAGValidationError::InvalidCoverage(node.id)); + } + } + if let LogicalASAPOperatorPayload::SummaryAgg { + family, grouping, .. + } = &node.payload + { + if node.coverage.is_none() { + return Err(LogicalASAPDAGValidationError::InvalidCoverage(node.id)); + } + if !node.output_schema.fields.iter().any(|field| &field.dtype == family) + || node.output_schema.fields.iter().any(|field| matches!(&field.dtype, FieldDataType::Sketch(_, actual) if actual != grouping)) { + return Err(LogicalASAPDAGValidationError::SummarySchemaMismatch(node.id)); + } + } + } + if self.roots.is_empty() { + return Err(LogicalASAPDAGValidationError::NoRoots); + } + let roots: Vec<_> = self + .roots + .iter() + .flat_map(LogicalASAPQueryRoot::operator_refs) + .collect(); + for root in &roots { + if !nodes.contains_key(root) { + return Err(LogicalASAPDAGValidationError::MissingNode(*root)); + } + } + let mut inputs: HashMap<_, Vec<_>> = HashMap::new(); + for edge in &self.edges { + let producer = nodes + .get(&edge.producer) + .ok_or(LogicalASAPDAGValidationError::MissingNode(edge.producer))?; + if !nodes.contains_key(&edge.consumer) { + return Err(LogicalASAPDAGValidationError::MissingNode(edge.consumer)); + } + if edge.intermediate_schema != producer.output_schema { + return Err(LogicalASAPDAGValidationError::EdgeSchemaMismatch( + edge.producer, + )); + } + inputs.entry(edge.consumer).or_default().push(edge.producer); + } + for node in &self.nodes { + if matches!(node.payload, LogicalASAPOperatorPayload::SummaryMerge) { + let coverage = inputs + .get(&node.id) + .into_iter() + .flatten() + .map(|id| { + nodes[id] + .coverage + .clone() + .ok_or(LogicalASAPDAGValidationError::InvalidCoverage(node.id)) + }) + .collect::, _>>()?; + let merged = + crate::ir::properties::summary_coverage::SummaryCoverage::merge_disjoint( + &coverage, + ) + .map_err(|_| LogicalASAPDAGValidationError::InvalidCoverage(node.id))?; + if node.coverage.as_ref() != Some(&merged) { + return Err(LogicalASAPDAGValidationError::InvalidCoverage(node.id)); + } + } + } + fn visit( + id: LogicalASAPNodeId, + inputs: &HashMap>, + active: &mut HashSet, + done: &mut HashSet, + ) -> Result<(), LogicalASAPDAGValidationError> { + if done.contains(&id) { + return Ok(()); + } + if !active.insert(id) { + return Err(LogicalASAPDAGValidationError::Cycle); + } + for child in inputs.get(&id).into_iter().flatten() { + visit(*child, inputs, active, done)?; + } + active.remove(&id); + done.insert(id); + Ok(()) + } + let mut done = HashSet::new(); + for root in roots { + visit(root, &inputs, &mut HashSet::new(), &mut done)?; + } + if let Some(id) = nodes.keys().find(|id| !done.contains(id)) { + return Err(LogicalASAPDAGValidationError::UnreachableNode(*id)); + } + Ok(()) + } +} + +/// Compiler-local identity mapping; IDs are local to this logical export. +#[derive(Debug, Clone)] +pub struct LogicalASAPNodeIdentityMap { + nodes_by_id: Vec>, +} + +impl LogicalASAPNodeIdentityMap { + pub fn node_id(&self, node: &Rc) -> Option { + self.nodes_by_id + .iter() + .position(|candidate| Rc::ptr_eq(candidate, node)) + .map(|id| LogicalASAPNodeId(id as u32)) + } + pub fn operator_node(&self, id: LogicalASAPNodeId) -> Option<&Rc> { + self.nodes_by_id.get(id.0 as usize) + } +} + +#[derive(Debug, Clone)] +pub struct LogicalASAPDAGCompilation { + pub dag: LogicalASAPDAG, + pub node_ids: LogicalASAPNodeIdentityMap, +} + +pub fn compile_logical_asap_dag( + root: &Rc, +) -> Result { + Ok(compile_logical_asap_dag_with_node_ids(root)?.dag) +} + +pub fn compile_logical_asap_dag_with_node_ids( + root: &Rc, +) -> Result { + compile_logical_asap_query_with_node_ids(&QueryRoot::Operator(Rc::clone(root))) +} + +pub fn compile_logical_asap_query( + root: &QueryRoot, +) -> Result { + Ok(compile_logical_asap_query_with_node_ids(root)?.dag) +} + +pub fn compile_logical_asap_query_with_node_ids( + root: &QueryRoot, +) -> Result { + compile_logical_asap_workload_with_node_ids(std::slice::from_ref(root)) +} + +/// Export a batch of queries as one DAG with one root per query. +pub fn compile_logical_asap_workload( + roots: &[QueryRoot], +) -> Result { + Ok(compile_logical_asap_workload_with_node_ids(roots)?.dag) +} + +pub fn compile_logical_asap_workload_with_node_ids( + roots: &[QueryRoot], +) -> Result { + let mut exporter = Exporter::default(); + let mut exported = Vec::with_capacity(roots.len()); + for root in roots { + root.validate_structure()?; + exported.push(match root { + QueryRoot::Operator(node) => LogicalASAPQueryRoot::Operator(exporter.visit(node)), + QueryRoot::Scalar(expr) => { + for node in expr.operator_refs() { + exporter.visit(node); + } + LogicalASAPQueryRoot::Scalar(WireScalarExpr::from_expr(expr, &mut |n| { + exporter.ids[&Rc::as_ptr(n)] + })) + } + }); + } + let dag = LogicalASAPDAG { + nodes: exporter.nodes, + edges: exporter.edges, + roots: exported, + }; + Ok(LogicalASAPDAGCompilation { + dag, + node_ids: LogicalASAPNodeIdentityMap { + nodes_by_id: exporter.nodes_by_id, + }, + }) +} + +#[derive(Default)] +struct Exporter { + ids: HashMap<*const OperatorNode, LogicalASAPNodeId>, + nodes: Vec, + edges: Vec, + nodes_by_id: Vec>, +} + +impl Exporter { + fn visit(&mut self, node: &Rc) -> LogicalASAPNodeId { + if let Some(id) = self.ids.get(&Rc::as_ptr(node)) { + return *id; + } + let mut producers = Vec::new(); + for (child, role) in input_edges(&node.operator) { + producers.push((self.visit(child), child, role)); + } + let scalars = match &node.operator { + Operator::NonASAP(op) => op.scalar_exprs(), + Operator::ASAP(ASAPOp::SummaryAgg { + filter: Some(filter), + .. + }) => vec![&filter.0], + _ => vec![], + }; + for expr in scalars { + for referenced in expr.operator_refs() { + producers.push((self.visit(referenced), referenced, EdgeRole::ScalarRef)); + } + } + let id = LogicalASAPNodeId(self.nodes.len() as u32); + let payload = payload_of(&node.operator, &mut |n| self.ids[&Rc::as_ptr(n)]); + self.nodes.push(LogicalASAPDAGNode { + id, + payload, + result_kind: node.result_kind, + output_schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + coverage: node.coverage.clone(), + }); + self.nodes_by_id.push(Rc::clone(node)); + self.ids.insert(Rc::as_ptr(node), id); + for (producer, child, role) in producers { + self.edges.push(LogicalASAPDAGEdge { + producer, + consumer: id, + role, + intermediate_schema: child.schema.clone(), + grouping: grouping_compatibility(&child.operator, &node.operator), + }); + } + id + } +} diff --git a/crates/types/src/ir/mod.rs b/crates/types/src/ir/mod.rs index d20bfd075..fb5e97847 100644 --- a/crates/types/src/ir/mod.rs +++ b/crates/types/src/ir/mod.rs @@ -1,17 +1,33 @@ -//! Unified operator and scalar representation from #511. -//! Graph algorithms are added in the next stack layer; legacy consumers -//! remain on their existing representation until the planner cutover. -pub mod aggregate_schema; -pub mod asap; -pub mod error; -pub mod node; -pub mod non_asap; -pub mod operator_properties; +//! The operator IR from #511: one operator DAG for every planning stage. +//! +//! - [`operator`] — §1 operators: [`OperatorNode`] and its operator families. +//! - [`scalar`] — §2.2 scalar expressions and column references. +//! - [`schema`] — §2.1 per-edge [`schema::Schema`] and summary state types. +//! - [`properties`] — §2.2–2.3 node properties: accuracy guarantees, execution +//! timing, and summary coverage. +//! - [`export`], [`physical_export`], [`cse`], [`canonicalize`] — DAG passes and +//! wire transport. +pub mod operator; +pub mod properties; pub mod query; pub mod scalar; -pub use asap::{ASAPOp, UNIMPLEMENTED_ASAP_OP}; -pub use error::SchemaDerivationError; -pub use node::{Operator, OperatorNode, OperatorResultKind}; -pub use non_asap::{BinaryOperator, NonASAPOp, TimeRangeKind}; +pub mod schema; +pub use operator::asap::{ASAPOp, UNIMPLEMENTED_ASAP_OP}; +pub use operator::node::{Operator, OperatorNode, OperatorResultKind}; +pub use operator::non_asap::{BinaryOperator, NonASAPOp, TimeRangeKind}; pub use query::QueryRoot; pub use scalar::{ExprSemantics, Predicate, ProjectItem, ScalarExpr, SortKey}; +pub use schema::error::SchemaDerivationError; + +pub mod canonicalize; +pub mod cse; +pub mod export; +/// Physical ASAP DAG transport: the logical payloads plus execution timing. +pub mod physical_export; +pub use properties::timing::{ + apply_materialization_timings, data_state, planned_data_state, split_shared_by_phase, + validate_maintained, MaterializationAssignment, TimingMemo, +}; +mod wire; + +pub mod schema_support; diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/ir/operator/agg_intent.rs similarity index 96% rename from crates/types/src/pre_asap/agg_intent.rs rename to crates/types/src/ir/operator/agg_intent.rs index c60dd55e3..c65ea5bed 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/ir/operator/agg_intent.rs @@ -10,28 +10,28 @@ //! heavy-hitter sketch when approximate — is a post-ASAP cost-aware decision, //! not encoded here. The semantic distinction that *is* made at lowering is //! intent vs operator: a heavy-hitter aggregate becomes `TopK`, whereas a -//! generic `ORDER BY value LIMIT k` stays as the `QueryExpr::Sort + Limit` +//! generic `ORDER BY value LIMIT k` stays as the `NonASAPOp::Sort + Limit` //! operator pair. use serde::{Deserialize, Serialize}; -use crate::pre_asap::query_expr::DataModel; -use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType}; +use crate::ir::operator::operator_properties::DataModel; +use crate::ir::schema::{ColumnId, DataType, Field, FieldDataType}; use crate::types::AccuracyTarget; /// "What to compute" — the vocabulary the planner pivots on. /// -/// Grouping for `TopK` rides on the enclosing `QueryExpr::Aggregate.by` +/// Grouping for `TopK` rides on the enclosing `NonASAPOp::Aggregate`'s `reduction` /// (positional `ColumnId`s), like every other aggregate; the intent itself /// carries only `k` + the accuracy target. /// /// The single-column reducers (`Sum` / `Min` / `Max` / `Avg` / `StdDev` / /// `Variance` / `Quantile`) carry `col: Option` — the input /// column they reduce, generic over the column-reference state the same way -/// [`QueryExpr`](super::query_expr::QueryExpr) is: positional `ColumnId` once -/// bound (the default, and every existing use of the bare `AggIntent` name), -/// or an unresolved name-based `ColumnRef` for a front end constructing this -/// intent directly, before the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has run. +/// the rest of the vocabulary is: positional `ColumnId` once bound (the +/// default, and every existing use of the bare `AggIntent` name), or an +/// unresolved name-based `ColumnRef` for a front end constructing this +/// intent directly, before name resolution (`asap_frontend_common`) has run. /// `None` is the PromQL convention "the time-series sample value"; SQL /// `SUM(bytes), AVG(latency)` sets distinct `Some(_)`s so a multi-aggregate /// node binds each reducer to the right column, and `plan::bind` knows which @@ -128,7 +128,7 @@ pub enum AggIntent { // ── Time-series streaming derivatives ──────────────────────────────── // Counter-reset adjustment; not equivalent to Sum/Count over a window. - // The temporal range lives on the enclosing `QueryExpr::TimeRange` node, + // The temporal range lives on the enclosing `NonASAPOp::TimeRange` node, // not in the intent — this keeps the intent vocabulary range-agnostic. Rate, /// PromQL `irate(v[w])` — reset-aware rate from the final two samples. @@ -234,7 +234,7 @@ pub enum AggIntent { /// A time / calendar accessor (issue #46) — `timestamp`, `minute`, `hour`, /// `day_of_week`, … over each sample's timestamp (or, for the no-arg forms, /// over the evaluation time). Label-preserving per-series value transform. - /// (`time()` is the evaluation time itself — a `QueryExpr::EvalTimestamp` leaf, + /// (`time()` is the evaluation time itself — a `ScalarExpr::EvalTimestamp` leaf, /// not this.) TimeFn(TimeFunc), @@ -378,9 +378,8 @@ pub enum MathFunc { } // `requires` / `is_per_series` / `output_column` never read `col`'s value — -// only its presence via a `{ .. }` pattern — so, unlike -// `QueryExpr::output_schema` (which genuinely cannot compile for an -// unresolved DAG — see its own doc), nothing stops these from being generic +// only its presence via a `{ .. }` pattern — so, unlike schema derivation +// (which needs bound positions), nothing stops these from being generic // over every `C`. And a front end constructing `AggIntent` // directly (issue #179) does need `is_per_series` pre-binding — it decides // the `PerEntity`/`Reduce` reduction shape right at construction time (see @@ -393,7 +392,7 @@ impl AggIntent { /// malformed selectors fail instead of acquiring a fabricated output type. pub fn arg_selector_columns( &self, - schema: &super::schema::Schema, + schema: &crate::ir::schema::Schema, ) -> Result, String> { let Self::Extension { ext_kind, payload } = self else { return Ok(None); @@ -408,14 +407,14 @@ impl AggIntent { return Err("arg selector requires arg_col and val_col only".into()); } let resolve = |field: &str| -> Result { - let reference: super::expr_ir::ColumnRef = serde_json::from_value( + let reference: crate::ir::scalar::ColumnRef = serde_json::from_value( fields .get(field) .ok_or_else(|| format!("missing arg selector {field}"))? .clone(), ) .map_err(|e| e.to_string())?; - super::column_resolution::resolve_column_ref(&reference, schema) + crate::ir::scalar::column_resolution::resolve_column_ref(&reference, schema) .map_err(|e| e.to_string()) }; Ok(Some((resolve("arg_col")?, resolve("val_col")?))) @@ -529,7 +528,7 @@ impl AggIntent { impl AggIntent { /// Output column name + type produced by this intent over `input`. - /// Used by `QueryExpr::Aggregate`'s schema-derivation rule. The PromQL + /// Used by `NonASAPOp::Aggregate`'s schema-derivation rule. The PromQL /// convention names the column after the intent kind so consumers can /// locate it without an alias lookup. pub fn output_column(&self, input: &Field) -> Field { @@ -549,8 +548,8 @@ impl AggIntent { DataType::Float64, false, ), - // TopK output is a per-row struct/list; modeled as Utf8 here - // (the post-ASAP sketch-bound IR upgrades the dtype). + // A single-measure `by` top-k instead returns its selected rows + // (`aggregate_schema::ranked_rows_schema`). AggIntent::TopK { k, .. } => col(&format!("topk_{k}"), DataType::Utf8, false), AggIntent::Cardinality { .. } => col("cardinality", DataType::Int64, false), AggIntent::FrequencyL2 { .. } => col("frequency_l2", DataType::Float64, false), @@ -739,7 +738,7 @@ pub fn default_quantile(q: f64) -> AggIntent { #[cfg(test)] mod tests { use super::*; - use crate::pre_asap::schema::{DataType, Field}; + use crate::ir::schema::{DataType, Field}; fn c(name: &str, dtype: DataType) -> Field { Field::plain(name, dtype, false) @@ -967,7 +966,8 @@ mod tests { #[cfg(test)] mod arg_selector_contract_tests { use super::*; - use crate::pre_asap::{ColumnRef, Schema}; + use crate::ir::scalar::ColumnRef; + use crate::ir::schema::Schema; #[test] fn arg_selector_rejects_missing_or_unresolved_arguments() { let schema = Schema::new(vec![Field::plain("value", DataType::Float64, false)]); diff --git a/crates/types/src/ir/asap.rs b/crates/types/src/ir/operator/asap.rs similarity index 72% rename from crates/types/src/ir/asap.rs rename to crates/types/src/ir/operator/asap.rs index faee35c5f..d1baee5e8 100644 --- a/crates/types/src/ir/asap.rs +++ b/crates/types/src/ir/operator/asap.rs @@ -5,12 +5,13 @@ use std::rc::Rc; use serde::{Deserialize, Serialize}; -use super::node::{OperatorNode, OperatorResultKind}; -use crate::ir::operator_properties::Reduction; +use crate::ir::operator::maintained_population::{MaintainedPopulation, PopulationStatistic}; +use crate::ir::operator::node::{OperatorNode, OperatorResultKind}; +use crate::ir::operator::operator_properties::Reduction; +use crate::ir::properties::summary_coverage::{CoverageError, SummaryCoverage}; +use crate::ir::schema::state_type::{GroupingStrategy, SketchStatistic, SummaryUpdate}; +use crate::ir::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; use crate::ir::SchemaDerivationError; -use crate::post_asap::maintained_population::{MaintainedPopulation, PopulationStatistic}; -use crate::post_asap::sketch::{GroupingStrategy, SketchStatistic, SummaryUpdate}; -use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; /// Why an ASAP operator cannot be used yet. pub const UNIMPLEMENTED_ASAP_OP: &str = @@ -29,7 +30,7 @@ pub enum ASAPOp { reduction: Reduction, grouping: GroupingStrategy, #[serde(default)] - filter: Option, + filter: Option, }, /// Read out a query result from built summary state. Output is a /// row-shaped schema. @@ -39,9 +40,7 @@ pub enum ASAPOp { }, /// Read an exact accumulator's state as its finalized value: the /// maintenance-to-read boundary before query-time operators. - FinalizeExactAccumulator { - child: Rc, - }, + FinalizeExactAccumulator { child: Rc }, /// Maintain the full declared population, including membership changes. MaintainPopulation { child: Rc, @@ -52,10 +51,9 @@ pub enum ASAPOp { child: Rc, evaluation: PopulationStatistic, }, - // ── Reserved: migrated but unimplemented (§1.3 of the proposal) ── - SummaryMerge { - children: Vec>, - }, + /// Merge compatible partial states for the same grouping and family. + SummaryMerge { children: Vec> }, + // ── Reserved: migrated but unimplemented ── SummarySubtract { left: Rc, right: Rc, @@ -118,7 +116,7 @@ impl ASAPOp { grouping: grouping.clone(), filter: filter .as_ref() - .map(|p| super::scalar::Predicate(p.0.map_operator_refs(&mut f))), + .map(|p| crate::ir::scalar::Predicate(p.0.map_operator_refs(&mut f))), }, SummaryEstimate { summary_input, @@ -186,11 +184,7 @@ impl ASAPOp { use ASAPOp::*; matches!( self, - SummaryMerge { .. } - | SummarySubtract { .. } - | SummaryDelete { .. } - | SummaryJoin { .. } - | Extension { .. } + SummarySubtract { .. } | SummaryDelete { .. } | SummaryJoin { .. } | Extension { .. } ) } @@ -202,10 +196,32 @@ impl ASAPOp { pub fn produced_state(&self) -> Option<&FieldDataType> { match self { ASAPOp::SummaryAgg { family, .. } | ASAPOp::SummaryJoin { family, .. } => Some(family), + ASAPOp::SummaryMerge { children } => children.first().and_then(|child| { + child + .schema + .fields + .iter() + .find(|field| !field.is_plain()) + .map(|field| &field.dtype) + }), _ => None, } } + /// Derive the merged node's coverage; unknown or overlapping inputs fail closed. + pub fn merged_coverage(&self) -> Result { + let ASAPOp::SummaryMerge { children } = self else { + return Err(SchemaDerivationError::InvalidScalarSignature( + "coverage merge requires SummaryMerge".into(), + )); + }; + let inputs = children + .iter() + .map(|child| child.coverage.clone().ok_or(CoverageError::UnknownInput)) + .collect::, _>>()?; + Ok(SummaryCoverage::merge_disjoint(&inputs)?) + } + /// Output schema derived from the operator and its children. Summary /// planning may retain a more specific schema (evaluation column naming) /// through [`OperatorNode::with_schema`]; all structural metadata must @@ -219,15 +235,15 @@ impl ASAPOp { reduction, .. } => { - let mut schema = crate::ir::aggregate_schema::aggregate_output_schema( + let mut schema = crate::ir::schema::aggregate_schema::aggregate_output_schema( &child.schema, reduction, - &[crate::pre_asap::AggIntent::Sum { col: None }], + &[crate::ir::operator::AggIntent::Sum { col: None }], &[], )?; let index = match reduction { - Reduction::PerEntity => crate::pre_asap::resolve_column_ref( - &crate::pre_asap::ColumnRef::SampleValue, + Reduction::PerEntity => crate::ir::scalar::resolve_column_ref( + &crate::ir::scalar::ColumnRef::SampleValue, &schema, ) .map_err(|e| SchemaDerivationError::InvalidScalarSignature(e.to_string()))?, @@ -247,7 +263,7 @@ impl ASAPOp { SketchStatistic::PointCount { .. } => ("count", DataType::Int64), SketchStatistic::FrequencyL2 => ("frequency_l2", DataType::Float64), SketchStatistic::FrequencyEntropy => ("frequency_entropy", DataType::Float64), - SketchStatistic::TopK { .. } => ("topk", DataType::Utf8), + SketchStatistic::TopK { .. } => return ranked_rows_schema(summary_input), }; if matches!( summary_input.asap(), @@ -277,11 +293,11 @@ impl ASAPOp { }) = child.asap() { { - use crate::post_asap::{ExactKind, SummaryInputExpr}; - use crate::pre_asap::AggIntent; + use crate::ir::operator::AggIntent; + use crate::ir::schema::{ExactKind, SummaryInputExpr}; let column = match &input.weight { SummaryInputExpr::Column(col) => Some( - crate::pre_asap::column_resolution::resolve_column_ref( + crate::ir::scalar::column_resolution::resolve_column_ref( col, &source.schema, ) @@ -304,7 +320,7 @@ impl ASAPOp { }; measure .map(|measure| { - super::NonASAPOp::Aggregate { + crate::ir::NonASAPOp::Aggregate { child: Rc::clone(source), reduction: reduction.clone(), measures: vec![measure], @@ -329,6 +345,9 @@ impl ASAPOp { for f in &mut out.fields { if let FieldDataType::ExactAggregate(kind, _) = &f.dtype { if let Some(result) = &value_result { + // The finalized value replaces the aggregate it + // realizes, so it takes that aggregate's column. + f.name = result.name.clone(); f.dtype = result.dtype.clone(); f.nullable = result.nullable; } else { @@ -340,8 +359,8 @@ impl ASAPOp { } MaintainPopulation { child, .. } => child.schema.clone(), EvaluatePopulation { child, evaluation } => { - use crate::post_asap::maintained_population::PopulationInput; - use crate::pre_asap::{AggIntent, GroupKeys}; + use crate::ir::operator::maintained_population::PopulationInput; + use crate::ir::operator::{AggIntent, GroupKeys}; let Some(MaintainPopulation { child: source, population, @@ -394,7 +413,7 @@ impl ASAPOp { PopulationStatistic::Average => AggIntent::Avg { col: column }, PopulationStatistic::TopK { .. } => unreachable!(), }; - super::NonASAPOp::Aggregate { + crate::ir::NonASAPOp::Aggregate { child: source.clone(), reduction: Reduction::Reduce(keys), measures: vec![measure], @@ -405,8 +424,11 @@ impl ASAPOp { .output_schema()? } } - SummaryMerge { .. } - | SummarySubtract { .. } + SummaryMerge { children } => { + self.validate_inputs()?; + children[0].schema.clone() + } + SummarySubtract { .. } | SummaryDelete { .. } | SummaryJoin { .. } | Extension { .. } => return Err(Self::unimplemented()), @@ -444,12 +466,52 @@ impl ASAPOp { } }; match self { + SummaryMerge { children } => { + let Some(first) = children.first() else { + return Err(SchemaDerivationError::InvalidScalarSignature( + "summary merge requires at least one state input".into(), + )); + }; + // Matching state parameters and grouping positions are necessary; + // matching names alone cannot prove two states compatible. + if first + .schema + .fields + .iter() + .filter(|field| !field.is_plain()) + .count() + != 1 + { + return Err(SchemaDerivationError::InvalidScalarSignature( + "summary merge requires exactly one state column".into(), + )); + } + for child in children { + needs_state(child, "SummaryMerge")?; + if child.schema != first.schema { + return Err(SchemaDerivationError::InvalidScalarSignature( + "summary merge inputs must have identical state and grouping schemas" + .into(), + )); + } + } + // Coverage records only time and population; what each state + // summarizes and how it is grouped come from the producers. + let update = first.summary_update(); + if update.is_none() || children.iter().any(|c| c.summary_update() != update) { + return Err(SchemaDerivationError::InvalidScalarSignature( + "summary merge inputs must share update expression and reduction".into(), + )); + } + self.merged_coverage()?; + Ok(()) + } SummaryEstimate { summary_input, query, } => { needs_state(summary_input, "SummaryEstimate")?; - use crate::post_asap::sketch::SketchCategory as C; + use crate::ir::schema::state_type::SketchCategory as C; let states: Vec<_> = summary_input .schema .fields @@ -521,13 +583,13 @@ impl ASAPOp { )); } fn check( - expr: &crate::post_asap::SummaryInputExpr, + expr: &crate::ir::schema::SummaryInputExpr, schema: &Schema, ) -> Result<(), SchemaDerivationError> { - use crate::post_asap::SummaryInputExpr; + use crate::ir::schema::SummaryInputExpr; match expr { SummaryInputExpr::Column(col) => { - crate::pre_asap::resolve_column_ref(col, schema).map_err(|e| { + crate::ir::scalar::resolve_column_ref(col, schema).map_err(|e| { SchemaDerivationError::InvalidScalarSignature(e.to_string()) })?; } @@ -566,9 +628,82 @@ impl ASAPOp { } } +/// A top-k readout returns the selected rows, as every executable top-k does +/// (exact Sort → Limit, `EvaluatePopulation`): the state's partition keys, the +/// ranked item's identity columns, and the item's estimated `value`. +fn ranked_rows_schema(state: &OperatorNode) -> Result { + use crate::ir::schema::state_type::{EntityIdentity, SummaryInputExpr}; + fn source(state: &OperatorNode) -> Option<(&Schema, &SummaryUpdate)> { + match state.asap()? { + ASAPOp::SummaryAgg { child, input, .. } => Some((&child.schema, input)), + ASAPOp::SummaryMerge { children } => source(children.first()?), + _ => None, + } + } + fn items( + item: &SummaryInputExpr, + source: &Schema, + fields: &mut Vec, + ) -> Result<(), SchemaDerivationError> { + match item { + SummaryInputExpr::Column(column) => { + let index = crate::ir::scalar::resolve_column_ref(column, source) + .map_err(|e| SchemaDerivationError::InvalidScalarSignature(e.to_string()))?; + fields.push(source.fields[index].clone()); + } + SummaryInputExpr::Tuple(parts) => { + for part in parts { + items(part, source, fields)?; + } + } + // A label set without its columns is read back as its encoded identity. + SummaryInputExpr::EntityIdentity(EntityIdentity::PromqlLabelSet { .. }) => { + fields.push(Field::plain( + crate::ir::schema::PROMQL_SERIES_IDENTITY, + DataType::Utf8, + false, + )) + } + SummaryInputExpr::Constant(_) => { + return Err(SchemaDerivationError::InvalidScalarSignature( + "top-k item identity cannot be a constant".into(), + )) + } + } + Ok(()) + } + let Some(( + source, + SummaryUpdate { + item: Some(item), .. + }, + )) = source(state) + else { + return Err(SchemaDerivationError::InvalidScalarSignature( + "top-k readout requires state keyed by an item identity".into(), + )); + }; + let mut fields: Vec<_> = state + .schema + .fields + .iter() + .filter(|f| f.is_plain()) + .cloned() + .collect(); + items(item, source, &mut fields)?; + let key = (0..fields.len()).collect(); + fields.push(Field::plain("value", DataType::Float64, false)); + Ok(Schema { + fields, + time_index: None, + unique_keys: vec![key], + closed: true, + }) +} + /// The plain value an exact accumulator finalizes to. -fn finalized_data_type(kind: &crate::post_asap::sketch::ExactKind) -> DataType { - use crate::post_asap::sketch::ExactKind; +fn finalized_data_type(kind: &crate::ir::schema::state_type::ExactKind) -> DataType { + use crate::ir::schema::state_type::ExactKind; match kind { ExactKind::Count => DataType::Int64, _ => DataType::Float64, @@ -579,11 +714,11 @@ fn finalized_data_type(kind: &crate::post_asap::sketch::ExactKind) -> DataType { /// of the relational input the state was built from. fn source_kind(node: &OperatorNode) -> OperatorResultKind { match &node.operator { - super::node::Operator::ASAP(op) => match op.children().first() { + crate::ir::operator::node::Operator::ASAP(op) => match op.children().first() { Some(child) => source_kind(child), None => OperatorResultKind::Relation, }, - super::node::Operator::NonASAP(_) => match node.result_kind { + crate::ir::operator::node::Operator::NonASAP(_) => match node.result_kind { OperatorResultKind::RangeVector => OperatorResultKind::InstantVector, OperatorResultKind::State => OperatorResultKind::Relation, other => other, diff --git a/crates/types/src/post_asap/maintained_population.rs b/crates/types/src/ir/operator/maintained_population.rs similarity index 57% rename from crates/types/src/post_asap/maintained_population.rs rename to crates/types/src/ir/operator/maintained_population.rs index 73ddd38ba..2d2eebc30 100644 --- a/crates/types/src/post_asap/maintained_population.rs +++ b/crates/types/src/ir/operator/maintained_population.rs @@ -1,4 +1,4 @@ -//! Language-independent maintained populations and their readouts. +//! Language-independent maintained populations and their evaluations. //! Resource limits, ingestion placement and data structures belong to the executor. use serde::{Deserialize, Serialize}; @@ -35,135 +35,32 @@ pub enum PopulationStatistic { Average, } -impl CurrentSeriesInput { - /// Verify the named contract against the canonical maintenance input. - pub fn matches_input(&self, input: &crate::pre_asap::QueryExpr) -> bool { - use crate::pre_asap::{CompareOpKind, DataType, QueryExpr, ScalarValue, Source}; - // PromQL instant selectors carry an ingestion-interval `TimeRange` as - // their input scope. The population must use the same expiry horizon; - // shifted and otherwise transformed inputs still fail below. - let input = match input { - QueryExpr::TimeRange { range, child } - if self.lookback_ms > 0 - && *range == std::time::Duration::from_millis(self.lookback_ms) => - { - child.as_ref() - } - QueryExpr::TimeRange { .. } => return false, - other if self.lookback_ms == 300_000 => other, - _ => return false, - }; - let QueryExpr::Scan { - source: Source::TimeSeries { metric }, - predicates, - schema, - } = input - else { - return false; - }; - if self.metric.is_empty() - || *metric != self.metric - || (schema.closed && !schema.has_promql_series_identity()) - || schema.time_index.is_none() - { - return false; - } - if self.grouping.iter().any(|label| { - !schema - .fields - .iter() - .any(|c| c.name == *label && c.dtype == DataType::Utf8) - }) { - return false; - } - let mut matchers = Vec::new(); - for predicate in predicates { - let QueryExpr::Compare { left, op, right } = predicate.0.as_ref() else { - return false; - }; - let (QueryExpr::Column(col), QueryExpr::Literal(ScalarValue::Utf8(value))) = - (left.as_ref(), right.as_ref()) - else { - return false; - }; - let Some(column) = schema.fields.get(*col) else { - return false; - }; - if column.dtype != DataType::Utf8 { - return false; - } - let operation = match op { - CompareOpKind::Eq => CurrentSeriesMatch::Equal, - CompareOpKind::Ne => CurrentSeriesMatch::NotEqual, - CompareOpKind::Regex => CurrentSeriesMatch::Regex, - CompareOpKind::NotRegex => CurrentSeriesMatch::NotRegex, - _ => return false, - }; - matchers.push(CurrentSeriesMatcher { - label: column.name.clone(), - value: value.clone(), - operation, - }); - } - matchers.sort(); - matchers.dedup(); - self.matchers == matchers && self.grouping.windows(2).all(|w| w[0] < w[1]) - } -} - /// Membership is part of state identity. Table rows must never acquire implicit /// latest-per-series selection, stale markers, or a PromQL lookback. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum PopulationInput { +pub enum PopulationInput { CurrentSeries(CurrentSeriesInput), Rows { input: std::rc::Rc, value_column: usize, - grouping: crate::pre_asap::GroupKeys, + grouping: crate::ir::operator::GroupKeys, }, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct MaintainedPopulation { +pub struct MaintainedPopulation { pub input: PopulationInput, pub max_k: usize, pub quantiles: bool, } -impl MaintainedPopulation { - pub fn matches_input(&self, input: &crate::pre_asap::QueryExpr) -> bool { - match &self.input { - PopulationInput::CurrentSeries(spec) => spec.matches_input(input), - PopulationInput::Rows { - input: expected, - value_column, - grouping, - } => { - use crate::pre_asap::{DataType, QueryExpr, Source}; - expected.as_ref() == input - && matches!(input, QueryExpr::Scan { source: Source::Table { .. }, schema, .. } - if schema.closed && schema.fields.get(*value_column).is_some_and(|c| c.dtype == DataType::Float64 && !c.nullable) - && !grouping.is_without() && grouping.keys().iter().all(|k| *k < schema.fields.len())) - } - } - } - - pub fn supports(&self, readout: &PopulationStatistic) -> bool { - match readout { - PopulationStatistic::Quantile { q } => self.quantiles && q.is_finite(), - PopulationStatistic::TopK { k } => *k <= self.max_k, - PopulationStatistic::Sum - | PopulationStatistic::Count - | PopulationStatistic::Average => true, - } - } -} - impl CurrentSeriesInput { /// Verify the named contract against the canonical maintenance input. pub fn matches_node(&self, input: &crate::ir::OperatorNode) -> bool { + use crate::ir::operator::Source; + use crate::ir::scalar::{CompareOpKind, ScalarValue}; + use crate::ir::schema::DataType; use crate::ir::{NonASAPOp, Operator, ScalarExpr, TimeRangeKind}; - use crate::pre_asap::{CompareOpKind, DataType, ScalarValue, Source}; let input = match &input.operator { Operator::NonASAP(NonASAPOp::TimeRange { range, @@ -242,8 +139,9 @@ impl CurrentSeriesInput { impl MaintainedPopulation { /// Whether `input` is the maintenance input this population declares. pub fn matches_node(&self, input: &crate::ir::OperatorNode) -> bool { + use crate::ir::operator::Source; + use crate::ir::schema::DataType; use crate::ir::{NonASAPOp, Operator}; - use crate::pre_asap::{DataType, Source}; match &self.input { PopulationInput::CurrentSeries(spec) => spec.matches_node(input), PopulationInput::Rows { diff --git a/crates/types/src/ir/operator/mod.rs b/crates/types/src/ir/operator/mod.rs new file mode 100644 index 000000000..09a2cbf2b --- /dev/null +++ b/crates/types/src/ir/operator/mod.rs @@ -0,0 +1,19 @@ +//! #511 §1: the operators of the unified DAG and their parameters. + +pub mod agg_intent; +pub mod asap; +pub mod maintained_population; +pub mod node; +pub mod non_asap; +pub mod operator_properties; + +pub use agg_intent::{ + agg_accuracy, agg_is_exact, agg_is_mergeable, default_cardinality, default_quantile, AggIntent, + MathFunc, TimeFunc, +}; +pub use operator_properties::{ + AtModifier, BinaryOpKind, ColState, ConcatDiscriminatorKey, DataModel, GroupKeys, GroupSide, + InfoMatcher, JoinKind, PromQLVectorSetOpKind, Reduction, RelationalSetOpKind, SampleKind, + Source, TimeShift, VectorGrouping, VectorMatch, VectorMatchKind, WindowFrame, WindowFrameBound, + WindowFrameOffset, WindowFrameUnits, WindowFuncKind, +}; diff --git a/crates/types/src/ir/node.rs b/crates/types/src/ir/operator/node.rs similarity index 79% rename from crates/types/src/ir/node.rs rename to crates/types/src/ir/operator/node.rs index d68f5aa03..d368d4a37 100644 --- a/crates/types/src/ir/node.rs +++ b/crates/types/src/ir/operator/node.rs @@ -8,12 +8,15 @@ use std::rc::Rc; use serde::{Deserialize, Serialize}; -use super::asap::ASAPOp; -use super::non_asap::NonASAPOp; +use crate::ir::operator::asap::ASAPOp; +use crate::ir::operator::non_asap::NonASAPOp; +use crate::ir::operator::operator_properties::Reduction; +use crate::ir::properties::execution::ExecutionTiming; +use crate::ir::properties::guarantee::ResultGuarantee; +use crate::ir::properties::summary_coverage::{CoverageError, SummaryCoverage}; +use crate::ir::schema::Schema; +use crate::ir::schema::SummaryUpdate; use crate::ir::SchemaDerivationError; -use crate::post_asap::execution_data_state::ExecutionTiming; -use crate::post_asap::guarantee::ResultGuarantee; -use crate::pre_asap::schema::Schema; /// The output category of an operator, derived from the operation and its /// inputs. Matching column schemas do not make categories interchangeable. @@ -86,7 +89,7 @@ impl Operator { /// `schema` and `result_kind` are derived from `operator` and its children /// at construction and retained. `guarantee` is `None` until accuracy /// assessment establishes one (`None` never means exact). `timing` is `None` -/// until a lifecycle assignment is applied; export rejects an executable +/// until a materialization assignment is applied; export rejects an executable /// node without one. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct OperatorNode { @@ -95,6 +98,8 @@ pub struct OperatorNode { pub schema: Schema, pub guarantee: Option, pub timing: Option, + #[serde(default)] + pub coverage: Option, } impl OperatorNode { @@ -103,7 +108,11 @@ impl OperatorNode { /// ASAP operator, ...). pub fn new(operator: Operator) -> Result { let schema = operator.output_schema()?; - Ok(Self::with_schema(operator, schema)) + let mut node = Self::with_schema(operator, schema); + if let Some(op @ ASAPOp::SummaryMerge { .. }) = node.asap() { + node.coverage = Some(op.merged_coverage()?); + } + Ok(node) } /// Build a node with caller-supplied output names and qualifiers. For @@ -117,6 +126,7 @@ impl OperatorNode { schema, guarantee: None, timing: None, + coverage: None, } } @@ -137,6 +147,40 @@ impl OperatorNode { self } + /// Attach caller-established coverage. Required on summary nodes; see + /// [`Self::requires_coverage`]. + pub fn with_coverage( + mut self, + coverage: SummaryCoverage, + ) -> Result { + coverage.validate()?; + if self.result_kind != OperatorResultKind::State { + return Err(CoverageError::NotState.into()); + } + self.coverage = Some(coverage); + Ok(self) + } + + /// What a summary state is updated with and how it is grouped: the + /// `SummaryAgg` fields, or those shared by a `SummaryMerge`'s inputs. + pub fn summary_update(&self) -> Option<(&SummaryUpdate, &Reduction)> { + match self.asap()? { + ASAPOp::SummaryAgg { + input, reduction, .. + } => Some((input, reduction)), + ASAPOp::SummaryMerge { children } => children.first()?.summary_update(), + _ => None, + } + } + + /// Summary nodes whose state can be composed must declare coverage. + pub fn requires_coverage(&self) -> bool { + matches!( + self.asap(), + Some(ASAPOp::SummaryAgg { .. } | ASAPOp::SummaryMerge { .. }) + ) + } + pub fn non_asap(&self) -> Option<&NonASAPOp> { match &self.operator { Operator::NonASAP(op) => Some(op), @@ -249,9 +293,9 @@ impl OperatorNode { "execution timing is unassigned".into(), ) })?; - if timing == crate::post_asap::ExecutionTiming::IngestionTime + if timing == crate::ir::properties::ExecutionTiming::IngestionTime && node.children().iter().any(|child| { - child.timing != Some(crate::post_asap::ExecutionTiming::IngestionTime) + child.timing != Some(crate::ir::properties::ExecutionTiming::IngestionTime) }) { return Err(SchemaDerivationError::InvalidScalarSignature( @@ -271,10 +315,9 @@ impl OperatorNode { pub fn validate_structure(self: &Rc) -> Result<(), SchemaDerivationError> { for node in Self::reachable(self) { if node.schema.time_index.is_some_and(|i| { - node.schema - .fields - .get(i) - .is_none_or(|f| f.plain_dtype() != Some(&crate::pre_asap::DataType::Timestamp)) + node.schema.fields.get(i).is_none_or(|f| { + f.plain_dtype() != Some(&crate::ir::schema::DataType::Timestamp) + }) }) || node .schema .unique_keys @@ -286,6 +329,18 @@ impl OperatorNode { "invalid time or identity column in schema".into(), )); } + match &node.coverage { + Some(coverage) => { + (*node.as_ref()).clone().with_coverage(coverage.clone())?; + } + None if node.requires_coverage() => return Err(CoverageError::Missing.into()), + None => {} + } + if let Some(op @ ASAPOp::SummaryMerge { .. }) = node.asap() { + if node.coverage.as_ref() != Some(&op.merged_coverage()?) { + return Err(CoverageError::MergeOutputMismatch.into()); + } + } node.operator.validate_inputs()?; if node.result_kind != node.operator.output_kind() { return Err(SchemaDerivationError::InvalidScalarSignature( diff --git a/crates/types/src/ir/non_asap.rs b/crates/types/src/ir/operator/non_asap.rs similarity index 98% rename from crates/types/src/ir/non_asap.rs rename to crates/types/src/ir/operator/non_asap.rs index 0daf3a0ce..3b71c3bfe 100644 --- a/crates/types/src/ir/non_asap.rs +++ b/crates/types/src/ir/operator/non_asap.rs @@ -6,16 +6,16 @@ use std::time::Duration; use serde::{Deserialize, Serialize}; -use super::node::{OperatorNode, OperatorResultKind}; -use super::scalar::{Predicate, ProjectItem, ScalarExpr, SortKey}; -use crate::ir::aggregate_schema::aggregate_output_schema; -use crate::ir::operator_properties::{ +use crate::ir::operator::agg_intent::AggIntent; +use crate::ir::operator::node::{OperatorNode, OperatorResultKind}; +use crate::ir::operator::operator_properties::{ BinaryOpKind, ConcatDiscriminatorKey, GroupKeys, InfoMatcher, JoinKind, Reduction, RelationalSetOpKind, SampleKind, Source, TimeShift, VectorMatch, WindowFrame, WindowFuncKind, }; +use crate::ir::scalar::{Predicate, ProjectItem, ScalarExpr, SortKey}; +use crate::ir::schema::aggregate_schema::aggregate_output_schema; +use crate::ir::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; use crate::ir::SchemaDerivationError; -use crate::pre_asap::agg_intent::AggIntent; -use crate::pre_asap::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; /// All semantics owned by a binary operator. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] @@ -761,8 +761,8 @@ impl NonASAPOp { .and_then(|m| m.grouping.as_ref()); let right_rows = matches!( operator.kind, - BinaryOpKind::Set(crate::pre_asap::PromQLVectorSetOpKind::Or) - ) || matches!(grouping, Some(g) if g.side == crate::pre_asap::GroupSide::Right); + BinaryOpKind::Set(crate::ir::operator::PromQLVectorSetOpKind::Or) + ) || matches!(grouping, Some(g) if g.side == crate::ir::operator::GroupSide::Right); let mut additions = Vec::new(); if right_rows { additions.extend( @@ -1087,11 +1087,11 @@ pub fn any_measure_filtered(filters: &[Option]) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::ir::operator_properties::{ + use crate::ir::operator::operator_properties::{ AtModifier, VectorMatchKind, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, }; use crate::ir::scalar::ExprSemantics; - use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind, ScalarValue}; + use crate::ir::scalar::{ArithmeticOpKind, CompareOpKind, ScalarValue}; fn col(name: &str, dtype: DataType, nullable: bool) -> Field { Field::plain(name, dtype, nullable) diff --git a/crates/types/src/ir/operator/operator_properties.rs b/crates/types/src/ir/operator/operator_properties.rs new file mode 100644 index 000000000..e5a04a720 --- /dev/null +++ b/crates/types/src/ir/operator/operator_properties.rs @@ -0,0 +1,565 @@ +//! Supporting parameter types used inside operator payloads. +//! +//! For example, `Aggregate.by` uses [`GroupKeys`], a join chooses [`JoinKind`], +//! and a SQL window carries [`WindowFrame`]. These types describe what an +//! operator does. Derived node metadata (schema, guarantee, timing) lives on +//! [`crate::ir::OperatorNode`], not in this module. +use crate::ir::scalar::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; +use crate::ir::schema::ColumnId; +use serde::{Deserialize, Serialize}; +/// The column-reference type an operator parameter is generic over: +/// positional [`ColumnId`] once bound, name-based [`ColumnRef`] before. +pub trait ColState: + Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de> +{ +} + +impl ColState for ColumnId {} + +impl ColState for ColumnRef {} + +// ── Leaf / supporting types ─────────────────────────────────────────────────── + +/// Positional grouping keys, shared by every "operate per group" operator: +/// `Aggregate.by` (reduce per group), `Sort.partition_by` (rank per group — +/// including generic `topk`/`bottomk`), and `SQLWindowFunc.partition_by` (window +/// per group). One spelling so grouping has a single home to evolve. Empty +/// (and `by`) = no grouping (a global operation). +/// +/// Heavy-hitter `AggIntent::TopK` carries its grouping here too, via the +/// enclosing `Aggregate.by` (issue #13) — so reduce, rank, and window groupings +/// all share this one type. +/// +/// ## `by` vs `without` (issue #39) +/// +/// The stored [`keys`](Self::keys) are **kept** labels for `by(...)` and +/// **excluded** labels for `without(...)`. PromQL's `without(labels)` groups by +/// every label *except* those listed; the complement can't be enumerated at +/// lowering time under an open (usage-derived) schema, so it is deferred to the +/// runtime — the excluded positions are stored, the kept set stays open. Only +/// `Aggregate` ever produces the `without` form; `Sort` / `SQLWindowFunc` / +/// `PromqlSeriesSample` groupings are always `by`. +/// +/// Serialises as a bare array for the (overwhelmingly common) `by` case — +/// wire-compatible with the `Vec` this field held before — and as +/// `{"without": [...]}` for the exclusion case. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct GroupKeys { + keys: Vec, + without: bool, +} + +// Not `#[derive(Default)]`: derive would add a `C: Default` bound, but an +// empty key set needs nothing from `C` — `ColumnRef` has no meaningful +// default anyway. +impl Default for GroupKeys { + fn default() -> Self { + Self { + keys: Vec::new(), + without: false, + } + } +} + +impl GroupKeys { + /// An empty key set — a global (ungrouped) operation. + pub fn none() -> Self { + Self::default() + } + /// `by(keys)` — group by exactly these columns. + pub fn by(keys: Vec) -> Self { + Self { + keys, + without: false, + } + } + /// `without(keys)` — group by every label *except* these (issue #39). The + /// kept set is runtime-resolved; only the excluded positions are stored. + pub fn without(keys: Vec) -> Self { + Self { + keys, + without: true, + } + } + /// Whether this is a `without(...)` exclusion grouping. + pub fn is_without(&self) -> bool { + self.without + } + /// The named keys — kept labels for `by`, excluded labels for `without`. + pub fn keys(&self) -> &[C] { + &self.keys + } +} + +impl std::ops::Deref for GroupKeys { + type Target = [C]; + fn deref(&self) -> &Self::Target { + &self.keys + } +} + +impl From> for GroupKeys { + fn from(keys: Vec) -> Self { + Self::by(keys) + } +} + +impl FromIterator for GroupKeys { + fn from_iter>(iter: I) -> Self { + Self::by(iter.into_iter().collect()) + } +} + +impl<'a, C> IntoIterator for &'a GroupKeys { + type Item = &'a C; + type IntoIter = std::slice::Iter<'a, C>; + fn into_iter(self) -> Self::IntoIter { + self.keys.iter() + } +} + +/// Compare directly against a `Vec` so call sites and tests can keep +/// writing `keys == vec![..]` / `assert_eq!(keys, &vec![..])`. A `without` +/// grouping never equals a bare `by` list. +impl PartialEq> for GroupKeys { + fn eq(&self, other: &Vec) -> bool { + !self.without && &self.keys == other + } +} + +/// (De)serialise as a bare array for `by`, or `{"without": [...]}` for the +/// exclusion form — keeping the `by` wire format identical to the old newtype. +/// Borrowed for `Serialize` (no `C: Clone` needed to write one out), owned for +/// `Deserialize` (there's nothing to borrow from). +#[derive(Serialize)] +#[serde(untagged)] +enum GroupKeysReprRef<'a, C> { + By(&'a [C]), + Without { without: &'a [C] }, +} + +#[derive(Deserialize)] +#[serde(untagged)] +enum GroupKeysRepr { + By(Vec), + Without { without: Vec }, +} + +impl Serialize for GroupKeys { + fn serialize(&self, serializer: S) -> Result { + if self.without { + GroupKeysReprRef::Without { + without: self.keys.as_slice(), + } + .serialize(serializer) + } else { + GroupKeysReprRef::By(self.keys.as_slice()).serialize(serializer) + } + } +} + +impl<'de, C: Deserialize<'de>> Deserialize<'de> for GroupKeys { + fn deserialize>(deserializer: D) -> Result { + Ok(match GroupKeysRepr::deserialize(deserializer)? { + GroupKeysRepr::By(keys) => Self::by(keys), + GroupKeysRepr::Without { without } => Self::without(without), + }) + } +} + +/// Which data model a `Source` / `AggIntent` operates over. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum DataModel { + TimeSeries, + Tabular, + Any, +} + +/// The leaf data source of a `Scan`. The schema itself rides on the +/// `Scan.schema` field (SchemaResolver-built); `Source` carries only the leaf's +/// identity. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum Source { + /// Time-series leaf — PromQL / DC lifecycle. Produces `(ts, value, *labels)`. + TimeSeries { metric: String }, + /// Tabular leaf — asap-fusion / future OLAP. Columns ride on `Scan.schema`. + Table { table_ref: String }, +} + +impl Source { + pub fn data_model(&self) -> DataModel { + match self { + Source::TimeSeries { .. } => DataModel::TimeSeries, + Source::Table { .. } => DataModel::Tabular, + } + } +} + +/// Operator on the query-level `BinaryOp` node. Reuses the scalar IR's +/// [`ArithmeticOpKind`] / [`CompareOpKind`] so every arithmetic/comparison +/// operator has exactly one representation (and one `Display`) across the IR. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum BinaryOpKind { + /// Arithmetic — `Add/Sub/Mul/Div/Mod` (shared with `ScalarExpr::Arithmetic`). + Arithmetic(ArithmeticOpKind), + /// Comparison — `Eq/Ne/Lt/Le/Gt/Ge` + `Like/ILike/Regex` family (shared + /// with `ScalarExpr::Compare`). PromQL keeps the matched series whose + /// comparison holds. + Compare(CompareOpKind), + /// PromQL comparison with the `bool` modifier: every matched series + /// yields 1 or 0 and loses its metric name. A separate variant, not a + /// flag, because only comparisons take `bool`. + CompareBool(CompareOpKind), + /// PromQL vector-set operation. + Set(PromQLVectorSetOpKind), +} + +impl std::fmt::Display for BinaryOpKind { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + BinaryOpKind::Arithmetic(op) => write!(f, "{op}"), + BinaryOpKind::Compare(op) => write!(f, "{op}"), + BinaryOpKind::CompareBool(op) => write!(f, "{op} bool"), + BinaryOpKind::Set(PromQLVectorSetOpKind::And) => f.write_str("AND"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Or) => f.write_str("OR"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) => f.write_str("unless"), + } + } +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum JoinKind { + Inner, + Left, + Right, + Full, + Cross, + /// Left semi-join — each left row that has **at least one** match, once. + /// `WHERE c IN (SELECT …)` / `WHERE EXISTS (…)` (issue #111). + /// + /// Output schema is the **left's alone**; the right side is a filter, not a + /// source of columns. The join predicate still resolves against the + /// concatenated `left ++ right` schema — its scope is deliberately wider + /// than the node's output. + Semi, + /// Left anti-join — each left row with **no** match. `WHERE NOT EXISTS (…)`. + /// Same schema rule as [`JoinKind::Semi`]. + /// + /// Note this is *not* `NOT IN (SELECT …)`: under SQL's three-valued logic a + /// NULL on the right makes `NOT IN` yield no rows at all, where an anti-join + /// yields every left row. The SQL front end rejects `NOT IN (subquery)` + /// rather than lower it here. + Anti, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum RelationalSetOpKind { + Union, + Intersect, + Except, +} + +/// PromQL vector-set operator used by [`BinaryOpKind::Set`]. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum PromQLVectorSetOpKind { + And, + Or, + Unless, +} + +/// SQL analytic window function (`fn(...) OVER (…)`). Distinct from a streaming +/// time `Window`: this is an analytic frame over already-materialised rows. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFuncKind { + RowNumber, + Rank, + DenseRank, + Lag, + Lead, + /// ClickHouse `lagInFrame`/`leadInFrame`: unlike [`Lag`](Self::Lag)/[`Lead`](Self::Lead), + /// these respect the window frame bounds (NULL/default past the frame edge) + /// rather than reaching arbitrarily far back/forward. Kept as distinct + /// variants so the frame clause is never silently discarded by conflating + /// them with `Lag`/`Lead` (#267). `WindowFuncKind` still has no frame + /// representation, so today these lower and behave exactly like + /// `Lag`/`Lead` — the tag is correct, the frame-respecting behavior isn't + /// implemented yet. See #231 for modeling window frames properly. + LagInFrame, + LeadInFrame, + FirstValue, + LastValue, + /// `NTH_VALUE(expr, n)` — `n` is resolved from the (literal) 2nd argument. + NthValue(Option), + Sum, + Avg, + Count, + Min, + Max, +} + +/// A window's frame-spec (`ROWS`/`RANGE BETWEEN … AND …`) — which rows around +/// the current one an analytic window function reads. `GROUPS` is rejected at +/// lowering time (issue #268): every SQL corpus in this repo uses only `ROWS`, +/// and nothing downstream interprets frame semantics yet, so it isn't worth +/// modelling untested. +/// +/// Meaningless (but harmless) on the rank-only and navigation functions +/// (`ROW_NUMBER`/`RANK`/`DENSE_RANK`/`LAG`/`LEAD`), which ignore the frame per +/// SQL semantics — DataFusion still attaches one, stored here verbatim. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WindowFrame { + pub units: WindowFrameUnits, + pub start_bound: WindowFrameBound, + pub end_bound: WindowFrameBound, +} + +/// A finite window-frame displacement. Intervals are normalized to Arrow's +/// month/day/nanosecond representation so SQL `RANGE INTERVAL ...` bounds +/// survive lowering without leaking DataFusion types into the canonical IR. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFrameOffset { + Scalar(ScalarValue), + Interval { + months: i32, + days: i32, + nanoseconds: i64, + }, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFrameUnits { + /// Boundaries count physical rows: `ROWS BETWEEN 2 PRECEDING AND CURRENT ROW`. + Rows, + /// Boundaries count by value-distance on the (single) `ORDER BY` column: + /// `RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW`. + Range, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum WindowFrameBound { + /// `UNBOUNDED PRECEDING` is + /// `Preceding(WindowFrameOffset::Scalar(ScalarValue::Null))`. + Preceding(WindowFrameOffset), + CurrentRow, + /// `UNBOUNDED FOLLOWING` is + /// `Following(WindowFrameOffset::Scalar(ScalarValue::Null))`. + Following(WindowFrameOffset), +} + +/// A symbolic label matcher on the **info metric** side of an +/// [`crate::ir::NonASAPOp::PromqlInfoEnrich`] (issue #84). Unlike a `Scan` predicate it is not +/// resolved positionally — it references the info metric's labels (`__name__` +/// picks the metric, the rest constrain data labels), which aren't in the input +/// vector's schema; the post-ASAP realization pass applies it against the info metric. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct InfoMatcher { + pub label: String, + /// One of `Eq` / `Ne` / `Regex` / `NotRegex` (PromQL `=`/`!=`/`=~`/`!~`). + pub op: CompareOpKind, + pub value: String, +} + +/// Series-sampling selection mode (PromQL `limitk` / `limit_ratio`, issue #86). +/// A [`crate::ir::NonASAPOp::PromqlSeriesSample`] keeps a *subset of whole series*, unchanged — it does +/// not rank or reduce, so it is distinct from `TopK` and from `Sort → Limit`. +#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum SampleKind { + /// `limitk(k, v)` — up to `k` series per group. Which series survive is + /// deterministic across evaluations but otherwise unspecified (no ordering). + LimitK(usize), + /// `limit_ratio(r, v)` — a deterministic `r`-fraction of series per group. + /// `r ∈ [-1, 1]`; a negative `r` selects the complementary fraction. + LimitRatio(f64), +} + +/// PromQL vector-match modifier (`on`/`ignoring` + `group_left`/`group_right`). +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct VectorMatch { + pub kind: VectorMatchKind, + pub labels: Vec, + pub grouping: Option, +} + +/// PromQL `@` modifier — pins a selector's evaluation time to an anchor instead +/// of the query evaluation time (issue #40). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum AtModifier { + /// `@ start()` — the query range's start instant. + Start, + /// `@ end()` — the query range's end instant. + End, + /// `@ ` — an absolute instant, milliseconds since the Unix epoch (may be + /// negative). PromQL writes the timestamp in seconds; the front end scales it. + Timestamp(i64), +} + +/// PromQL per-selector **time-shift** modifiers — `offset` and `@` (issue #40). +/// Neither changes a selector's *schema*; both move *when* it is evaluated, so +/// the shift is a pass-through wrapper ([`crate::ir::NonASAPOp::TimeShift`]) over the +/// selector rather than a new leaf shape. The runtime resolves the anchor and +/// applies the offset. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] +pub struct TimeShift { + /// `offset ` as signed milliseconds — a positive value shifts the + /// lookback *back* in time (`offset 5m`), a negative value shifts it + /// *forward* (`offset -5m`). `0` = no offset. + pub offset_ms: i64, + /// `@` anchor; `None` = evaluate at the query time. + pub at: Option, +} + +impl TimeShift { + /// Whether this shift is the identity (no `offset`, no `@`) — the state of + /// every selector that carries neither modifier. + pub fn is_identity(&self) -> bool { + self.offset_ms == 0 && self.at.is_none() + } +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum VectorMatchKind { + On, + Ignoring, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct VectorGrouping { + pub side: GroupSide, + pub labels: Vec, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub enum GroupSide { + Left, + Right, +} + +// ── Intent algebra IR ──────────────────────────────────────────────────────── + +/// What kind of computation an `Aggregate` node performs — orthogonal to +/// *which* columns it groups by (that's still [`GroupKeys`], inside +/// `Reduce`). Explicit, decided once by whichever pass constructs the node +/// (structural, at front-end lowering time), rather than inferred downstream from +/// whether a grouping-key list happens to be empty or from a neighboring +/// node's shape. See design proposal #165. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum Reduction { + /// Collapses input rows via `by` — `by`/`without` semantics are exactly + /// [`GroupKeys`]'s. May still collapse every row into one (an empty, + /// non-`without` `by`) — that's a genuine reduction with zero grouping + /// columns, not "no grouping concept." + Reduce(GroupKeys), + /// No grouping concept at all: preserves one output row per input + /// entity (e.g. a per-series windowed computation with no `by(...)` + /// clause to begin with, because there's no aggregation operator here + /// for such a clause to attach to). Never merges across entities, and + /// never collapses an entity's own row structure (e.g. a time axis) — + /// unlike `Reduce(GroupKeys::without(vec![]))` ("group by every + /// label"), which is still a genuine reduction and does collapse it. + PerEntity, +} + +impl Reduction { + /// Shorthand for the common case — group by these (possibly empty) + /// keys, kept rather than excluded. + pub fn by(keys: Vec) -> Self { + Self::Reduce(GroupKeys::by(keys)) + } + + /// The grouping keys, if this is a genuine reduction — `None` for + /// `PerEntity`, which has no grouping-keys concept to report. + pub fn group_keys(&self) -> Option<&GroupKeys> { + match self { + Self::Reduce(by) => Some(by), + Self::PerEntity => None, + } + } + + /// The grouping keys, panicking if this is `PerEntity` — for call sites + /// (tests, mostly) that already know, from the shape they built or are + /// asserting on, that this must be a genuine reduction. Prefer + /// [`group_keys`](Self::group_keys) wherever the caller can't assume that. + pub fn expect_reduce(&self) -> &GroupKeys { + match self { + Self::Reduce(by) => by, + Self::PerEntity => panic!("expected Reduction::Reduce, got PerEntity"), + } + } +} + +/// A caller-proven compound unique key for a [`crate::ir::NonASAPOp::Concat`] (issue +/// #228) — built only via [`ConcatDiscriminatorKey::new`], never by naming `discriminator` directly +/// in a struct literal (both fields are private): from *other Rust code*, +/// the only way to end up with one of these is to hand over a specific +/// column as the discriminator, by name, at the call site. +/// +/// Caveat: this is a Rust-API-level guarantee, not a data-level one. The +/// derived `Deserialize` impl below builds a `ConcatDiscriminatorKey` +/// directly from field values, bypassing `new()`. Deserialization is therefore +/// equivalent to a caller supplying the assertion directly; it does not prove +/// either fact below. An external boundary accepting IR data must +/// reject this field or validate both obligations before treating it as +/// uniqueness evidence. +/// +/// # Soundness +/// +/// `Concat`'s default (see its own doc) is to drop `unique_keys` +/// unconditionally, because a key unique **within** one branch is not unique +/// **across** the concatenation unless the branches' value sets for that key +/// are provably disjoint — nothing about matching schemas or matching +/// per-branch keys establishes that on its own. Two different branches can +/// trivially emit the same `inner_key` value (e.g. two PromQL +/// `histogram_quantiles` branches keyed on `(host, le)` can both produce a +/// `(host, le)` pair for different φ). +/// +/// Prepending `discriminator` restores a compound key only when two facts +/// hold: `inner_key` uniquely identifies rows **within every branch**, and +/// `discriminator`'s value is **guaranteed to differ between branches** — a +/// literal the producer just tagged the branch with (PromQL φ riding along via +/// [`crate::ir::NonASAPOp::PromqlRelabel`], a Postgres-style synthetic `GROUPING()` id +/// for `ROLLUP`/`CUBE`, …), never something inferred structurally from the +/// branches' own data — then `discriminator` alone partitions rows into +/// disjoint sets independent of what the branches actually contain, so +/// `(discriminator, inner_key)` is sound even when otherwise-identical +/// `inner_key` values occur in different branches. Neither fact is verified +/// here; both are part of the caller-proven claim. +/// +/// This is a **caller-proven claim, not something `Concat` can verify**: +/// nothing stops a caller from asserting a discriminator that in fact +/// repeats across branches, in which case the resulting `unique_keys` claim +/// is simply wrong — `output_schema` trusts it without checking. The +/// obligation is on the constructor call site, exactly as it is on +/// [`crate::ir::NonASAPOp::Dedup`]'s `cols` or any other unverified `unique_keys` +/// producer in this module. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] +pub struct ConcatDiscriminatorKey { + discriminator: C, + inner_key: Vec, +} + +impl ConcatDiscriminatorKey { + /// The only constructor — `discriminator` must be named explicitly by + /// the caller. See the type's doc for the soundness obligation this + /// puts on that caller. + pub fn new(discriminator: C, inner_key: Vec) -> Self { + Self { + discriminator, + inner_key, + } + } + + pub fn discriminator(&self) -> &C { + &self.discriminator + } + + pub fn inner_key(&self) -> &[C] { + &self.inner_key + } +} diff --git a/crates/types/src/ir/operator_properties.rs b/crates/types/src/ir/operator_properties.rs deleted file mode 100644 index 4737e6278..000000000 --- a/crates/types/src/ir/operator_properties.rs +++ /dev/null @@ -1,8 +0,0 @@ -//! Operator parameters shared with the existing dag during migration. -//! Definitions move here when legacy dag consumers are removed. -pub use crate::pre_asap::query_expr::{ - AtModifier, BinaryOpKind, ColState, ConcatDiscriminatorKey, DataModel, GroupKeys, GroupSide, - InfoMatcher, JoinKind, PromQLVectorSetOpKind, Reduction, RelationalSetOpKind, SampleKind, - Source, TimeShift, VectorGrouping, VectorMatch, VectorMatchKind, WindowFrame, WindowFrameBound, - WindowFrameOffset, WindowFrameUnits, WindowFuncKind, -}; diff --git a/crates/types/src/ir/physical_export.rs b/crates/types/src/ir/physical_export.rs new file mode 100644 index 000000000..3e8caa36a --- /dev/null +++ b/crates/types/src/ir/physical_export.rs @@ -0,0 +1,452 @@ +//! Physical ASAP DAG transport (planner-layering stage 2 output). +//! +//! Same operator payloads as the logical export, plus the execution timing +//! (data state) of every node and edge. The input must already be timed +//! ([`crate::ir::properties::timing::apply_materialization_timings`]); export reads each node's +//! timing and does not re-run data-state validation. + +use std::collections::{BTreeMap, HashMap, HashSet}; +use std::rc::Rc; + +use serde::{Deserialize, Serialize}; +use thiserror::Error; + +use crate::ir::operator::asap::ASAPOp; +use crate::ir::operator::node::{Operator, OperatorNode}; +use crate::ir::properties::execution::{ + ExecutionDataState, ExecutionDataStateError, ExecutionTiming, +}; +use crate::ir::properties::guarantee::ResultGuarantee; +use crate::ir::properties::timing::data_state; +use crate::ir::schema::{FieldDataType, Schema}; +use crate::ir::wire::{grouping_compatibility, input_edges, payload_of}; +use crate::ir::wire::{EdgeRole, GroupingEdgeCompatibility, LogicalASAPOperatorPayload}; + +pub const PHYSICAL_ASAP_DAG_WIRE_VERSION: u32 = 8; + +/// Operator payloads are shared with the logical export. +pub type PhysicalASAPOperatorPayload = LogicalASAPOperatorPayload; + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum WindowEdgeCompatibility { + /// Physical lowering must prove equal pane/query phase or install an + /// exact boundary residual. The logical DAG alone cannot make that claim. + #[serde(rename = "RequiresAlignedPanePhaseOrExactBoundaryResidual")] + RequiresAlignedPanePhaseOrExactWindowEdgeResidual, + NotApplicable, +} + +/// Node ids are shared with the logical export, since payloads embed them. +pub type PhysicalASAPNodeId = crate::ir::wire::LogicalASAPNodeId; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PhysicalASAPDAGNode { + pub id: PhysicalASAPNodeId, + /// The payload variant is the sole operator identity (`payload.kind` in JSON). + pub payload: PhysicalASAPOperatorPayload, + /// Phase is a placement choice for every operator, independent of payload kind. + pub output_state: ExecutionDataState, + pub output_schema: Schema, + pub guarantee: Option, + #[serde(default)] + pub coverage: Option, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PhysicalASAPDAGEdge { + pub producer: PhysicalASAPNodeId, + pub consumer: PhysicalASAPNodeId, + pub role: EdgeRole, + pub intermediate_schema: Schema, + pub data_state: ExecutionDataState, + pub grouping: GroupingEdgeCompatibility, + pub window: WindowEdgeCompatibility, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PhysicalASAPDAG { + pub nodes: Vec, + pub edges: Vec, + /// Semantic workload root. Physical query/precompute sinks are selected + /// downstream by the control plane. + /// One root per query of the batch, in workload order. Scalar query roots + /// are not physical nodes yet. + pub roots: Vec, +} + +/// Versioned transport envelope for a physical ASAP DAG. +/// +/// Process boundaries exchange this envelope and call [`Self::validate`]. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PhysicalASAPDAGDocument { + pub schema_version: u32, + pub dag: PhysicalASAPDAG, +} + +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum PhysicalASAPDAGValidationError { + #[error("phase assignment must name every DAG node exactly once")] + IncompletePhaseAssignment, + #[error("ingestion node {consumer:?} depends on query node {producer:?}")] + QueryDependencyInIngestion { + producer: PhysicalASAPNodeId, + consumer: PhysicalASAPNodeId, + }, + #[error("unsupported physical ASAP DAG schema version {0}")] + UnsupportedVersion(u32), + #[error("duplicate physical ASAP node id {0:?}")] + DuplicateNodeId(PhysicalASAPNodeId), + #[error("physical ASAP DAG root {0:?} does not name a node")] + MissingRoot(PhysicalASAPNodeId), + #[error("edge endpoint {0:?} does not name a node")] + MissingEdgeEndpoint(PhysicalASAPNodeId), + #[error("edge {producer:?}->{consumer:?} schema differs from producer output")] + EdgeSchemaMismatch { + producer: PhysicalASAPNodeId, + consumer: PhysicalASAPNodeId, + }, + #[error("edge {producer:?}->{consumer:?} data state differs from producer output")] + EdgeDataStateMismatch { + producer: PhysicalASAPNodeId, + consumer: PhysicalASAPNodeId, + }, + #[error("physical ASAP DAG has no query roots")] + NoRoots, + #[error("physical ASAP DAG contains a cycle")] + Cycle, + #[error("physical ASAP node {0:?} is not reachable from the root")] + UnreachableNode(PhysicalASAPNodeId), + #[error("summary aggregate node {node:?} output schema does not contain its declared family")] + SummaryFamilySchemaMismatch { node: PhysicalASAPNodeId }, + #[error( + "summary aggregate node {node:?} declares grouping inconsistent with its sketch state" + )] + SummaryGroupingMismatch { node: PhysicalASAPNodeId }, +} + +impl PhysicalASAPDAGDocument { + pub fn new(dag: PhysicalASAPDAG) -> Self { + Self { + schema_version: PHYSICAL_ASAP_DAG_WIRE_VERSION, + dag, + } + } + + pub fn validate(&self) -> Result<(), PhysicalASAPDAGValidationError> { + if self.schema_version != PHYSICAL_ASAP_DAG_WIRE_VERSION { + return Err(PhysicalASAPDAGValidationError::UnsupportedVersion( + self.schema_version, + )); + } + self.dag.validate() + } +} + +impl PhysicalASAPDAG { + /// Assign execution phases without changing operator semantics. Phase choices + /// do not prove deployment support: callers must bind concrete implementations + /// and storage boundaries before installing this plan. + pub fn with_execution_phases( + &self, + phases: &BTreeMap, + ) -> Result { + self.validate()?; + if phases.len() != self.nodes.len() + || self.nodes.iter().any(|node| !phases.contains_key(&node.id)) + { + return Err(PhysicalASAPDAGValidationError::IncompletePhaseAssignment); + } + let mut dag = self.clone(); + for node in &mut dag.nodes { + node.output_state.timing = phases[&node.id]; + } + let states: HashMap<_, _> = dag.nodes.iter().map(|n| (n.id, n.output_state)).collect(); + for edge in &mut dag.edges { + edge.data_state = states[&edge.producer]; + } + dag.validate()?; + Ok(dag) + } + + pub fn validate(&self) -> Result<(), PhysicalASAPDAGValidationError> { + let mut nodes = HashMap::new(); + for node in &self.nodes { + if nodes.insert(node.id, node).is_some() { + return Err(PhysicalASAPDAGValidationError::DuplicateNodeId(node.id)); + } + if let PhysicalASAPOperatorPayload::SummaryAgg { + family, grouping, .. + } = &node.payload + { + let mut found_family = false; + for field in &node.output_schema.fields { + if &field.dtype == family { + found_family = true; + } + if let FieldDataType::Sketch(_, schema_grouping) = &field.dtype { + if schema_grouping != grouping { + return Err(PhysicalASAPDAGValidationError::SummaryGroupingMismatch { + node: node.id, + }); + } + } + } + if !found_family { + return Err( + PhysicalASAPDAGValidationError::SummaryFamilySchemaMismatch { + node: node.id, + }, + ); + } + } + } + if self.roots.is_empty() { + return Err(PhysicalASAPDAGValidationError::NoRoots); + } + for root in &self.roots { + if !nodes.contains_key(root) { + return Err(PhysicalASAPDAGValidationError::MissingRoot(*root)); + } + } + let mut children: HashMap> = HashMap::new(); + for edge in &self.edges { + let producer = nodes.get(&edge.producer).ok_or( + PhysicalASAPDAGValidationError::MissingEdgeEndpoint(edge.producer), + )?; + if !nodes.contains_key(&edge.consumer) { + return Err(PhysicalASAPDAGValidationError::MissingEdgeEndpoint( + edge.consumer, + )); + } + if producer.output_state.timing == ExecutionTiming::QueryTime + && nodes[&edge.consumer].output_state.timing == ExecutionTiming::IngestionTime + { + return Err(PhysicalASAPDAGValidationError::QueryDependencyInIngestion { + producer: edge.producer, + consumer: edge.consumer, + }); + } + if edge.intermediate_schema != producer.output_schema { + return Err(PhysicalASAPDAGValidationError::EdgeSchemaMismatch { + producer: edge.producer, + consumer: edge.consumer, + }); + } + if edge.data_state != producer.output_state { + return Err(PhysicalASAPDAGValidationError::EdgeDataStateMismatch { + producer: edge.producer, + consumer: edge.consumer, + }); + } + children + .entry(edge.consumer) + .or_default() + .push(edge.producer); + } + fn visit( + id: PhysicalASAPNodeId, + children: &HashMap>, + visiting: &mut HashSet, + visited: &mut HashSet, + ) -> bool { + if visited.contains(&id) { + return true; + } + if !visiting.insert(id) { + return false; + } + if children + .get(&id) + .into_iter() + .flatten() + .any(|child| !visit(*child, children, visiting, visited)) + { + return false; + } + visiting.remove(&id); + visited.insert(id); + true + } + let mut visited = HashSet::new(); + for root in &self.roots { + if !visit(*root, &children, &mut HashSet::new(), &mut visited) { + return Err(PhysicalASAPDAGValidationError::Cycle); + } + } + fn mark( + id: PhysicalASAPNodeId, + children: &HashMap>, + reachable: &mut HashSet, + ) { + if !reachable.insert(id) { + return; + } + for child in children.get(&id).into_iter().flatten() { + mark(*child, children, reachable); + } + } + let mut reachable = HashSet::new(); + for root in &self.roots { + mark(*root, &children, &mut reachable); + } + if let Some(id) = nodes.keys().find(|id| !reachable.contains(id)) { + return Err(PhysicalASAPDAGValidationError::UnreachableNode(*id)); + } + Ok(()) + } +} + +// ── Compilation from the IR ────────────────────────────────────────────── + +/// Compiler-local identity assignment. It deliberately retains `Rc` handles +/// and is not serialized; deployed artifacts persist the physical ASAP node ID +/// together with their physical materialization/query IDs. +#[derive(Debug, Clone)] +pub struct PhysicalASAPNodeIdentityMap { + nodes_by_id: Vec>, +} + +impl PhysicalASAPNodeIdentityMap { + pub fn node_id(&self, node: &Rc) -> Option { + self.nodes_by_id + .iter() + .position(|candidate| Rc::ptr_eq(candidate, node)) + .map(|id| crate::ir::wire::LogicalASAPNodeId(id as u32)) + } + + pub fn operator_node(&self, id: PhysicalASAPNodeId) -> Option<&Rc> { + self.nodes_by_id.get(id.0 as usize) + } +} + +#[derive(Debug, Clone)] +pub struct PhysicalASAPDAGCompilation { + pub dag: PhysicalASAPDAG, + pub node_ids: PhysicalASAPNodeIdentityMap, +} + +pub fn compile_physical_asap_dag( + root: &Rc, +) -> Result { + Ok(compile_physical_asap_dag_with_node_ids(root)?.dag) +} + +/// Export the timed DAG below `root`. Every reachable node must carry a +/// timing (see [`crate::ir::properties::timing::apply_materialization_timings`]); the data-state +/// rules were checked by that pass and are not re-run here. +pub fn compile_physical_asap_dag_with_node_ids( + root: &Rc, +) -> Result { + compile_physical_asap_workload_with_node_ids(std::slice::from_ref(root)) +} + +/// Export a timed batch as one DAG with one root per query. +pub fn compile_physical_asap_workload( + roots: &[Rc], +) -> Result { + Ok(compile_physical_asap_workload_with_node_ids(roots)?.dag) +} + +pub fn compile_physical_asap_workload_with_node_ids( + roots: &[Rc], +) -> Result { + let mut exporter = Exporter::default(); + let roots = roots + .iter() + .map(|root| exporter.visit(root)) + .collect::, _>>()?; + let dag = PhysicalASAPDAG { + nodes: exporter.nodes, + edges: exporter.edges, + roots, + }; + dag.validate() + .expect("compiler emits a valid physical ASAP DAG"); + Ok(PhysicalASAPDAGCompilation { + dag, + node_ids: PhysicalASAPNodeIdentityMap { + nodes_by_id: exporter.nodes_by_id, + }, + }) +} + +#[derive(Default)] +struct Exporter { + ids: HashMap<*const OperatorNode, PhysicalASAPNodeId>, + nodes: Vec, + edges: Vec, + nodes_by_id: Vec>, +} + +impl Exporter { + fn visit( + &mut self, + node: &Rc, + ) -> Result { + if let Some(id) = self.ids.get(&Rc::as_ptr(node)) { + return Ok(*id); + } + let output_state = data_state(node).ok_or(ExecutionDataStateError::UntimedNode { + operator: node.operator.kind_name(), + })?; + // Operator inputs first, then the nodes read from scalar expressions. + let mut producers = Vec::new(); + for (child, role) in input_edges(&node.operator) { + producers.push((self.visit(child)?, child, role)); + } + { + let scalars = match &node.operator { + Operator::NonASAP(op) => op.scalar_exprs(), + Operator::ASAP(ASAPOp::SummaryAgg { + filter: Some(filter), + .. + }) => vec![&filter.0], + _ => vec![], + }; + for expr in scalars { + for referenced in expr.operator_refs() { + producers.push((self.visit(referenced)?, referenced, EdgeRole::ScalarRef)); + } + } + } + let id = crate::ir::wire::LogicalASAPNodeId(self.nodes.len() as u32); + let payload = { + let ids = &self.ids; + let mut id_of = |n: &Rc| ids[&Rc::as_ptr(n)]; + payload_of(&node.operator, &mut id_of) + }; + self.nodes.push(PhysicalASAPDAGNode { + id, + payload, + output_state, + output_schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + coverage: node.coverage.clone(), + }); + self.nodes_by_id.push(Rc::clone(node)); + self.ids.insert(Rc::as_ptr(node), id); + for (producer, child, role) in producers { + let producer_state = self.nodes[producer.0 as usize].output_state; + let maintenance_dependency = producer_state.timing == ExecutionTiming::IngestionTime + && output_state.timing == ExecutionTiming::IngestionTime; + self.edges.push(PhysicalASAPDAGEdge { + producer, + consumer: id, + role, + intermediate_schema: child.schema.clone(), + data_state: producer_state, + grouping: grouping_compatibility(&child.operator, &node.operator), + window: if maintenance_dependency { + WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual + } else { + WindowEdgeCompatibility::NotApplicable + }, + }); + } + Ok(id) + } +} diff --git a/crates/types/src/ir/properties/execution.rs b/crates/types/src/ir/properties/execution.rs new file mode 100644 index 000000000..22545824c --- /dev/null +++ b/crates/types/src/ir/properties/execution.rs @@ -0,0 +1,182 @@ +//! Execution timing and data-state vocabulary of the operator IR. +//! +//! [`ExecutionTiming`] says when a node's value is produced (ingestion vs. +//! query time); [`ExecutionDataState`] pairs it with the [`DataPrimitive`] +//! the edge carries (raw values vs. summary state). The rules that assign +//! and check them over a DAG live in [`crate::ir::properties::timing`], which reports +//! violations as [`ExecutionDataStateError`]. + +use thiserror::Error; + +/// When a post-ASAP value is produced. +#[derive( + Default, Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize, +)] +#[serde(rename_all = "snake_case")] +pub enum ExecutionTiming { + IngestionTime, + #[default] + QueryTime, +} + +impl ExecutionTiming { + pub fn is_query_time(&self) -> bool { + *self == Self::QueryTime + } + pub fn as_str(self) -> &'static str { + match self { + Self::IngestionTime => "ingestion_time", + Self::QueryTime => "query_time", + } + } +} + +/// The primitive representation carried by a post-ASAP edge. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] +pub enum DataPrimitive { + /// Directly usable values, including approximate summary evaluations. + /// This does not imply original input data or an exact guarantee. + Raw, + SummaryState, +} + +impl DataPrimitive { + pub fn as_str(self) -> &'static str { + match self { + Self::Raw => "raw", + Self::SummaryState => "summary_state", + } + } +} + +/// The two-dimensional edge contract: when a value exists and which data +/// primitive it carries. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] +pub struct ExecutionDataState { + pub timing: ExecutionTiming, + pub primitive: DataPrimitive, +} + +impl ExecutionDataState { + pub const INGESTION_ROWS: Self = Self { + timing: ExecutionTiming::IngestionTime, + primitive: DataPrimitive::Raw, + }; + pub const INGESTION_SUMMARY: Self = Self { + timing: ExecutionTiming::IngestionTime, + primitive: DataPrimitive::SummaryState, + }; + pub const QUERY_ROWS: Self = Self { + timing: ExecutionTiming::QueryTime, + primitive: DataPrimitive::Raw, + }; +} + +impl std::fmt::Display for ExecutionDataState { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}/{}", self.timing.as_str(), self.primitive.as_str()) + } +} + +/// A plan-construction-time data_state violation. Typed (not a string) so a +/// strategy can degrade to a conservative fallback on the specific variant +/// it expects, and so tests can assert the *reason* a plan was rejected. +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum ExecutionDataStateError { + #[error("invalid maintained-population maintenance/evaluation contract")] + InvalidMaintainedPopulation, + /// A query-time value (a `SummaryEstimate` or query-time operator output) + /// placed beneath a maintained summary — the one shape issue #171's + /// data_state split exists to make unrepresentable. + #[error( + "evaluation value under maintenance: {edge} received a {child} input, but a maintained \ + summary can only consume update-path values (or exact accumulator state)" + )] + EvaluationUnderMaintenance { + edge: &'static str, + child: ExecutionDataState, + }, + /// Any other edge whose child data_state the parent does not accept + /// (e.g. plain update rows fed straight into a `SummaryEstimate`, or a + /// sketch's opaque state fed into a query-time operator). + #[error("{edge} does not accept a {child} input")] + IllegalChildDataState { + edge: &'static str, + child: ExecutionDataState, + }, + /// A `SummaryAgg` whose child is summary state of a family other than an + /// exact accumulator — re-accumulating opaque sketch/sample/… state on + /// the update path has no defined semantics here. + #[error( + "SummaryAgg.child carries {family} summary state; only exact accumulator state can be \ + composed into another maintained summary" + )] + UnsupportedStateComposition { family: String }, + /// One shared node assigned two different execution timings by its + /// consumers; no single execution of it can serve both. + #[error("shared node is assigned conflicting timings: {first} and {second}")] + ConflictingTiming { + first: ExecutionDataState, + second: ExecutionDataState, + }, + /// A maintenance-time value operation at the root of a plan: + /// nothing maintains state above it, so its output is never read. + #[error("A maintenance-time value operation cannot be a plan root: its update-path output feeds nothing")] + MaintenanceRowsAtRoot, + #[error("unsupported maintenance binary schema or operator")] + InvalidMaintenanceBinary, + #[error("checked division requires one valid guard on a read-time division operator")] + InvalidCheckedDivision, + /// An exact operator whose input columns are not all `Plain` at its + /// declared data_state. + #[error("exact operator consumes non-plain column {column:?} ({dtype})")] + NonPlainOperand { column: String, dtype: String }, + /// A reserved ASAP operator (`SummaryMerge`, `SummarySubtract`, + /// `SummaryDelete`, `SummaryJoin`, `Extension`) in an executable plan. + #[error("{operator} is a reserved operator with no execution contract yet")] + UnimplementedOperator { operator: &'static str }, + /// A node reached by export without a timing: the materialization timing pass + /// was not applied to the DAG first. + #[error( + "{operator} node has no execution timing; apply materialization timings before export" + )] + UntimedNode { operator: &'static str }, +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Both execution phases use raw values, distinct from maintained state. + #[test] + fn raw_primitive_labels() { + assert_eq!( + ExecutionDataState::INGESTION_ROWS.primitive, + DataPrimitive::Raw + ); + assert_eq!(ExecutionDataState::QUERY_ROWS.primitive, DataPrimitive::Raw); + assert_eq!( + ExecutionDataState::INGESTION_ROWS.to_string(), + "ingestion_time/raw" + ); + assert_eq!(ExecutionDataState::QUERY_ROWS.to_string(), "query_time/raw"); + assert_eq!(DataPrimitive::SummaryState.as_str(), "summary_state"); + } + + #[test] + fn execution_phase_wire_names_are_ingestion_and_query_time() { + for (phase, name) in [ + (ExecutionTiming::IngestionTime, "ingestion_time"), + (ExecutionTiming::QueryTime, "query_time"), + ] { + assert_eq!(phase.as_str(), name); + assert_eq!(serde_json::to_value(phase).unwrap(), name); + assert_eq!( + serde_json::from_value::(serde_json::json!(name)).unwrap(), + phase + ); + } + assert!(serde_json::from_str::("\"maintenance_time\"").is_err()); + assert!(serde_json::from_str::("\"MaintenanceTime\"").is_err()); + } +} diff --git a/crates/types/src/post_asap/guarantee.rs b/crates/types/src/ir/properties/guarantee.rs similarity index 97% rename from crates/types/src/post_asap/guarantee.rs rename to crates/types/src/ir/properties/guarantee.rs index cfc6f3f01..6bc4810db 100644 --- a/crates/types/src/post_asap/guarantee.rs +++ b/crates/types/src/ir/properties/guarantee.rs @@ -9,16 +9,16 @@ //! failure-probability expressions, the provenance trail, and the typed //! rejection reasons. The *algebra* that composes these (the `AccuracyModel` //! trait, its default conservative rules, and budget allocation) lives one -//! layer up in `asap_aware_mapping::accuracy`, the same layering +//! layer up in `asap_logical_optimizer::accuracy`, the same layering //! [`crate::dag_export`] keeps for cost decisions: this crate defines the //! shapes, the planning crate decides. //! //! ## What a guarantee says //! //! [`ResultGuarantee`] is attached to a finalized, caller-visible value — -//! [`super::SummaryNode::guarantee`] on a `SummaryEstimate` readout, an +//! [`crate::ir::OperatorNode::guarantee`] on a `SummaryEstimate` evaluation, an //! exact accumulator, or a kept pre-ASAP sub-DAG — never to raw summary -//! state (a `SummaryAgg` sketch node carries `None`; its readout carries the +//! state (a `SummaryAgg` sketch node carries `None`; its evaluation carries the //! guarantee). Its statement is: //! //! ```text @@ -325,13 +325,13 @@ pub enum GuaranteeSource { /// Deterministic exact computation — zero error by construction. Exact { /// What made it exact (e.g. `"ExactAggregate(Sum)"`, - /// `"KeepPreAsap"`). + /// `"RetainedExact"`). reason: String, }, - /// The target this readout's sketch was sized against. + /// The target this evaluation's sketch was sized against. AccuracyTarget { target: AccuracyTarget }, - /// The concrete sketch a readout's local guarantee was derived from. - SketchReadout { + /// The concrete sketch a evaluation's local guarantee was derived from. + SketchEvaluation { algorithm: String, /// Stable estimator/analysis contract used to derive this guarantee. #[serde(default)] @@ -421,8 +421,8 @@ impl ResultGuarantee { self.bound.is_zero() && self.failure_probability.is_zero() } - /// How many approximate sketch readouts contributed to this value — - /// `1` for a plain readout, `0` for an exact value, and the transitive + /// How many approximate sketch evaluations contributed to this value — + /// `1` for a plain evaluation, `0` for an exact value, and the transitive /// count through every [`GuaranteeSource::ChildGuarantee`] for a /// composition. An `AccuracyBudgetAllocator` uses this as the number /// of layers a budget must be split across. @@ -430,7 +430,7 @@ impl ResultGuarantee { self.provenance .iter() .map(|source| match source { - GuaranteeSource::SketchReadout { .. } => 1, + GuaranteeSource::SketchEvaluation { .. } => 1, GuaranteeSource::ChildGuarantee { guarantee, .. } => { guarantee.approximate_layer_count() } diff --git a/crates/types/src/ir/properties/mod.rs b/crates/types/src/ir/properties/mod.rs new file mode 100644 index 000000000..a51c74808 --- /dev/null +++ b/crates/types/src/ir/properties/mod.rs @@ -0,0 +1,13 @@ +//! #511 §2.2–2.3: node properties beyond the schema — the accuracy +//! guarantee, execution timing and data state, and summary coverage. + +pub mod execution; +pub mod guarantee; +pub mod summary_coverage; +pub mod timing; + +pub use execution::{DataPrimitive, ExecutionDataState, ExecutionDataStateError, ExecutionTiming}; +pub use guarantee::{ + AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, GuaranteeSource, ProbabilityExpr, + ResultGuarantee, +}; diff --git a/crates/types/src/ir/properties/summary_coverage.rs b/crates/types/src/ir/properties/summary_coverage.rs new file mode 100644 index 000000000..19ae10eb8 --- /dev/null +++ b/crates/types/src/ir/properties/summary_coverage.rs @@ -0,0 +1,126 @@ +//! Joint time/population coverage for summary composition, independent of schema. +//! Equality predicates are a deliberately narrow proof vocabulary. Unsupported +//! predicates cannot be declared disjoint merely by giving them different names. +use crate::ir::operator::Source; +use serde::{Deserialize, Serialize}; +use std::collections::BTreeMap; +use std::ops::Range; +use thiserror::Error; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SummaryCoverage { + /// Observation data source, as named by `Scan`: a table or a time series. + /// Region time bounds refer to its time column. + pub source: Source, + /// Union of joint regions; never the Cartesian product of independent bounds. + pub regions: Vec, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct CoverageRegion { + /// Half-open bounds on the source's time column, in milliseconds. `None` + /// means no time restriction, e.g. a source without a time column. + pub time_ms: Option>, + /// Conjunction of non-null equality predicates; empty means unrestricted. + pub population: BTreeMap, +} + +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum CoverageError { + #[error("coverage interval must have start < end")] + InvalidInterval, + #[error("population dimension names cannot be empty")] + InvalidPopulation, + #[error("summary coverage sources differ")] + SourceMismatch, + #[error("coverage overlap is not proven absent")] + PossibleOverlap, + #[error("coverage merge requires at least one input")] + EmptyMerge, + #[error("summary coverage requires state output")] + NotState, + #[error("summary node requires coverage")] + Missing, + #[error("summary merge requires known coverage on every input")] + UnknownInput, + #[error("retained merge coverage disagrees with input union")] + MergeOutputMismatch, +} + +impl SummaryCoverage { + pub fn validate(&self) -> Result<(), CoverageError> { + for (index, region) in self.regions.iter().enumerate() { + if region.time_ms.as_ref().is_some_and(Range::is_empty) { + return Err(CoverageError::InvalidInterval); + } + if region.population.keys().any(String::is_empty) { + return Err(CoverageError::InvalidPopulation); + } + if self.regions[..index] + .iter() + .any(|other| region.may_overlap(other)) + { + return Err(CoverageError::PossibleOverlap); + } + } + Ok(()) + } + + /// Every observation in a region is assumed to contribute once to the state. + /// Compose once-per-observation summaries only when their joint regions are + /// provably disjoint. Update/reduction compatibility, family merge capability + /// and accuracy are checked by `SummaryMerge`, not here. + pub fn merge_disjoint(inputs: &[Self]) -> Result { + let first = inputs.first().ok_or(CoverageError::EmptyMerge)?; + let mut merged = first.clone(); + merged.regions.clear(); + for input in inputs { + input.validate()?; + if input.source != first.source { + return Err(CoverageError::SourceMismatch); + } + merged.regions.extend(input.regions.iter().cloned()); + } + merged.validate()?; + // Coalesce adjacent intervals only for identical population predicates. + merged.regions.sort_by(|a, b| { + a.population.cmp(&b.population).then( + a.time_ms + .as_ref() + .map(|t| t.start) + .cmp(&b.time_ms.as_ref().map(|t| t.start)), + ) + }); + let mut normalized: Vec = Vec::new(); + for region in merged.regions { + if let Some(last) = normalized.last_mut() { + if let (Some(last_time), Some(time)) = (&mut last.time_ms, ®ion.time_ms) { + if last.population == region.population && last_time.end == time.start { + last_time.end = time.end; + continue; + } + } + } + normalized.push(region); + } + merged.regions = normalized; + Ok(merged) + } +} +impl CoverageRegion { + fn may_overlap(&self, other: &Self) -> bool { + let time_overlaps = match (&self.time_ms, &other.time_ms) { + (Some(a), Some(b)) => a.start < b.end && b.start < a.end, + _ => true, + }; + time_overlaps + && !self.population.iter().any(|(dimension, value)| { + other + .population + .get(dimension) + .is_some_and(|other| other != value) + }) + } +} diff --git a/crates/types/src/ir/properties/timing.rs b/crates/types/src/ir/properties/timing.rs new file mode 100644 index 000000000..3d2154ee8 --- /dev/null +++ b/crates/types/src/ir/properties/timing.rs @@ -0,0 +1,950 @@ +//! Execution timing: written into every node from a materialization +//! assignment, then validated against each operator's kind and its consuming +//! edges. +//! +//! The logical DAG carries no timing. Materialization decides per summary +//! state whether it is maintained at ingestion time or computed at query +//! time; [`MaterializationAssignment`] records that choice per `SummaryAgg` +//! and [`apply_materialization_timings`] expands it into a timing on every node: +//! +//! - a node of fixed kind takes its kind's timing (`SummaryEstimate` and +//! `EvaluatePopulation` run at query time, `MaintainPopulation` at ingestion +//! time); +//! - a `SummaryAgg` takes the assignment's timing (default: query time, until +//! Stage 2 materialization (#509) chooses otherwise), unless something below +//! it can only exist at query time; +//! - every other node runs when its consumer runs: everything that feeds a +//! maintained state runs at ingestion time, everything above a evaluation at +//! query time. +//! +//! A node reached from two consumers that need different timings cannot be +//! executed once for both; [`split_shared_by_phase`] copies such a sub-DAG +//! for one side before the assignment is applied, and the pass itself +//! rejects a conflict it still finds. +//! +//! ## Edge rules (checked after the write) +//! +//! | Consumer | Accepts from an input | +//! |---|---| +//! | `SummaryAgg.child` | Rows, or exact-accumulator state, never a query-time value when the state is maintained | +//! | `SummaryEstimate.summary_input` | Summary state at either phase | +//! | `FinalizeExactAccumulator.child` | Exact-accumulator state | +//! | `EvaluatePopulation.child` | A `MaintainPopulation` at ingestion time | +//! | `MaintainPopulation.child` | Ingestion-time rows matching the population's input | +//! | any `NonASAP` consumer | Rows (or exact-accumulator state for a projection-like operator) at the consumer's own timing; ingestion work never reads a query-time value | + +use std::collections::HashMap; +use std::rc::Rc; + +use crate::ir::operator::asap::ASAPOp; +use crate::ir::operator::node::{Operator, OperatorNode}; +use crate::ir::operator::non_asap::NonASAPOp; +use crate::ir::operator::operator_properties::BinaryOpKind; +use crate::ir::properties::execution::{ + DataPrimitive, ExecutionDataState, ExecutionDataStateError, ExecutionTiming, +}; +use crate::ir::schema::{DataType, FieldDataType, Schema}; + +/// The per-state materialization choice: for each `SummaryAgg` node (by +/// identity), whether its state is maintained at ingestion time or computed +/// at query time. A state absent from the map takes the assignment's default. +/// `Default` is [`Self::all_query_time`]: nothing is materialized until Stage 2 +/// materialization (#509) decides otherwise. +#[derive(Debug, Clone, Default)] +pub struct MaterializationAssignment { + summary_timings: HashMap<*const OperatorNode, ExecutionTiming>, + default_timing: ExecutionTiming, +} + +impl MaterializationAssignment { + /// Every summary state computed at query time. + pub fn all_query_time() -> Self { + Self::default() + } + + /// Every summary state maintained at ingestion time. + pub fn all_ingestion_time() -> Self { + Self { + summary_timings: HashMap::new(), + default_timing: ExecutionTiming::IngestionTime, + } + } + + pub fn set(&mut self, summary: &Rc, timing: ExecutionTiming) { + self.summary_timings.insert(Rc::as_ptr(summary), timing); + } + + pub fn summary_timing(&self, summary: &Rc) -> ExecutionTiming { + self.summary_timings + .get(&Rc::as_ptr(summary)) + .copied() + .unwrap_or(self.default_timing) + } +} + +/// Memo of one [`apply_materialization_timings`] pass: `input node → timed node`, +/// shared by every root of a workload so a node shared by two roots stays +/// one `Rc`. Re-reaching a node with a different timing is a conflict. +#[derive(Default)] +pub struct TimingMemo { + done: HashMap<*const OperatorNode, Rc>, +} + +impl TimingMemo { + pub fn new() -> Self { + Self::default() + } + + /// The timed node produced for `input`, if the pass has reached it. + pub fn timed(&self, input: &Rc) -> Option<&Rc> { + self.done.get(&Rc::as_ptr(input)) + } +} + +/// The data state a timed node's output carries. +pub fn data_state(node: &OperatorNode) -> Option { + Some(ExecutionDataState { + timing: node.timing?, + primitive: match &node.operator { + Operator::ASAP(op) if op.produced_state().is_some() => DataPrimitive::SummaryState, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) => DataPrimitive::SummaryState, + _ => DataPrimitive::Raw, + }, + }) +} + +/// Whether the sub-DAG below `node` contains a node that can only run at +/// query time (a evaluation), which forces every consumer above it to query +/// time as well. +fn forces_query_time(node: &OperatorNode, seen: &mut HashMap<*const OperatorNode, bool>) -> bool { + let key = node as *const OperatorNode; + if let Some(&cached) = seen.get(&key) { + return cached; + } + let forced = match &node.operator { + _ if node.timing == Some(ExecutionTiming::QueryTime) => true, + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + | Operator::ASAP(ASAPOp::EvaluatePopulation { .. }) => true, + Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) + if node + .schema + .fields + .iter() + .any(|f| f.name == crate::ir::schema::PROMQL_SERIES_IDENTITY) + && per_series_rows(lhs).is_none_or(|rows| per_series_rows(rhs) != Some(rows)) => + { + true + } + _ => node + .children() + .iter() + .any(|child| forces_query_time(child, seen)), + }; + seen.insert(key, forced); + forced +} + +/// Write the timings of `assignment` into every node reachable from `root`, +/// top-down, then validate every edge. Returns the timed copy of `root`; +/// `memo` carries the sharing across the roots of one workload. +pub fn apply_materialization_timings( + root: &Rc, + assignment: &MaterializationAssignment, + memo: &mut TimingMemo, +) -> Result, ExecutionDataStateError> { + let mut forced = HashMap::new(); + let timed = write( + root, + ExecutionTiming::QueryTime, + assignment, + memo, + &mut forced, + )?; + if timed.timing == Some(ExecutionTiming::IngestionTime) + && data_state(&timed).map(|s| s.primitive) == Some(DataPrimitive::Raw) + { + return Err(ExecutionDataStateError::MaintenanceRowsAtRoot); + } + validate(&timed, &mut HashMap::new())?; + Ok(timed) +} + +/// Validate the sub-DAG below `root` with every summary maintained at +/// ingestion time ([`MaterializationAssignment::all_ingestion_time`]) and +/// `root` consumed at `root_timing`. For planning-time legality checks of a +/// candidate before it is assembled into a workload DAG: a candidate must stay +/// executable if materialization later maintains its states. Nothing is kept. +pub fn validate_maintained( + root: &Rc, + root_timing: ExecutionTiming, +) -> Result<(), ExecutionDataStateError> { + let assignment = MaterializationAssignment::all_ingestion_time(); + let mut memo = TimingMemo::new(); + let mut forced = HashMap::new(); + let timed = write(root, root_timing, &assignment, &mut memo, &mut forced)?; + validate(&timed, &mut HashMap::new()) +} + +/// The data state `node` produces with every summary maintained at ingestion +/// time when its consumer runs at `consumer` — the planning-time answer to +/// "what does this candidate's output look like", consistent with +/// [`validate_maintained`]. +pub fn planned_data_state( + node: &Rc, + consumer: ExecutionTiming, +) -> ExecutionDataState { + let mut forced = HashMap::new(); + let timing = own_timing( + node, + consumer, + &MaterializationAssignment::all_ingestion_time(), + &mut forced, + ); + ExecutionDataState { + timing, + primitive: match &node.operator { + Operator::ASAP(op) if op.produced_state().is_some() => DataPrimitive::SummaryState, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) => DataPrimitive::SummaryState, + _ => DataPrimitive::Raw, + }, + } +} + +/// The timing `node` takes when its consumer runs at `consumer`. +fn own_timing( + node: &Rc, + consumer: ExecutionTiming, + assignment: &MaterializationAssignment, + forced: &mut HashMap<*const OperatorNode, bool>, +) -> ExecutionTiming { + // A placement fixed when the candidate was built (an exact-state read + // boundary that must run at query time, or one that feeds maintenance) + // is honored; a conflicting consumer is rejected by validation. + if let Some(placed) = node.timing { + return placed; + } + match &node.operator { + Operator::ASAP(ASAPOp::SummaryEstimate { .. }) + | Operator::ASAP(ASAPOp::EvaluatePopulation { .. }) => ExecutionTiming::QueryTime, + Operator::ASAP(ASAPOp::MaintainPopulation { .. }) => ExecutionTiming::IngestionTime, + Operator::ASAP(ASAPOp::SummaryAgg { child, .. }) => { + if forces_query_time(child, forced) { + ExecutionTiming::QueryTime + } else { + assignment.summary_timing(node) + } + } + _ => consumer, + } +} + +fn write( + node: &Rc, + consumer: ExecutionTiming, + assignment: &MaterializationAssignment, + memo: &mut TimingMemo, + forced: &mut HashMap<*const OperatorNode, bool>, +) -> Result, ExecutionDataStateError> { + let timing = own_timing(node, consumer, assignment, forced); + if let Some(done) = memo.done.get(&Rc::as_ptr(node)) { + let previous = done.timing.expect("memoized node is timed"); + if previous != timing { + return Err(ExecutionDataStateError::ConflictingTiming { + first: ExecutionDataState { + timing: previous, + primitive: data_state(done).map_or(DataPrimitive::Raw, |s| s.primitive), + }, + second: ExecutionDataState { + timing, + primitive: data_state(done).map_or(DataPrimitive::Raw, |s| s.primitive), + }, + }); + } + return Ok(Rc::clone(done)); + } + let mut error = None; + let operator = + node.operator.map_children( + |child| match write(child, timing, assignment, memo, forced) { + Ok(timed) => timed, + Err(e) => { + error.get_or_insert(e); + Rc::clone(child) + } + }, + ); + if let Some(e) = error { + return Err(e); + } + let timed = Rc::new(OperatorNode { + operator, + result_kind: node.result_kind, + schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + timing: Some(timing), + // Timing copies the same logical sub-DAG; its observations are unchanged. + coverage: node.coverage.clone(), + }); + memo.done.insert(Rc::as_ptr(node), Rc::clone(&timed)); + Ok(timed) +} + +fn state_of(node: &OperatorNode) -> ExecutionDataState { + data_state(node).expect("timed node") +} + +/// Check every edge below `node` against the module-level rules. +fn validate( + node: &Rc, + seen: &mut HashMap<*const OperatorNode, ()>, +) -> Result<(), ExecutionDataStateError> { + if seen.insert(Rc::as_ptr(node), ()).is_some() { + return Ok(()); + } + let timing = node.timing.expect("timed node"); + match &node.operator { + Operator::ASAP(op) => validate_asap(node, op, timing)?, + Operator::NonASAP(op) => validate_non_asap(node, op, timing)?, + } + for child in node.children() { + validate(child, seen)?; + } + Ok(()) +} + +fn is_exact_accumulator_state(schema: &Schema) -> Result<(), ExecutionDataStateError> { + for field in &schema.fields { + match &field.dtype { + FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) => {} + other => { + return Err(ExecutionDataStateError::UnsupportedStateComposition { + family: format!("{other:?}"), + }) + } + } + } + Ok(()) +} + +fn validate_asap( + node: &OperatorNode, + op: &ASAPOp, + timing: ExecutionTiming, +) -> Result<(), ExecutionDataStateError> { + match op { + ASAPOp::SummaryAgg { child, .. } => { + let avail = state_of(child); + match avail { + ExecutionDataState::INGESTION_ROWS | ExecutionDataState::QUERY_ROWS => {} + s if s.primitive == DataPrimitive::SummaryState => { + is_exact_accumulator_state(&child.schema)? + } + other => { + return Err(ExecutionDataStateError::EvaluationUnderMaintenance { + edge: "SummaryAgg.child", + child: other, + }) + } + } + if timing == ExecutionTiming::IngestionTime + && avail.timing == ExecutionTiming::QueryTime + { + return Err(ExecutionDataStateError::EvaluationUnderMaintenance { + edge: "SummaryAgg.child", + child: avail, + }); + } + Ok(()) + } + ASAPOp::SummaryEstimate { summary_input, .. } => { + let s = state_of(summary_input); + if s.primitive != DataPrimitive::SummaryState { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: "SummaryEstimate.summary_input", + child: s, + }); + } + if timing != ExecutionTiming::QueryTime { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: "SummaryEstimate", + child: state_of(node), + }); + } + Ok(()) + } + ASAPOp::FinalizeExactAccumulator { child } => { + let s = state_of(child); + if s.primitive != DataPrimitive::SummaryState + || is_exact_accumulator_state(&child.schema).is_err() + || (timing == ExecutionTiming::IngestionTime && s.timing != timing) + { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: "FinalizeExactAccumulator.child", + child: s, + }); + } + Ok(()) + } + ASAPOp::MaintainPopulation { child, population } => { + let valid = population.matches_node(child) + && state_of(child) + == ExecutionDataState { + timing, + primitive: DataPrimitive::Raw, + }; + if !valid { + return Err(ExecutionDataStateError::InvalidMaintainedPopulation); + } + Ok(()) + } + ASAPOp::EvaluatePopulation { child, evaluation } => { + let valid = timing == ExecutionTiming::QueryTime + && matches!( + &child.operator, + Operator::ASAP(ASAPOp::MaintainPopulation { population, .. }) + if population.supports(evaluation) + && child.timing.is_some() + ); + if !valid { + return Err(ExecutionDataStateError::InvalidMaintainedPopulation); + } + Ok(()) + } + ASAPOp::SummaryMerge { .. } + | ASAPOp::SummarySubtract { .. } + | ASAPOp::SummaryDelete { .. } + | ASAPOp::SummaryJoin { .. } + | ASAPOp::Extension { .. } => Err(ExecutionDataStateError::UnimplementedOperator { + operator: op.kind_name(), + }), + } +} + +fn check_plain_or_exact_values(input: &Schema) -> Result<(), ExecutionDataStateError> { + for field in &input.fields { + if !matches!( + field.dtype, + FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) + ) { + return Err(ExecutionDataStateError::NonPlainOperand { + column: field.name.clone(), + dtype: format!("{:?}", field.dtype), + }); + } + } + Ok(()) +} + +fn check_all_plain(input: &Schema) -> Result<(), ExecutionDataStateError> { + for field in &input.fields { + if !field.is_plain() { + return Err(ExecutionDataStateError::NonPlainOperand { + column: field.name.clone(), + dtype: format!("{:?}", field.dtype), + }); + } + } + Ok(()) +} + +fn validate_non_asap( + node: &OperatorNode, + op: &NonASAPOp, + timing: ExecutionTiming, +) -> Result<(), ExecutionDataStateError> { + // Every input is rows at this node's own timing. Ingestion work never + // reads a query-time value; exact-accumulator state may pass through + // the projection-like operators unchanged. + for child in op.children() { + let s = state_of(child); + let passes_state = matches!( + op, + NonASAPOp::Project { .. } + | NonASAPOp::Filter { .. } + | NonASAPOp::Sort { .. } + | NonASAPOp::Limit { .. } + ) && s.primitive == DataPrimitive::SummaryState + && is_exact_accumulator_state(&child.schema).is_ok(); + if s.timing != timing || (s.primitive != DataPrimitive::Raw && !passes_state) { + return Err(ExecutionDataStateError::IllegalChildDataState { + edge: op.kind_name(), + child: s, + }); + } + } + match op { + NonASAPOp::Project { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => check_plain_or_exact_values(&child.schema)?, + NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } => { + let mut referenced: Vec = reduction + .group_keys() + .map(|keys| keys.keys().to_vec()) + .unwrap_or_default(); + for m in measures { + referenced.extend(m.input_cols()); + } + let implicit = measures.iter().any(|m| m.input_cols().is_empty()); + for (i, field) in child.schema.fields.iter().enumerate() { + if (implicit || referenced.contains(&i)) && !field.is_plain() { + return Err(ExecutionDataStateError::NonPlainOperand { + column: field.name.clone(), + dtype: format!("{:?}", field.dtype), + }); + } + } + } + NonASAPOp::BinaryOp { + operator, lhs, rhs, .. + } => { + let is_div = matches!( + operator.kind, + BinaryOpKind::Arithmetic(crate::ir::scalar::ArithmeticOpKind::Div) + ); + if (operator.checked_relative_division && operator.checked_finite_division) + || ((operator.checked_relative_division || operator.checked_finite_division) + && (timing != ExecutionTiming::QueryTime || !is_div)) + { + return Err(ExecutionDataStateError::InvalidCheckedDivision); + } + if timing == ExecutionTiming::IngestionTime { + let plain_float_or_ts = |schema: &Schema| { + schema.fields.iter().all(|field| { + !field.nullable + && (matches!( + field.dtype, + FieldDataType::Plain(DataType::Float64 | DataType::Timestamp) + ) || (field.name == crate::ir::schema::PROMQL_SERIES_IDENTITY + && field.dtype == FieldDataType::Plain(DataType::Utf8))) + }) + }; + let float_count = node + .schema + .fields + .iter() + .filter(|f| matches!(f.dtype, FieldDataType::Plain(DataType::Float64))) + .count(); + if operator.vector_match.is_some() + || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) + || lhs.schema != rhs.schema + || lhs.schema != node.schema + || node + .schema + .fields + .iter() + .filter(|f| f.name == crate::ir::schema::PROMQL_SERIES_IDENTITY) + .count() + > 1 + || (node + .schema + .fields + .iter() + .any(|f| f.name == crate::ir::schema::PROMQL_SERIES_IDENTITY) + && per_series_rows(lhs) + .is_none_or(|rows| per_series_rows(rhs) != Some(rows))) + || !plain_float_or_ts(&node.schema) + || float_count != 1 + { + return Err(ExecutionDataStateError::InvalidMaintenanceBinary); + } + } + } + _ => { + for child in op.children() { + check_all_plain(&child.schema)?; + } + } + } + Ok(()) +} + +/// Copy, for one consumer, every sub-DAG that `assignment` would reach with +/// two different timings, so that a workload whose CSE shared a `Scan` +/// between an ingestion-time summary and a query-time computation can still +/// be timed. Only the conflicting sub-DAGs are copied; a sub-DAG reached with +/// one timing stays one `Rc`. Returns the (possibly rewritten) root. +pub fn split_shared_by_phase( + root: &Rc, + assignment: &MaterializationAssignment, +) -> Rc { + // First pass: the set of timings each node is reached with. + let mut reached: HashMap<*const OperatorNode, Vec> = HashMap::new(); + let mut forced = HashMap::new(); + fn collect( + node: &Rc, + consumer: ExecutionTiming, + assignment: &MaterializationAssignment, + reached: &mut HashMap<*const OperatorNode, Vec>, + forced: &mut HashMap<*const OperatorNode, bool>, + ) { + let timing = own_timing(node, consumer, assignment, forced); + let entry = reached.entry(Rc::as_ptr(node)).or_default(); + if entry.contains(&timing) { + return; + } + entry.push(timing); + for child in node.children() { + collect(child, timing, assignment, reached, forced); + } + } + collect( + root, + ExecutionTiming::QueryTime, + assignment, + &mut reached, + &mut forced, + ); + if reached.values().all(|timings| timings.len() <= 1) { + return Rc::clone(root); + } + // Second pass: rebuild, giving each (node, timing) pair its own copy. + let mut copies: HashMap<(*const OperatorNode, ExecutionTiming), Rc> = + HashMap::new(); + fn rebuild( + node: &Rc, + consumer: ExecutionTiming, + assignment: &MaterializationAssignment, + reached: &HashMap<*const OperatorNode, Vec>, + copies: &mut HashMap<(*const OperatorNode, ExecutionTiming), Rc>, + forced: &mut HashMap<*const OperatorNode, bool>, + ) -> Rc { + let timing = own_timing(node, consumer, assignment, forced); + let key = (Rc::as_ptr(node), timing); + if let Some(done) = copies.get(&key) { + return Rc::clone(done); + } + let conflicted = reached + .get(&Rc::as_ptr(node)) + .is_some_and(|timings| timings.len() > 1); + let mut changed = conflicted; + let operator = node.operator.map_children(|child| { + let rebuilt = rebuild(child, timing, assignment, reached, copies, forced); + changed |= !Rc::ptr_eq(&rebuilt, child); + rebuilt + }); + let out = if changed { + Rc::new(OperatorNode { + operator, + result_kind: node.result_kind, + schema: node.schema.clone(), + guarantee: node.guarantee.clone(), + timing: node.timing, + coverage: node.coverage.clone(), + }) + } else { + Rc::clone(node) + }; + copies.insert(key, Rc::clone(&out)); + out + } + rebuild( + root, + ExecutionTiming::QueryTime, + assignment, + &reached, + &mut copies, + &mut forced, + ) +} + +/// Maintenance arithmetic needs the same per-series population on both sides. +fn per_series_rows(node: &OperatorNode) -> Option<&OperatorNode> { + use crate::ir::schema::ExactKind; + match &node.operator { + Operator::ASAP(ASAPOp::FinalizeExactAccumulator { child }) => match &child.operator { + Operator::ASAP(ASAPOp::SummaryAgg { + child, + family: FieldDataType::ExactAggregate(ExactKind::Sum | ExactKind::Count, _), + reduction: crate::ir::operator::Reduction::PerEntity, + filter: None, + .. + }) => Some(child), + _ => None, + }, + Operator::NonASAP(NonASAPOp::BinaryOp { lhs, rhs, .. }) => { + let rows = per_series_rows(lhs)?; + (per_series_rows(rhs) == Some(rows)).then_some(rows) + } + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ir::operator::agg_intent::AggIntent; + use crate::ir::operator::non_asap::NonASAPOp; + use crate::ir::operator::operator_properties::{Reduction, Source}; + use crate::ir::scalar::ColumnRef; + use crate::ir::schema::state_type::{ + ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, + SketchStatistic, SummaryUpdate, + }; + use crate::ir::schema::Field; + + fn scan_with(fields: Vec) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Scan { + source: Source::TimeSeries { metric: "m".into() }, + predicates: vec![], + schema: Schema::with_time_index(fields, 0, vec![]), + })) + .unwrap() + } + + fn scan() -> Rc { + scan_with(vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("value", DataType::Float64, false), + Field::plain("zone", DataType::Utf8, true), + ]) + } + + fn kll() -> FieldDataType { + FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), + GroupingStrategy::default(), + ) + } + + fn exact_sum() -> FieldDataType { + FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum) + } + + fn agg(child: Rc, family: FieldDataType) -> Rc { + std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::SummaryAgg { + child, + family: family.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + }), + Schema::lifted(vec![Field::new("state", family, false)], None), + ) + .with_guarantee(None), + ) + } + + fn estimate(child: Rc) -> Rc { + std::rc::Rc::new( + OperatorNode::with_schema( + crate::ir::Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: child, + query: SketchStatistic::Quantile { q: 0.99 }, + }), + Schema::lifted( + vec![Field::plain("quantile_0_99", DataType::Float64, false)], + None, + ), + ) + .with_guarantee(None), + ) + } + + fn aggregate(measure: AggIntent, child: Rc) -> Rc { + OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Aggregate { + reduction: Reduction::by(vec![]), + measures: vec![measure], + output_names: vec![], + filters: vec![], + having: None, + child, + })) + .unwrap() + } + + fn max(child: Rc) -> Rc { + aggregate(AggIntent::Max { col: None }, child) + } + + /// `node` with its timing fixed in advance, as a candidate builder does + /// for an operator that must feed maintenance. + fn placed_at_ingestion(node: Rc) -> Rc { + Rc::new( + (*node) + .clone() + .with_timing(Some(ExecutionTiming::IngestionTime)), + ) + } + + /// Apply with every summary maintained, the placement whose edge rules + /// these tests exercise. + fn apply(root: &Rc) -> Result, ExecutionDataStateError> { + apply_materialization_timings( + root, + &MaterializationAssignment::all_ingestion_time(), + &mut TimingMemo::new(), + ) + } + + fn child(node: &Rc) -> Rc { + Rc::clone(node.children()[0]) + } + + /// Without a materialization decision, a summary and its input run at + /// query time; an explicit per-state choice overrides the default. + #[test] + fn default_assignment_materializes_nothing() { + let summary = agg(scan(), kll()); + let root = apply_materialization_timings( + &summary, + &MaterializationAssignment::default(), + &mut TimingMemo::new(), + ) + .unwrap(); + assert_eq!( + data_state(&root), + Some(ExecutionDataState { + timing: ExecutionTiming::QueryTime, + primitive: DataPrimitive::SummaryState, + }) + ); + assert_eq!( + data_state(&child(&root)), + Some(ExecutionDataState::QUERY_ROWS) + ); + let mut assignment = MaterializationAssignment::all_query_time(); + assignment.set(&summary, ExecutionTiming::IngestionTime); + let root = + apply_materialization_timings(&summary, &assignment, &mut TimingMemo::new()).unwrap(); + assert_eq!( + data_state(&root), + Some(ExecutionDataState::INGESTION_SUMMARY) + ); + } + + #[test] + fn summary_agg_input_runs_at_ingestion_time() { + let root = apply(&agg(scan(), kll())).unwrap(); + assert_eq!( + data_state(&root), + Some(ExecutionDataState::INGESTION_SUMMARY) + ); + assert_eq!( + data_state(&child(&root)), + Some(ExecutionDataState::INGESTION_ROWS) + ); + } + + #[test] + fn exact_accumulator_state_may_feed_another_summary_agg() { + let inner = agg(scan(), exact_sum()); + assert!(apply(&estimate(agg(inner, kll()))).is_ok()); + } + + #[test] + fn evaluation_can_feed_summary_construction_at_query_time() { + let inner = estimate(agg(scan(), kll())); + let root = apply(&estimate(agg(inner, kll()))).unwrap(); + assert_eq!(child(&root).timing, Some(ExecutionTiming::QueryTime)); + } + + /// Any non-ASAP operator over a evaluation runs at query time. + #[test] + fn query_time_operation_over_evaluation_is_legal_and_root_is_evaluation() { + let evaluation = || estimate(agg(scan(), kll())); + let sorted = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Sort { + keys: vec![], + partition_by: Default::default(), + child: evaluation(), + })) + .unwrap(); + for root in [max(evaluation()), sorted] { + let root = apply(&root).unwrap(); + assert_eq!(data_state(&root), Some(ExecutionDataState::QUERY_ROWS)); + } + } + + #[test] + fn query_time_values_can_feed_query_time_summary_construction() { + let post = max(estimate(agg(scan(), kll()))); + let root = apply(&estimate(agg(post, kll()))).unwrap(); + assert_eq!(child(&root).timing, Some(ExecutionTiming::QueryTime)); + } + + #[test] + fn function_under_summary_agg_is_legal_but_not_at_root() { + let operation = placed_at_ingestion(max(scan())); + assert_eq!( + apply(&operation).err(), + Some(ExecutionDataStateError::MaintenanceRowsAtRoot) + ); + let root = apply(&estimate(agg(operation, kll()))).unwrap(); + let timed_operation = child(&child(&root)); + assert_eq!( + data_state(&timed_operation), + Some(ExecutionDataState::INGESTION_ROWS) + ); + } + + #[test] + fn function_over_evaluation_is_rejected() { + let operation = placed_at_ingestion(max(estimate(agg(scan(), kll())))); + assert!(matches!( + apply(&estimate(agg(operation, kll()))), + Err(ExecutionDataStateError::IllegalChildDataState { + child: ExecutionDataState::QUERY_ROWS, + .. + }) + )); + } + + /// One shared sub-DAG reached as maintenance input and as query-time + /// input cannot be executed once for both; splitting it by phase first + /// makes the plan timeable. + #[test] + fn a_shared_subtree_reached_at_two_timings_conflicts() { + let shared = scan(); + let root = OperatorNode::new_shared(crate::ir::Operator::NonASAP(NonASAPOp::Concat { + children: vec![ + max(estimate(agg(Rc::clone(&shared), kll()))), + max(Rc::clone(&shared)), + ], + discriminator_unique_key: None, + })) + .unwrap(); + assert_eq!( + apply(&root).err(), + Some(ExecutionDataStateError::ConflictingTiming { + first: ExecutionDataState::INGESTION_ROWS, + second: ExecutionDataState::QUERY_ROWS, + }) + ); + let split = split_shared_by_phase(&root, &MaterializationAssignment::all_ingestion_time()); + assert!(apply(&split).is_ok()); + } + + /// Both paired operands must be plain; an unrelated state column is not + /// an input. + #[test] + fn pearson_corr_checks_both_operand_states() { + let corr_over = |state_column: usize| { + let mut fields = vec![ + Field::plain("ts", DataType::Timestamp, false), + Field::plain("x", DataType::Float64, false), + Field::plain("y", DataType::Float64, false), + Field::plain("unused", DataType::Float64, false), + ]; + fields[state_column].dtype = kll(); + aggregate( + AggIntent::PearsonCorr { left: 1, right: 2 }, + scan_with(fields), + ) + }; + for operand in [1, 2] { + assert!(matches!( + validate_maintained(&corr_over(operand), ExecutionTiming::QueryTime), + Err(ExecutionDataStateError::NonPlainOperand { .. }) + )); + } + validate_maintained(&corr_over(3), ExecutionTiming::QueryTime).unwrap(); + } +} diff --git a/crates/types/src/ir/query.rs b/crates/types/src/ir/query.rs index 682c59465..56c469cb3 100644 --- a/crates/types/src/ir/query.rs +++ b/crates/types/src/ir/query.rs @@ -1,6 +1,6 @@ //! Query results are either an operator result or a standalone scalar expression. //! The root discriminator is not an operator and never creates a dag node. -use super::{OperatorNode, ScalarExpr}; +use crate::ir::{OperatorNode, ScalarExpr}; use serde::{Deserialize, Serialize}; use std::rc::Rc; #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] @@ -18,7 +18,7 @@ impl QueryRoot { match self { Self::Operator(node) => node.validate_structure(), Self::Scalar(expr) => { - expr.scalar_type(&crate::pre_asap::Schema::default())?; + expr.scalar_type(&crate::ir::schema::Schema::default())?; for node in expr.operator_refs() { node.validate_structure()?; } diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/ir/scalar/column_resolution.rs similarity index 52% rename from crates/types/src/pre_asap/column_resolution.rs rename to crates/types/src/ir/scalar/column_resolution.rs index cfafbe9e0..587d182eb 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/ir/scalar/column_resolution.rs @@ -1,23 +1,14 @@ //! Schema-driven column resolution. //! -//! Front ends (issue #179) emit `ColumnRef` (name-based, optionally -//! table-qualified); the canonical DAG uses positional [`ColumnId`] resolved -//! against a per-node [`Schema`]. These helpers bridge the two — the -//! [`SchemaResolver`](super::schema_resolver) builds the schema, and [`resolve_column_refs`] -//! turns name-based refs (group keys, dedup columns) into positional ids, -//! qualifier-aware. - -use std::rc::Rc; +//! Front ends emit `ColumnRef` (name-based, optionally table-qualified); the +//! IR uses positional [`ColumnId`] resolved against a per-node [`Schema`]. +//! These helpers turn name-based refs into positional ids, qualifier-aware. +//! Front-end name resolution (`asap_frontend_common::resolve`) calls them. use thiserror::Error; -use super::agg_intent::AggIntent; -use super::expr_ir::ColumnRef; -use super::query_expr::{ - aggregate_output_schema, GroupKeys, QueryExpr, QueryExprError, Reduction, ResolvedQueryExpr, - UnresolvedQueryExpr, -}; -use super::schema::{ColumnId, DataType, FieldDataType, Schema}; +use crate::ir::scalar::ColumnRef; +use crate::ir::schema::{ColumnId, DataType, FieldDataType, Schema}; /// Errors returned by the resolution helpers. #[derive(Debug, Error, PartialEq, Eq)] @@ -115,106 +106,10 @@ pub fn resolve_group_keys_promql( .collect() } -/// Resolve a name-based scalar [`UnresolvedQueryExpr`] (one of `QueryExpr`'s scalar -/// variants, issue #205) into a positional [`ResolvedQueryExpr`] by resolving every -/// column reference against `schema`. Structural otherwise. `expr` must be -/// one of the scalar variants — an operator variant here is a construction -/// bug, not a shape this needs to handle silently. -pub fn resolve_expr( - expr: &UnresolvedQueryExpr, - schema: &Schema, -) -> Result { - let rc = |e: &UnresolvedQueryExpr| -> Result, ResolveError> { - Ok(Rc::new(resolve_expr(e, schema)?)) - }; - let each = |es: &[UnresolvedQueryExpr]| -> Result, ResolveError> { - es.iter().map(|e| resolve_expr(e, schema)).collect() - }; - Ok(match expr { - QueryExpr::Column(c) => QueryExpr::Column(resolve_column_ref(c, schema)?), - QueryExpr::Literal(s) => QueryExpr::Literal(s.clone()), - QueryExpr::EvalTimestamp => QueryExpr::EvalTimestamp, - QueryExpr::CurrentTimestamp => QueryExpr::CurrentTimestamp, - QueryExpr::Compare { left, op, right } => QueryExpr::Compare { - left: rc(left)?, - op: op.clone(), - right: rc(right)?, - }, - QueryExpr::BoolAnd(v) => QueryExpr::BoolAnd(each(v)?), - QueryExpr::BoolOr(v) => QueryExpr::BoolOr(each(v)?), - QueryExpr::Not(e) => QueryExpr::Not(rc(e)?), - QueryExpr::IsNull(e) => QueryExpr::IsNull(rc(e)?), - QueryExpr::IsNotNull(e) => QueryExpr::IsNotNull(rc(e)?), - QueryExpr::Cast { expr, to, try_cast } => QueryExpr::Cast { - expr: rc(expr)?, - to: to.clone(), - try_cast: *try_cast, - }, - QueryExpr::InList { - expr, - list, - negated, - } => QueryExpr::InList { - expr: rc(expr)?, - list: each(list)?, - negated: *negated, - }, - QueryExpr::FunctionCall { name, args } => QueryExpr::FunctionCall { - name: name.clone(), - args: each(args)?, - }, - QueryExpr::Arithmetic { op, left, right } => QueryExpr::Arithmetic { - op: op.clone(), - left: rc(left)?, - right: rc(right)?, - }, - QueryExpr::Case { - operand, - branches, - else_expr, - } => QueryExpr::Case { - operand: operand.as_deref().map(rc).transpose()?, - branches: branches - .iter() - .map(|(w, t)| Ok((resolve_expr(w, schema)?, resolve_expr(t, schema)?))) - .collect::, ResolveError>>()?, - else_expr: else_expr.as_deref().map(rc).transpose()?, - }, - other => unreachable!("resolve_expr called on a non-scalar QueryExpr variant: {other:?}"), - }) -} - -/// Output schema produced by an `Aggregate { by, measures }` over `input`. -/// Mirrors `QueryExpr::output_schema_in`'s `Aggregate` arm; out-of-range `by` -/// ids are silently dropped (callers needing the strict check resolve `by` -/// via [`resolve_column_refs`], which surfaces `NotFound`). -pub fn output_schema_for_aggregate( - input: &Schema, - by: &GroupKeys, - measures: &[AggIntent], - output_names: &[String], -) -> Result { - // Delegate to the single canonical derivation so HAVING resolution can never - // drift from `QueryExpr::output_schema_in` (issue #41). HAVING is SQL-only - // and cross-series (SQL has no `without`), but detect the child-independent - // per-entity case anyway (a lone `rate`/`increase`/`*_over_time` intent) so - // the two agree on every shared input — the range-window child marker the - // canonical arm also keys off is not visible here, and never co-occurs with - // HAVING. - let per_entity = - by.is_empty() && !by.is_without() && measures.len() == 1 && measures[0].is_per_series(); - let reduction = if per_entity { - Reduction::PerEntity - } else { - Reduction::Reduce(by.clone()) - }; - aggregate_output_schema(input, &reduction, measures, output_names) -} - #[cfg(test)] mod tests { use super::*; - use crate::pre_asap::schema::Field; + use crate::ir::schema::Field; /// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. fn ts_value_schema() -> Schema { @@ -326,77 +221,4 @@ mod tests { Err(ResolveError::NotFound { .. }) )); } - - #[test] - fn aggregate_strips_time_and_keeps_unique_keys() { - let mut input = ts_value_schema(); - input - .fields - .push(Field::plain("host", DataType::Utf8, false)); - let out = output_schema_for_aggregate( - &input, - &GroupKeys::by(vec![2]), - &[AggIntent::Sum { col: None }], - &[], - ) - .expect("valid group-by column"); - assert_eq!(out.fields.len(), 2); // host, sum - assert_eq!(out.fields[0].name, "host"); - assert_eq!(out.fields[1].name, "sum"); - assert!(out.time_index.is_none()); - assert_eq!(out.unique_keys, vec![vec![0]]); - } - - #[test] - fn having_schema_agrees_with_canonical_for_a_per_series_reduction() { - // Issue #41: `output_schema_for_aggregate` (HAVING resolution) and the - // canonical `QueryExpr::output_schema_in` must produce identical schemas - // for the same aggregate. Before the dedup this diverged on a per-series - // reduction — the HAVING mirror lacked the per-series branch and would - // collapse `[ts, value]` to a single `rate` column. - use crate::pre_asap::query_expr::Source; - use std::time::Duration; - - let leaf_schema = Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ); - let scan = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: leaf_schema.clone(), - }; - // Aggregate{ reduction: PerEntity, [Rate], child: TimeRange{ Scan } } — - // a per-series reduction (label-preserving). - let agg = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Rate], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan), - }), - }; - let canonical = agg.output_schema().expect("canonical schema"); - - // The HAVING-resolution derivation gets only the input schema (the - // TimeRange passes the leaf schema through). - let having_side = - output_schema_for_aggregate(&leaf_schema, &GroupKeys::none(), &[AggIntent::Rate], &[]) - .unwrap(); - - assert_eq!( - canonical, having_side, - "the two aggregate-schema derivations must agree (issue #41)" - ); - // Sanity: it really is the label-preserving per-series shape, not `[rate]`. - assert!(having_side.fields.iter().any(|c| c.name == "value")); - assert!(having_side.time_index.is_some()); - } } diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/ir/scalar/expr_ir.rs similarity index 73% rename from crates/types/src/pre_asap/expr_ir.rs rename to crates/types/src/ir/scalar/expr_ir.rs index 21ffe21bf..50606f185 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/ir/scalar/expr_ir.rs @@ -1,29 +1,20 @@ -//! Column-reference and scalar-operator vocabulary shared by the whole -//! canonical [`QueryExpr`](super::query_expr::QueryExpr) DAG. -//! -//! Issue #205: the scalar expression shapes (`Column`/`Literal`/`Compare`/…) -//! used to live in a separate, self-recursive `Expr` DAG here, reachable -//! from `QueryExpr` only through wrapper fields (`Predicate`, `ProjectItem`, -//! `SortKey`). They're variants of `QueryExpr` itself now — one recursive -//! DAG, not two type families joined by wrappers — generic over the same -//! column-reference state `C` the rest of `QueryExpr` already carries -//! (issue #179): [`ColumnRef`] (name-based, front-end-emitted) or -//! [`ColumnId`](super::schema::ColumnId) (positional, once bound). -//! -//! What's left here is the vocabulary those scalar variants are built from — -//! [`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`] — the **union** of what the two +//! Column-reference and scalar-operator vocabulary shared by the IR's scalar +//! expressions ([`crate::ir::ScalarExpr`]) and the front ends' unresolved +//! form: [`ColumnRef`] (name-based, front-end-emitted; positional +//! [`ColumnId`](crate::ir::schema::ColumnId) once bound), and [`ScalarValue`], +//! [`CompareOpKind`], [`ArithmeticOpKind`] — the **union** of what the two //! front ends need: PromQL contributes `Regex` / `NotRegex` (`=~` / `!~`); SQL //! contributes arithmetic, `CASE`, `IN`, `CAST`, `IS [NOT] NULL`, scalar //! function calls, and the `LIKE` / `ILIKE` comparison family. use serde::{Deserialize, Serialize}; -/// A name-based column reference — the front-end-emitted, unresolved state of -/// [`QueryExpr::Column`](super::query_expr::QueryExpr::Column) (`C = -/// ColumnRef`); the [`SchemaResolver`](super::schema_resolver::SchemaResolver) resolves it to a -/// positional [`ColumnId`](super::schema::ColumnId). This is a logical reference, -/// not schema metadata or a runtime data array. `SampleValue` names the implicit -/// PromQL sample column; `Wildcard` represents an all-columns/rows request. +/// A name-based column reference — the front-end-emitted, unresolved form of +/// [`ScalarExpr::Column`](crate::ir::ScalarExpr::Column); front-end name +/// resolution turns it into a positional [`ColumnId`](crate::ir::schema::ColumnId). +/// This is a logical reference, not schema metadata or a runtime data array. +/// `SampleValue` names the implicit PromQL sample column; `Wildcard` represents +/// an all-columns/rows request. #[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] pub enum ColumnRef { Named(String), diff --git a/crates/types/src/ir/scalar.rs b/crates/types/src/ir/scalar/mod.rs similarity index 97% rename from crates/types/src/ir/scalar.rs rename to crates/types/src/ir/scalar/mod.rs index c1a0b90a1..6877b3456 100644 --- a/crates/types/src/ir/scalar.rs +++ b/crates/types/src/ir/scalar/mod.rs @@ -3,20 +3,26 @@ //! //! A [`ScalarExpr`] never produces a table. It is owned by value by an operator //! field (`Filter.pred`, `ProjectItem.expr`, `SortKey.expr`, `HAVING`, window -//! arguments, relabel values) or by a [`super::QueryRoot::Scalar`] query root. +//! arguments, relabel values) or by a [`crate::ir::QueryRoot::Scalar`] query root. //! The only operator references inside a scalar tree are the explicit //! plan-reading variants (`PromqlScalarFromVector`, `ScalarSubquery`, `Exists`, //! `InSubquery`); every traversal of the operator DAG follows them. +pub mod column_resolution; +mod expr_ir; +pub mod scalar_type_rules; + +pub use column_resolution::{resolve_column_ref, resolve_column_refs, ResolveError}; +pub use expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; + use std::rc::Rc; use serde::{Deserialize, Serialize}; -use super::node::OperatorNode; +use crate::ir::operator::node::OperatorNode; +use crate::ir::scalar::scalar_type_rules::MapScalarFunction; +use crate::ir::schema::{ColumnId, DataType, Schema}; use crate::ir::SchemaDerivationError; -use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind, ScalarValue}; -use crate::pre_asap::scalar_type_rules::MapScalarFunction; -use crate::pre_asap::schema::{ColumnId, DataType, Schema}; /// Which language's numeric and comparison rules an expression follows. /// Both languages use `Float64`, so a result type alone does not preserve @@ -489,7 +495,8 @@ impl ScalarExpr { (to.clone(), *try_cast || nullable) } ScalarExpr::FunctionCall { name, args } => { - if let Some(arity) = crate::pre_asap::scalar_type_rules::promql_function_arity(name) + if let Some(arity) = + crate::ir::scalar::scalar_type_rules::promql_function_arity(name) { if args.len() != arity || args @@ -559,7 +566,7 @@ impl ScalarExpr { // `scalar(v)` is one float sample (NaN when the vector is not // exactly one series); a scalar subquery is its single column. ScalarExpr::PromqlScalarFromVector(node) => { - if node.result_kind != super::OperatorResultKind::InstantVector { + if node.result_kind != crate::ir::OperatorResultKind::InstantVector { return Err(signature("scalar() requires an instant vector")); } (DataType::Float64, false) @@ -603,7 +610,7 @@ fn common_scalar_type(a: &DataType, b: &DataType) -> Result Result<(), SchemaDerivationError> { - if node.result_kind == super::OperatorResultKind::Relation { + if node.result_kind == crate::ir::OperatorResultKind::Relation { Ok(()) } else { Err(signature("SQL subquery requires a relation")) @@ -611,7 +618,7 @@ fn relation(node: &OperatorNode) -> Result<(), SchemaDerivationError> { } fn scalar_subquery_field( node: &OperatorNode, -) -> Result<&crate::pre_asap::Field, SchemaDerivationError> { +) -> Result<&crate::ir::schema::Field, SchemaDerivationError> { relation(node)?; match node.schema.fields.as_slice() { [field] => Ok(field), @@ -747,9 +754,9 @@ pub fn element_access_type( #[cfg(test)] mod tests { use super::*; - use crate::ir::operator_properties::Source; + use crate::ir::operator::operator_properties::Source; + use crate::ir::schema::{Field, FieldDataType}; use crate::ir::{NonASAPOp, OperatorNode}; - use crate::pre_asap::schema::{Field, FieldDataType}; fn call(name: &str, args: Vec) -> ScalarExpr { ScalarExpr::FunctionCall { diff --git a/crates/types/src/ir/scalar/scalar_type_rules.rs b/crates/types/src/ir/scalar/scalar_type_rules.rs new file mode 100644 index 000000000..81be9e466 --- /dev/null +++ b/crates/types/src/ir/scalar/scalar_type_rules.rs @@ -0,0 +1,213 @@ +//! Shared type and nullability rules used to validate scalar expressions. +//! These rules do not evaluate expressions or define physical representations. +use crate::ir::schema::DataType; + +/// Names are resolved once against this closed builtin set; unknown functions +/// remain outside these type rules. Map access keeps the first duplicate key +/// and returns the value type's default when the key is absent. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MapScalarFunction { + Construct, + Concat, + Access, +} +impl MapScalarFunction { + pub fn from_name(name: &str) -> Option { + match name.to_ascii_lowercase().as_str() { + "map" => Some(Self::Construct), + "mapconcat" => Some(Self::Concat), + "asap_map_access" => Some(Self::Access), + _ => None, + } + } + pub fn output_type(self, args: &[(DataType, bool)]) -> Result<(DataType, bool), String> { + match self { + Self::Construct => { + let (pairs, remainder) = args.as_chunks::<2>(); + if !remainder.is_empty() { + return Err("map construction requires key/value pairs".into()); + } + let mut key = DataType::Null; + let mut value = DataType::Null; + let mut value_nullable = false; + for pair in pairs { + if pair[0].1 || pair[0].0 == DataType::Null { + return Err("map keys must be non-null".into()); + } + key = common_type(&key, &pair[0].0)?; + value = common_type(&value, &pair[1].0)?; + value_nullable |= pair[1].1 || pair[1].0 == DataType::Null; + } + Ok(( + DataType::Map { + key: Box::new(key), + value: Box::new(value), + value_nullable, + }, + false, + )) + } + Self::Concat => { + if args.is_empty() { + return Err("map concatenation requires at least one map".into()); + } + let mut key = DataType::Null; + let mut value = DataType::Null; + let mut value_nullable = false; + for (argument, nullable) in args { + if *nullable { + return Err("nullable map containers are unsupported".into()); + } + let DataType::Map { + key: k, + value: v, + value_nullable: n, + } = argument + else { + return Err("map concatenation requires map arguments".into()); + }; + key = common_type(&key, k)?; + value = common_type(&value, v)?; + value_nullable |= *n; + } + Ok(( + DataType::Map { + key: Box::new(key), + value: Box::new(value), + value_nullable, + }, + false, + )) + } + Self::Access => { + let [(map, map_nullable), (index, index_nullable)] = args else { + return Err("map access requires a map and key".into()); + }; + if *map_nullable { + return Err("nullable map containers are unsupported".into()); + } + let DataType::Map { + key, + value, + value_nullable, + } = map + else { + return Err("map access requires a map".into()); + }; + if **key == DataType::Null { + return Err("map lookup requires a concrete map key type".into()); + } + if *index != DataType::Null && common_type(key, index)? != **key { + return Err("map lookup key requires a lossy or unsupported coercion".into()); + } + Ok(( + (**value).clone(), + *value_nullable + || *index_nullable + || *index == DataType::Null + || **value == DataType::Null, + )) + } + } + } +} +fn common_type(left: &DataType, right: &DataType) -> Result { + if left == right || *right == DataType::Null { + return Ok(left.clone()); + } + if *left == DataType::Null { + return Ok(right.clone()); + } + Err(format!( + "incompatible map scalar types: {left:?} and {right:?}" + )) +} + +/// Closed, namespaced contracts for PromQL pointwise float functions. +/// Date functions consume Unix seconds; `timestamp` remains a sample-selection +/// operation because its operand is a sample timestamp rather than its value. +pub fn promql_function_arity(name: &str) -> Option { + Some(match name.strip_prefix("promql_")? { + "abs" | "ceil" | "floor" | "exp" | "ln" | "log2" | "log10" | "sqrt" | "sgn" | "sin" + | "cos" | "tan" | "asin" | "acos" | "atan" | "sinh" | "cosh" | "tanh" | "asinh" + | "acosh" | "atanh" | "deg" | "rad" | "minute" | "hour" | "day_of_week" + | "day_of_month" | "day_of_year" | "month" | "year" | "days_in_month" => 1, + "round" | "clamp_min" | "clamp_max" => 2, + "clamp" => 3, + _ => return None, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn empty_map_is_bottom_typed_and_concat_resolves_it() { + let empty = MapScalarFunction::Construct.output_type(&[]).unwrap(); + assert_eq!( + empty, + ( + DataType::Map { + key: Box::new(DataType::Null), + value: Box::new(DataType::Null), + value_nullable: false + }, + false + ) + ); + assert!(MapScalarFunction::Access + .output_type(&[empty.clone(), (DataType::Utf8, false)]) + .is_err()); + let concrete = MapScalarFunction::Construct + .output_type(&[(DataType::Utf8, false), (DataType::Int64, false)]) + .unwrap(); + assert_eq!( + MapScalarFunction::Concat + .output_type(&[empty, concrete.clone()]) + .unwrap(), + concrete.clone() + ); + assert_eq!( + MapScalarFunction::Access + .output_type(&[concrete, (DataType::Utf8, false)]) + .unwrap(), + (DataType::Int64, false) + ); + } + #[test] + fn nullable_lookup_and_invalid_signatures_are_explicit() { + let map = MapScalarFunction::Construct + .output_type(&[(DataType::Utf8, false), (DataType::Int64, true)]) + .unwrap(); + assert_eq!( + MapScalarFunction::Access + .output_type(&[map, (DataType::Utf8, false)]) + .unwrap(), + (DataType::Int64, true) + ); + let nonnull = MapScalarFunction::Construct + .output_type(&[(DataType::Utf8, false), (DataType::Int64, false)]) + .unwrap(); + assert_eq!( + MapScalarFunction::Access + .output_type(&[nonnull, (DataType::Utf8, true)]) + .unwrap(), + (DataType::Int64, true) + ); + assert!(MapScalarFunction::Construct + .output_type(&[(DataType::Utf8, true), (DataType::Int64, false)]) + .is_err()); + assert!(MapScalarFunction::Concat + .output_type(&[(DataType::Int64, false)]) + .is_err()); + // ClickHouse can choose Variant(Float64, Int64), not lossless Float64. + assert!(MapScalarFunction::Construct + .output_type(&[ + (DataType::Utf8, false), + (DataType::Int64, false), + (DataType::Utf8, false), + (DataType::Float64, false), + ]) + .is_err()); + } +} diff --git a/crates/types/src/ir/aggregate_schema.rs b/crates/types/src/ir/schema/aggregate_schema.rs similarity index 83% rename from crates/types/src/ir/aggregate_schema.rs rename to crates/types/src/ir/schema/aggregate_schema.rs index 10b33d2ae..c8f7239cb 100644 --- a/crates/types/src/ir/aggregate_schema.rs +++ b/crates/types/src/ir/schema/aggregate_schema.rs @@ -5,9 +5,11 @@ //! aggregate result; a PromQL per-series range reduction preserves labels //! and produces a Float64 sample value. This module derives schemas, not //! aggregate values or summary candidates. -use super::operator_properties::*; -use super::SchemaDerivationError; -use crate::pre_asap::{AggIntent, ColumnId, ColumnRef, DataType, Field, FieldDataType, Schema}; +use crate::ir::operator::operator_properties::*; +use crate::ir::operator::AggIntent; +use crate::ir::scalar::ColumnRef; +use crate::ir::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; +use crate::ir::SchemaDerivationError; /// Output schema of a *per-series* window/range reduction (`rate`/`increase`, /// or an `*_over_time` reducer under a time `Window`). Such a reduction emits /// one value per series, so every label column of `input` is preserved and only @@ -20,7 +22,7 @@ fn per_series_reduction_schema( let vi = if let Some(index) = agg.input_cols().first() { *index } else { - crate::pre_asap::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, input) + crate::ir::scalar::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, input) .map_err(|error| SchemaDerivationError::InvalidSampleColumn(error.to_string()))? }; if !matches!( @@ -90,6 +92,10 @@ pub fn aggregate_output_schema( return without_output_schema(in_schema, by.keys(), measures, output_names); } + if let [AggIntent::TopK { .. }] = measures { + return ranked_rows_schema(in_schema, by.keys()); + } + let mut out_cols: Vec = Vec::with_capacity(by.len() + measures.len()); for &id in by.keys() { let c = in_schema @@ -101,10 +107,12 @@ pub fn aggregate_output_schema( ))?; out_cols.push(c.clone()); } - let value_col_idx = - crate::pre_asap::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema) - .ok() - .or_else(|| (0..in_schema.fields.len()).find(|i| !by.contains(i))); + let value_col_idx = crate::ir::scalar::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + in_schema, + ) + .ok() + .or_else(|| (0..in_schema.fields.len()).find(|i| !by.contains(i))); let probe = value_col_idx .and_then(|i| in_schema.fields.get(i)) .cloned() @@ -177,6 +185,63 @@ pub fn aggregate_output_schema( }) } +/// A top-k returns the selected rows, the shape every realization of it +/// produces (exact Sort → Limit, a sketch readout): the partition keys, the +/// ranked item's identity, and its ranking `value`. A PromQL item is its +/// series: the identity column when rows carry it, else the encoded label set +/// a sketch readout returns. A SQL item is every other non-time column. +fn ranked_rows_schema( + in_schema: &Schema, + keys: &[ColumnId], +) -> Result { + use crate::ir::schema::PROMQL_SERIES_IDENTITY; + let field = |id: ColumnId| { + in_schema + .fields + .get(id) + .cloned() + .ok_or(SchemaDerivationError::InvalidGroupByColumn( + id, + in_schema.fields.len(), + )) + }; + let mut fields = keys + .iter() + .map(|&id| field(id)) + .collect::, _>>()?; + let identity = in_schema + .fields + .iter() + .position(|f| f.name == PROMQL_SERIES_IDENTITY); + match identity { + Some(id) if in_schema.has_promql_series_identity() => fields.push(field(id)?), + _ if !in_schema.closed => { + fields.push(Field::plain(PROMQL_SERIES_IDENTITY, DataType::Utf8, false)) + } + _ => { + let value = crate::ir::scalar::column_resolution::resolve_column_ref( + &ColumnRef::SampleValue, + in_schema, + ) + .ok() + .or_else(|| (0..in_schema.fields.len()).rfind(|i| !keys.contains(i))); + for id in 0..in_schema.fields.len() { + if !keys.contains(&id) && Some(id) != value && Some(id) != in_schema.time_index { + fields.push(field(id)?); + } + } + } + } + let key = (0..fields.len()).collect(); + fields.push(Field::plain("value", DataType::Float64, false)); + Ok(Schema { + fields, + time_index: None, + unique_keys: vec![key], + closed: true, + }) +} + /// Output schema of a `without(excluded)` aggregate: the kept labels (every /// input label column except the `excluded` positions, the time axis, and the /// sample-value column) followed by the aggregate output column(s). Unlike the @@ -200,7 +265,7 @@ fn without_output_schema( let mut out_cols: Vec = Vec::new(); for (i, col) in in_schema.fields.iter().enumerate() { let is_time = in_schema.time_index == Some(i); - let is_value = crate::pre_asap::column_resolution::resolve_column_ref( + let is_value = crate::ir::scalar::column_resolution::resolve_column_ref( &ColumnRef::SampleValue, in_schema, ) diff --git a/crates/types/src/ir/error.rs b/crates/types/src/ir/schema/error.rs similarity index 83% rename from crates/types/src/ir/error.rs rename to crates/types/src/ir/schema/error.rs index 4b6849965..d4f3b05b5 100644 --- a/crates/types/src/ir/error.rs +++ b/crates/types/src/ir/schema/error.rs @@ -3,7 +3,7 @@ //! [`SchemaDerivationError`] distinguishes invalid scalar signatures, out-of-range //! grouping columns, empty concatenations, and invalid sample columns. //! Structural DAG and execution-timing validation have separate error types. -use crate::pre_asap::ColumnId; +use crate::ir::schema::ColumnId; use thiserror::Error; /// Errors from schema and type derivation over an operator DAG. #[derive(Debug, Error)] @@ -16,4 +16,6 @@ pub enum SchemaDerivationError { EmptyConcat, #[error("invalid per-series sample column: {0}")] InvalidSampleColumn(String), + #[error("invalid summary coverage: {0}")] + Coverage(#[from] crate::ir::properties::summary_coverage::CoverageError), } diff --git a/crates/types/src/pre_asap/schema.rs b/crates/types/src/ir/schema/mod.rs similarity index 84% rename from crates/types/src/pre_asap/schema.rs rename to crates/types/src/ir/schema/mod.rs index 77c8ebe92..3ea89792d 100644 --- a/crates/types/src/pre_asap/schema.rs +++ b/crates/types/src/ir/schema/mod.rs @@ -13,13 +13,21 @@ #![allow(dead_code)] -use serde::{Deserialize, Serialize}; - -use crate::post_asap::sketch::{ - ExactKind, ExactParams, GroupingStrategy, SamplingKind, SamplingParams, SketchKind, - StatModelKind, StatModelParams, WaveletKind, WaveletParams, +pub mod aggregate_schema; +pub mod error; +pub mod state_type; + +pub use aggregate_schema::aggregate_output_schema; +pub use error::SchemaDerivationError; +pub use state_type::{ + default_hydra_params, hydra_kind_for, EntityIdentity, ExactKind, ExactParams, GroupingStrategy, + HydraKind, HydraParams, NonNegativeWeightProof, SamplingKind, SamplingParams, SketchAlgorithm, + SketchCategory, SketchKind, SketchParams, SketchStatistic, StatModelKind, StatModelParams, + SummaryInputExpr, SummaryUpdate, WaveletKind, WaveletParams, WeightDomain, }; +use serde::{Deserialize, Serialize}; + /// Zero-based position of a column in a particular operator's input or output. /// /// During planning, the position indexes [`Schema::fields`] to obtain metadata; @@ -325,82 +333,6 @@ impl TryFrom for Schema { /// label map. `$` cannot occur in a user PromQL label name. pub const PROMQL_SERIES_IDENTITY: &str = "$promql_series_identity"; -/// Resolve a PromQL root to rows carrying [`PROMQL_SERIES_IDENTITY`] before -/// candidate search. `closed` describes physical columns here: the final -/// column contains every dynamic source label. It does not assert that the -/// query's projected labels are the full label set. -/// -/// This realization supports explicit `by` grouping and per-series computation. -/// Operators that rewrite or implicitly match dynamic label sets require their -/// own realization; they must not accidentally treat the opaque identity as a -/// user label or silently discard it. -pub fn with_promql_series_identity(root: &super::QueryExpr) -> Result { - use super::{QueryExpr, Source}; - use std::rc::Rc; - let mut root = root.clone(); - fn visit(node: &mut QueryExpr) -> Result<(), String> { - match node { - QueryExpr::Scan { - source: Source::TimeSeries { .. }, - schema, - .. - } => { - if schema - .fields - .iter() - .any(|column| column.name == PROMQL_SERIES_IDENTITY) - { - return Err("source already contains a physical series identity".into()); - } - if schema.closed { - return Err("dynamic series identity requires an open PromQL source".into()); - } - schema - .fields - .push(Field::plain(PROMQL_SERIES_IDENTITY, DataType::Utf8, false)); - schema.closed = true; - Ok(()) - } - QueryExpr::TimeRange { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::TimeShift { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - | QueryExpr::PromqlScalarFromVector(child) - | QueryExpr::PromqlRelabel { child, .. } => visit(Rc::make_mut(child)), - // Constants read no series. - QueryExpr::PromqlScalarBridge(_) - | QueryExpr::EvalTimestamp - | QueryExpr::Literal(super::ScalarValue::Float64(_)) => Ok(()), - QueryExpr::PromqlVectorFromScalar(child) => visit(Rc::make_mut(child)), - QueryExpr::BinaryOp { lhs, rhs, .. } => { - visit(Rc::make_mut(lhs))?; - visit(Rc::make_mut(rhs)) - } - QueryExpr::Concat { children, .. } => { - for child in children { - visit(child)?; - } - Ok(()) - } - QueryExpr::Aggregate { child, .. } => visit(Rc::make_mut(child)), - QueryExpr::Sort { - child, - partition_by, - .. - } => { - if partition_by.is_without() { - return Err("dynamic without ranking requires label-set projection".into()); - } - visit(Rc::make_mut(child)) - } - _ => Err("operator has no dynamic series-identity realization".into()), - } - } - visit(&mut root)?; - root.output_schema().map_err(|error| error.to_string())?; - Ok(root) -} - impl Schema { pub fn has_promql_series_identity(&self) -> bool { self.closed @@ -605,12 +537,4 @@ mod tests { assert_eq!(back, c); assert_eq!(back.table.as_deref(), Some("hosts")); } - // Direct scalar literals remain valid vector inputs when series typing runs. - #[test] - fn series_identity_accepts_direct_vector_literal() { - let root = super::super::QueryExpr::PromqlVectorFromScalar(std::rc::Rc::new( - super::super::QueryExpr::Literal(super::super::ScalarValue::Float64(1.0)), - )); - assert!(with_promql_series_identity(&root).is_ok()); - } } diff --git a/crates/types/src/post_asap/sketch.rs b/crates/types/src/ir/schema/state_type.rs similarity index 93% rename from crates/types/src/post_asap/sketch.rs rename to crates/types/src/ir/schema/state_type.rs index a5e1edcc1..8146b652f 100644 --- a/crates/types/src/post_asap/sketch.rs +++ b/crates/types/src/ir/schema/state_type.rs @@ -1,12 +1,30 @@ +//! Summary-state types carried by [`FieldDataType`](crate::ir::schema::FieldDataType). +//! +//! Where an intent ([`crate::ir::operator::AggIntent`]) says *what* to compute +//! ("a quantile to ε accuracy"), these types say *how* a summary realizes it: +//! the family, kind/algorithm and parameters are committed — one +//! `(Kind, Params)` pair per family ([`ExactKind`]/[`ExactParams`], +//! [`SamplingKind`]/[`SamplingParams`], [`WaveletKind`]/[`WaveletParams`], +//! [`StatModelKind`]/[`StatModelParams`]). The `Sketch` family nests a third +//! level, [`SketchKind`] (quantile/cardinality/frequency/top-k), carrying the +//! committed [`SketchAlgorithm`] and [`SketchParams`], because it is the one +//! family with more than one algorithm per purpose. +//! +//! [`GroupingStrategy`] is a second, orthogonal axis: how many physical +//! instances of a summary exist across a grouped aggregate's `by` +//! subpopulations (per-subpopulation vs. one shared Hydra instance — see +//! `asap_logical_optimizer::pass1::grouping`). It rides on `ASAPOp::SummaryAgg` and on +//! sketch-valued edge types. + use serde::{Deserialize, Serialize}; -use crate::pre_asap::ColumnRef; +use crate::ir::scalar::ColumnRef; // ── Exact accumulators ────────────────────────────────────────────────────── /// An exact, mergeable accumulator family — zero approximation error. The /// partial state built for one of these *is* the answer; no -/// `SummaryEstimate` readout is needed to get a value out of it. +/// `SummaryEstimate` evaluation is needed to get a value out of it. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] pub enum ExactKind { /// Exact sum accumulator (mergeable by addition). @@ -48,7 +66,7 @@ pub enum ExactParams { /// [`SketchKind::new`] is where that classification is made. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] pub enum SketchAlgorithm { - /// Universal frequency-vector summary with shared statistic readouts. + /// Universal frequency-vector summary with shared statistic evaluations. UnivMon, /// KLL quantile sketch (mergeable, ε-accurate rank queries). Kll, @@ -134,7 +152,7 @@ pub enum SketchCategory { /// quantile-style, cardinality-style, frequency-style, or heavy-hitter/ /// top-k-style estimation — together with the concrete [`SketchAlgorithm`] /// and [`SketchParams`] realizing it. Sits between -/// [`FieldDataType::Sketch`](super::schema::FieldDataType::Sketch) +/// [`FieldDataType::Sketch`](crate::ir::schema::FieldDataType::Sketch) /// (the `Sketch` family as a whole, sibling to `Sample`/`Wavelet`/ /// `StatModel`) and the bare algorithm: `Kll` vs. `DDSketch` is a choice /// *within* `Quantile`, not a choice *of* `SketchKind` — every `Quantile` @@ -143,7 +161,7 @@ pub enum SketchCategory { /// [`SketchKind::new`] is the one place `(SketchAlgorithm, SketchParams)` /// pairs get classified into a category; construct through it rather than /// naming a variant directly, so a new algorithm can't drift out of sync -/// with its category. See `asap_aware_mapping::summary_candidates` for the +/// with its category. See `asap_logical_optimizer::summary_candidates` for the /// `AggIntent -> [SketchAlgorithm]` candidate list this ultimately groups. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct SketchKind { @@ -304,7 +322,7 @@ pub enum StatModelParams { // family/kind answers an intent — any family could in principle grow its own // per-subpopulation vs. shared-multi-subpopulation variant, so it is a // second, independent axis, not a member of any one family's own kind -// vocabulary. See `asap_aware_mapping::grouping`'s module docs for where this +// vocabulary. See `asap_logical_optimizer::pass1::grouping`'s module docs for where this // axis actually plugs into the post-ASAP IR and the legality rules gating // when `SharedMultiSubpopulation` is offered as a candidate at all. @@ -345,7 +363,7 @@ pub enum StatModelParams { /// really the universal-sketch composition (L layers of Count-Sketch plus a /// heavy-hitter heap, Theorems 1+2 combined) estimating entropy/L1-norm/ /// L2-norm/cardinality/frequency-moments as one instance. Standalone UnivMon -/// and its frequency readouts are represented here, but sharing a Hydra +/// and its frequency evaluations are represented here, but sharing a Hydra /// grid across populations still needs its own collision/error contract; /// standalone support does not establish that contract. #[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] @@ -393,7 +411,7 @@ pub enum HydraParams { /// subpopulation. Correctly sizing this against an estimated /// subpopulation cardinality is a cost-model concern — out of scope /// for the legality axis this type lives on (see - /// `asap_aware_mapping::grouping`'s module docs) — so this is + /// `asap_logical_optimizer::pass1::grouping`'s module docs) — so this is /// deliberately not derived from any cardinality estimate here. shared_buckets: u32, }, @@ -450,7 +468,7 @@ pub fn hydra_kind_for(algorithm: &SketchAlgorithm) -> Option { /// belong to the [`SketchAlgorithm`] `kind` wraps: a caller bug, since /// [`hydra_kind_for`] and the algorithm a `SketchParams` came from must /// agree; callers that got both from the same already-ranked -/// `Realization` (as `asap_aware_mapping::grouping` does) cannot hit +/// `Realization` (as `asap_logical_optimizer::pass1::grouping` does) cannot hit /// this. /// /// This function is generic over which inner sketch type `kind` wraps @@ -500,7 +518,7 @@ pub fn default_hydra_params( /// enums themselves, for exactly the reason explained in this section's /// module docs above. /// -/// Carried both on `SummaryExpr::SummaryAgg` (where planning consults it) +/// Carried both on `ASAPOp::SummaryAgg` (where planning consults it) /// and on sketch-valued `FieldDataType` edges (where it prevents /// incompatible shared and independent physical states from type-checking /// as merge-compatible). @@ -590,7 +608,8 @@ pub enum SummaryInputExpr { EntityIdentity(EntityIdentity), } -/// What to extract from a built summary. Carried by `SummaryEstimate`. +/// The statistic computed from summary state by `SummaryEstimate`, for example +/// `Quantile { q: 0.99 }`. This is a result operation, not a workload query. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub enum SketchStatistic { /// sqrt(sum_v frequency(v)^2), not the norm of numeric input values. @@ -605,8 +624,8 @@ pub enum SketchStatistic { /// `value: Some(v)` is a per-item point lookup (e.g. /// `count(cms_metric{item="checkout"})` — `key` is `item`, `value` is /// `"checkout"`). `value` is carried here rather than resolved by the - /// `SummaryExecutor` from a `Filter` predicate because `readout`'s - /// trait signature has no DAG access — see `CostModel::readout_extension`. + /// `SummaryExecutor` from a `Filter` predicate because `evaluation`'s + /// signature has no dag access. PointCount { key: ColumnRef, value: Option, @@ -622,7 +641,7 @@ mod tests { use super::*; #[test] - fn keyed_summary_input_and_topk_readout_round_trip() { + fn keyed_summary_input_and_topk_evaluation_round_trip() { for input in [ SummaryUpdate { item: Some(SummaryInputExpr::EntityIdentity( diff --git a/crates/types/src/ir/schema_support.rs b/crates/types/src/ir/schema_support.rs new file mode 100644 index 000000000..415c9801f --- /dev/null +++ b/crates/types/src/ir/schema_support.rs @@ -0,0 +1,100 @@ +//! Series-identity realization for the unified dag. +use crate::ir::schema::*; +/// Resolve a PromQL root to rows carrying [`PROMQL_SERIES_IDENTITY`] before +/// candidate search. `closed` describes physical columns here: the final +/// column contains every dynamic source label. It does not assert that the +/// query's projected labels are the full label set. +/// +/// This realization supports explicit `by` grouping and per-series computation. +/// Operators that rewrite or implicitly match dynamic label sets require their +/// own realization; they must not accidentally treat the opaque identity as a +/// user label or silently discard it. +pub fn with_promql_series_identity( + root: &std::rc::Rc, +) -> Result, String> { + use crate::ir::operator::Source; + use crate::ir::{NonASAPOp, Operator, OperatorNode}; + use std::{collections::HashMap, rc::Rc}; + fn visit( + node: &Rc, + memo: &mut HashMap<*const OperatorNode, Rc>, + ) -> Result, String> { + if let Some(found) = memo.get(&Rc::as_ptr(node)) { + return Ok(Rc::clone(found)); + } + let mut error = None; + let mut operator = node + .operator + .map_children(|child| match visit(child, memo) { + Ok(child) => child, + Err(e) => { + error = Some(e); + Rc::clone(child) + } + }); + if let Some(error) = error { + return Err(error); + } + match &mut operator { + Operator::NonASAP(NonASAPOp::Scan { + source: Source::TimeSeries { .. }, + schema, + .. + }) => { + if schema + .fields + .iter() + .any(|field| field.name == PROMQL_SERIES_IDENTITY) + { + if !schema.has_promql_series_identity() { + return Err("invalid physical series identity".into()); + } + memo.insert(Rc::as_ptr(node), Rc::clone(node)); + return Ok(Rc::clone(node)); + } + if schema.closed { + return Err("dynamic series identity requires an open PromQL source".into()); + } + schema.fields.push(Field::new( + PROMQL_SERIES_IDENTITY, + FieldDataType::Plain(DataType::Utf8), + false, + )); + schema.closed = true; + // A series has at most one sample per timestamp, so the full + // identity and the timestamp key each row. CSE's sharing + // legality reads this key. + if let Some(time) = schema.time_index { + schema.add_unique_key(vec![schema.fields.len() - 1, time]); + } + } + Operator::NonASAP(NonASAPOp::Sort { partition_by, .. }) + if partition_by.is_without() => + { + return Err("dynamic without ranking requires label-set projection".into()); + } + Operator::NonASAP( + NonASAPOp::TimeRange { .. } + | NonASAPOp::Limit { .. } + | NonASAPOp::Project { .. } + | NonASAPOp::Filter { .. } + | NonASAPOp::TimeShift { .. } + | NonASAPOp::PromqlSubquery { .. } + | NonASAPOp::PromqlRelabel { .. } + | NonASAPOp::PromqlVectorFromScalar(_) + | NonASAPOp::BinaryOp { .. } + | NonASAPOp::Concat { .. } + | NonASAPOp::Aggregate { .. } + | NonASAPOp::Sort { .. }, + ) => {} + _ => return Err("operator has no dynamic series-identity realization".into()), + } + let mut rebuilt = OperatorNode::new(operator).map_err(|e| e.to_string())?; + rebuilt.guarantee = node.guarantee.clone(); + rebuilt.timing = node.timing; + let rebuilt = Rc::new(rebuilt); + memo.insert(Rc::as_ptr(node), Rc::clone(&rebuilt)); + Ok(rebuilt) + } + visit(root, &mut HashMap::new()) +} diff --git a/crates/types/src/ir/wire.rs b/crates/types/src/ir/wire.rs new file mode 100644 index 000000000..eb6dd6423 --- /dev/null +++ b/crates/types/src/ir/wire.rs @@ -0,0 +1,746 @@ +//! Flat operator/scalar payloads shared by logical transport. +use std::rc::Rc; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; + +use crate::ir::operator::agg_intent::AggIntent; +use crate::ir::operator::asap::ASAPOp; +use crate::ir::operator::maintained_population::{MaintainedPopulation, PopulationStatistic}; +use crate::ir::operator::node::{Operator, OperatorNode}; +use crate::ir::operator::non_asap::{BinaryOperator, NonASAPOp, TimeRangeKind}; +use crate::ir::operator::operator_properties::{ + ConcatDiscriminatorKey, GroupKeys, InfoMatcher, JoinKind, Reduction, RelationalSetOpKind, + SampleKind, Source, TimeShift, WindowFrame, WindowFuncKind, +}; +use crate::ir::scalar::{ArithmeticOpKind, CompareOpKind, ScalarValue}; +use crate::ir::scalar::{ExprSemantics, Predicate, ProjectItem, ScalarExpr, SortKey}; +use crate::ir::schema::state_type::{GroupingStrategy, SketchStatistic, SummaryUpdate}; +use crate::ir::schema::{ColumnId, DataType, FieldDataType, Schema}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum EdgeRole { + Input, + Left, + Right, + /// The consumer reads the producer from inside one of its scalar + /// expressions (`scalar(v)`, a scalar subquery, `EXISTS`, `IN`). + ScalarRef, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub enum GroupingEdgeCompatibility { + Identical, + ConsumerCoarsensProducer, + Incompatible, + NotApplicable, +} + +/// Stable identity of a node within one exported logical ASAP DAG. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct LogicalASAPNodeId(pub u32); + +// ── Wire mirrors of the scalar language ────────────────────────────────── + +/// [`ScalarExpr`] with every operator reference replaced by the id of the +/// exported node (connected to the owner by an [`EdgeRole::ScalarRef`] edge). +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub enum WireScalarExpr { + Column(ColumnId), + Literal(ScalarValue), + Negative { + expr: Box, + semantics: ExprSemantics, + }, + Compare { + left: Box, + op: CompareOpKind, + right: Box, + semantics: ExprSemantics, + }, + BoolAnd(Vec), + BoolOr(Vec), + Not(Box), + IsNull(Box), + IsNotNull(Box), + Cast { + expr: Box, + to: DataType, + try_cast: bool, + }, + InList { + expr: Box, + list: Vec, + negated: bool, + }, + FunctionCall { + name: String, + args: Vec, + }, + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + semantics: ExprSemantics, + }, + Case { + operand: Option>, + branches: Vec<(WireScalarExpr, WireScalarExpr)>, + else_expr: Option>, + }, + CurrentTimestamp, + EvalTimestamp, + PromqlScalarFromVector(LogicalASAPNodeId), + ScalarSubquery(LogicalASAPNodeId), + Exists { + subquery: LogicalASAPNodeId, + negated: bool, + }, + InSubquery { + expr: Box, + subquery: LogicalASAPNodeId, + negated: bool, + }, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WirePredicate(pub WireScalarExpr); + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WireProjectItem { + pub alias: Option, + pub expr: WireScalarExpr, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct WireSortKey { + pub expr: WireScalarExpr, + pub ascending: bool, + pub nulls_first: bool, +} + +impl WireScalarExpr { + /// Explicit producer IDs recursively referenced by this scalar tree. + pub fn operator_refs(&self) -> Vec { + fn collect(expr: &WireScalarExpr, out: &mut Vec) { + use WireScalarExpr::*; + match expr { + PromqlScalarFromVector(id) | ScalarSubquery(id) => out.push(*id), + Exists { subquery, .. } => out.push(*subquery), + InSubquery { expr, subquery, .. } => { + out.push(*subquery); + collect(expr, out); + } + Negative { expr, .. } + | Not(expr) + | IsNull(expr) + | IsNotNull(expr) + | Cast { expr, .. } => collect(expr, out), + Compare { left, right, .. } | Arithmetic { left, right, .. } => { + collect(left, out); + collect(right, out); + } + BoolAnd(args) | BoolOr(args) | FunctionCall { args, .. } => { + for expr in args { + collect(expr, out); + } + } + InList { expr, list, .. } => { + collect(expr, out); + for expr in list { + collect(expr, out); + } + } + Case { + operand, + branches, + else_expr, + } => { + if let Some(expr) = operand { + collect(expr, out); + } + for (when, then) in branches { + collect(when, out); + collect(then, out); + } + if let Some(expr) = else_expr { + collect(expr, out); + } + } + Column(_) | Literal(_) | CurrentTimestamp | EvalTimestamp => {} + } + } + let mut out = Vec::new(); + collect(self, &mut out); + out + } + + /// Mirror `expr`, resolving every operator reference through `id_of`. + pub fn from_expr( + expr: &ScalarExpr, + id_of: &mut impl FnMut(&Rc) -> LogicalASAPNodeId, + ) -> Self { + fn boxed( + e: &ScalarExpr, + id_of: &mut impl FnMut(&Rc) -> LogicalASAPNodeId, + ) -> Box { + Box::new(WireScalarExpr::from_expr(e, id_of)) + } + fn list( + es: &[ScalarExpr], + id_of: &mut impl FnMut(&Rc) -> LogicalASAPNodeId, + ) -> Vec { + es.iter() + .map(|e| WireScalarExpr::from_expr(e, id_of)) + .collect() + } + match expr { + ScalarExpr::Column(id) => WireScalarExpr::Column(*id), + ScalarExpr::Literal(v) => WireScalarExpr::Literal(v.clone()), + ScalarExpr::Negative { expr, semantics } => WireScalarExpr::Negative { + expr: boxed(expr, id_of), + semantics: *semantics, + }, + ScalarExpr::Compare { + left, + op, + right, + semantics, + } => WireScalarExpr::Compare { + left: boxed(left, id_of), + op: op.clone(), + right: boxed(right, id_of), + semantics: *semantics, + }, + ScalarExpr::BoolAnd(parts) => WireScalarExpr::BoolAnd(list(parts, id_of)), + ScalarExpr::BoolOr(parts) => WireScalarExpr::BoolOr(list(parts, id_of)), + ScalarExpr::Not(e) => WireScalarExpr::Not(boxed(e, id_of)), + ScalarExpr::IsNull(e) => WireScalarExpr::IsNull(boxed(e, id_of)), + ScalarExpr::IsNotNull(e) => WireScalarExpr::IsNotNull(boxed(e, id_of)), + ScalarExpr::Cast { expr, to, try_cast } => WireScalarExpr::Cast { + expr: boxed(expr, id_of), + to: to.clone(), + try_cast: *try_cast, + }, + ScalarExpr::InList { + expr, + list: items, + negated, + } => WireScalarExpr::InList { + expr: boxed(expr, id_of), + list: list(items, id_of), + negated: *negated, + }, + ScalarExpr::FunctionCall { name, args } => WireScalarExpr::FunctionCall { + name: name.clone(), + args: list(args, id_of), + }, + ScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } => WireScalarExpr::Arithmetic { + op: op.clone(), + left: boxed(left, id_of), + right: boxed(right, id_of), + semantics: *semantics, + }, + ScalarExpr::Case { + operand, + branches, + else_expr, + } => WireScalarExpr::Case { + operand: operand.as_ref().map(|e| boxed(e, id_of)), + branches: branches + .iter() + .map(|(w, t)| (Self::from_expr(w, id_of), Self::from_expr(t, id_of))) + .collect(), + else_expr: else_expr.as_ref().map(|e| boxed(e, id_of)), + }, + ScalarExpr::CurrentTimestamp => WireScalarExpr::CurrentTimestamp, + ScalarExpr::EvalTimestamp => WireScalarExpr::EvalTimestamp, + ScalarExpr::PromqlScalarFromVector(node) => { + WireScalarExpr::PromqlScalarFromVector(id_of(node)) + } + ScalarExpr::ScalarSubquery(node) => WireScalarExpr::ScalarSubquery(id_of(node)), + ScalarExpr::Exists { subquery, negated } => WireScalarExpr::Exists { + subquery: id_of(subquery), + negated: *negated, + }, + ScalarExpr::InSubquery { + expr, + subquery, + negated, + } => WireScalarExpr::InSubquery { + expr: boxed(expr, id_of), + subquery: id_of(subquery), + negated: *negated, + }, + } + } +} + +impl WirePredicate { + fn from_pred( + p: &Predicate, + id_of: &mut impl FnMut(&Rc) -> LogicalASAPNodeId, + ) -> Self { + WirePredicate(WireScalarExpr::from_expr(&p.0, id_of)) + } +} + +impl WireSortKey { + fn from_keys( + keys: &[SortKey], + id_of: &mut impl FnMut(&Rc) -> LogicalASAPNodeId, + ) -> Vec { + keys.iter() + .map(|k| WireSortKey { + expr: WireScalarExpr::from_expr(&k.expr, id_of), + ascending: k.ascending, + nulls_first: k.nulls_first, + }) + .collect() + } +} + +// ── Wire mirror of the non-ASAP operator vocabulary ────────────────────── + +/// [`NonASAPOp`] without its child fields (children are edges) and with +/// every scalar expression mirrored as [`WireScalarExpr`]. Fields named +/// `kind` in the IR are renamed so they do not collide with the variant tag. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case")] +pub enum NonASAPOpKind { + Scan { + source: Source, + #[serde(default)] + predicates: Vec, + schema: Schema, + }, + Values { + rows: Vec>, + schema: Schema, + }, + Filter { + pred: WirePredicate, + }, + Project { + cols: Vec, + #[serde(default)] + qualifier: Option, + }, + Aggregate { + reduction: Reduction, + measures: Vec, + #[serde(default)] + output_names: Vec, + #[serde(default)] + filters: Vec>, + #[serde(default)] + having: Option, + }, + Join { + join_kind: JoinKind, + pred: WirePredicate, + }, + SetOp { + set_kind: RelationalSetOpKind, + all: bool, + }, + Concat { + #[serde(default)] + discriminator_unique_key: Option, + }, + Dedup { + cols: Vec, + }, + Sort { + keys: Vec, + #[serde(default)] + partition_by: GroupKeys, + }, + Limit { + n: Option, + offset: usize, + #[serde(default)] + partition_by: GroupKeys, + }, + BinaryOp { + operator: BinaryOperator, + #[serde(default)] + return_bool: bool, + }, + #[serde(rename = "sql_window_func")] + SQLWindowFunc { + func: WindowFuncKind, + args: Vec, + partition_by: GroupKeys, + order_by: Vec, + #[serde(default)] + frame: Option, + output_name: String, + }, + TimeRange { + range: Duration, + range_kind: TimeRangeKind, + }, + TimeShift { + shift: TimeShift, + }, + PromqlVectorFromScalar { + expr: WireScalarExpr, + }, + PromqlRelabel { + dst: String, + value: WireScalarExpr, + }, + PromqlInfoEnrich { + #[serde(default)] + selector: Vec, + }, + PromqlSeriesSample { + #[serde(default)] + by: GroupKeys, + sample_kind: SampleKind, + }, + PromqlSubquery { + range: Duration, + #[serde(default)] + resolution: Option, + }, +} + +impl NonASAPOpKind { + /// Mirror `op`, resolving every operator node its scalar expressions + /// reference through `id_of`. + pub fn from_op( + op: &NonASAPOp, + id_of: &mut impl FnMut(&Rc) -> LogicalASAPNodeId, + ) -> Self { + use NonASAPOp as Op; + match op { + Op::Scan { + source, + predicates, + schema, + } => NonASAPOpKind::Scan { + source: source.clone(), + predicates: predicates + .iter() + .map(|p| WirePredicate::from_pred(p, id_of)) + .collect(), + schema: schema.clone(), + }, + Op::Values { rows, schema } => NonASAPOpKind::Values { + rows: rows + .iter() + .map(|row| { + row.iter() + .map(|e| WireScalarExpr::from_expr(e, id_of)) + .collect() + }) + .collect(), + schema: schema.clone(), + }, + Op::Filter { pred, .. } => NonASAPOpKind::Filter { + pred: WirePredicate::from_pred(pred, id_of), + }, + Op::Project { + cols, qualifier, .. + } => NonASAPOpKind::Project { + cols: cols + .iter() + .map(|ProjectItem { alias, expr }| WireProjectItem { + alias: alias.clone(), + expr: WireScalarExpr::from_expr(expr, id_of), + }) + .collect(), + qualifier: qualifier.clone(), + }, + Op::Aggregate { + reduction, + measures, + output_names, + filters, + having, + .. + } => NonASAPOpKind::Aggregate { + reduction: reduction.clone(), + measures: measures.clone(), + output_names: output_names.clone(), + filters: filters + .iter() + .map(|p| p.as_ref().map(|p| WirePredicate::from_pred(p, id_of))) + .collect(), + having: having.as_ref().map(|p| WirePredicate::from_pred(p, id_of)), + }, + Op::Join { kind, pred, .. } => NonASAPOpKind::Join { + join_kind: kind.clone(), + pred: WirePredicate::from_pred(pred, id_of), + }, + Op::SetOp { kind, all, .. } => NonASAPOpKind::SetOp { + set_kind: kind.clone(), + all: *all, + }, + Op::Concat { + discriminator_unique_key, + .. + } => NonASAPOpKind::Concat { + discriminator_unique_key: discriminator_unique_key.clone(), + }, + Op::Dedup { cols, .. } => NonASAPOpKind::Dedup { cols: cols.clone() }, + Op::Sort { + keys, partition_by, .. + } => NonASAPOpKind::Sort { + keys: WireSortKey::from_keys(keys, id_of), + partition_by: partition_by.clone(), + }, + Op::Limit { + n, + offset, + partition_by, + .. + } => NonASAPOpKind::Limit { + n: *n, + offset: *offset, + partition_by: partition_by.clone(), + }, + Op::BinaryOp { + operator, + return_bool, + .. + } => NonASAPOpKind::BinaryOp { + operator: operator.clone(), + return_bool: *return_bool, + }, + Op::SQLWindowFunc { + func, + args, + partition_by, + order_by, + frame, + output_name, + .. + } => NonASAPOpKind::SQLWindowFunc { + func: func.clone(), + args: args + .iter() + .map(|e| WireScalarExpr::from_expr(e, id_of)) + .collect(), + partition_by: partition_by.clone(), + order_by: WireSortKey::from_keys(order_by, id_of), + frame: frame.clone(), + output_name: output_name.clone(), + }, + Op::TimeRange { range, kind, .. } => NonASAPOpKind::TimeRange { + range: *range, + range_kind: *kind, + }, + Op::TimeShift { shift, .. } => NonASAPOpKind::TimeShift { shift: *shift }, + Op::PromqlVectorFromScalar(e) => NonASAPOpKind::PromqlVectorFromScalar { + expr: WireScalarExpr::from_expr(e, id_of), + }, + Op::PromqlRelabel { dst, value, .. } => NonASAPOpKind::PromqlRelabel { + dst: dst.clone(), + value: WireScalarExpr::from_expr(value, id_of), + }, + Op::PromqlInfoEnrich { selector, .. } => NonASAPOpKind::PromqlInfoEnrich { + selector: selector.clone(), + }, + Op::PromqlSeriesSample { by, kind, .. } => NonASAPOpKind::PromqlSeriesSample { + by: by.clone(), + sample_kind: *kind, + }, + Op::PromqlSubquery { + range, resolution, .. + } => NonASAPOpKind::PromqlSubquery { + range: *range, + resolution: *resolution, + }, + } + } +} + +// ── The exported DAG ───────────────────────────────────────────────────── + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] +pub enum LogicalASAPOperatorPayload { + Relational { + operator: NonASAPOpKind, + }, + SummaryAgg { + family: FieldDataType, + input: SummaryUpdate, + reduction: Reduction, + grouping: GroupingStrategy, + #[serde(default)] + filter: Option, + }, + SummaryEstimate { + query: SketchStatistic, + }, + FinalizeExactAccumulator, + MaintainPopulation { + population: MaintainedPopulation, + }, + EvaluatePopulation { + evaluation: PopulationStatistic, + }, + SummaryMerge, + SummarySubtract, + SummaryDelete { + key: ColumnId, + }, + SummaryJoin { + key: ColumnId, + family: FieldDataType, + }, + Extension { + name: String, + }, +} + +/// The operator's own inputs with their edge roles, in field order. +pub(super) fn input_edges(operator: &Operator) -> Vec<(&Rc, EdgeRole)> { + match operator { + Operator::NonASAP(op) => match op { + NonASAPOp::Join { left, right, .. } + | NonASAPOp::SetOp { left, right, .. } + | NonASAPOp::BinaryOp { + lhs: left, + rhs: right, + .. + } => vec![(left, EdgeRole::Left), (right, EdgeRole::Right)], + NonASAPOp::Concat { children, .. } => { + children.iter().map(|c| (c, EdgeRole::Input)).collect() + } + NonASAPOp::Filter { child, .. } + | NonASAPOp::Project { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Dedup { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::SQLWindowFunc { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } + | NonASAPOp::PromqlRelabel { child, .. } + | NonASAPOp::PromqlInfoEnrich { child, .. } + | NonASAPOp::PromqlSeriesSample { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => vec![(child, EdgeRole::Input)], + NonASAPOp::Scan { .. } + | NonASAPOp::Values { .. } + | NonASAPOp::PromqlVectorFromScalar(_) => vec![], + }, + Operator::ASAP(op) => match op { + ASAPOp::SummarySubtract { left, right } + | ASAPOp::SummaryJoin { + outer: left, + inner: right, + .. + } => vec![(left, EdgeRole::Left), (right, EdgeRole::Right)], + ASAPOp::SummaryMerge { children } => { + children.iter().map(|c| (c, EdgeRole::Input)).collect() + } + ASAPOp::SummaryAgg { child, .. } + | ASAPOp::FinalizeExactAccumulator { child } + | ASAPOp::MaintainPopulation { child, .. } + | ASAPOp::EvaluatePopulation { child, .. } + | ASAPOp::Extension { child, .. } => vec![(child, EdgeRole::Input)], + ASAPOp::SummaryEstimate { summary_input, .. } + | ASAPOp::SummaryDelete { summary_input, .. } => { + vec![(summary_input, EdgeRole::Input)] + } + }, + } +} + +pub(super) fn payload_of( + operator: &Operator, + id_of: &mut impl FnMut(&Rc) -> LogicalASAPNodeId, +) -> LogicalASAPOperatorPayload { + match operator { + Operator::NonASAP(op) => LogicalASAPOperatorPayload::Relational { + operator: NonASAPOpKind::from_op(op, id_of), + }, + Operator::ASAP(op) => match op { + ASAPOp::SummaryAgg { + family, + input, + reduction, + grouping, + filter, + .. + } => LogicalASAPOperatorPayload::SummaryAgg { + family: family.clone(), + input: input.clone(), + reduction: reduction.clone(), + grouping: grouping.clone(), + filter: filter.as_ref().map(|p| WirePredicate::from_pred(p, id_of)), + }, + ASAPOp::SummaryEstimate { query, .. } => LogicalASAPOperatorPayload::SummaryEstimate { + query: query.clone(), + }, + ASAPOp::FinalizeExactAccumulator { .. } => { + LogicalASAPOperatorPayload::FinalizeExactAccumulator + } + ASAPOp::MaintainPopulation { population, .. } => { + LogicalASAPOperatorPayload::MaintainPopulation { + population: population.clone(), + } + } + ASAPOp::EvaluatePopulation { evaluation, .. } => { + LogicalASAPOperatorPayload::EvaluatePopulation { + evaluation: evaluation.clone(), + } + } + ASAPOp::SummaryMerge { .. } => LogicalASAPOperatorPayload::SummaryMerge, + ASAPOp::SummarySubtract { .. } => LogicalASAPOperatorPayload::SummarySubtract, + ASAPOp::SummaryDelete { key, .. } => { + LogicalASAPOperatorPayload::SummaryDelete { key: *key } + } + ASAPOp::SummaryJoin { key, family, .. } => LogicalASAPOperatorPayload::SummaryJoin { + key: *key, + family: family.clone(), + }, + ASAPOp::Extension { name, .. } => { + LogicalASAPOperatorPayload::Extension { name: name.clone() } + } + }, + } +} + +/// Grouping compatibility between two `SummaryAgg`s by their reductions. +pub(super) fn grouping_compatibility( + producer: &Operator, + consumer: &Operator, +) -> GroupingEdgeCompatibility { + let ( + Operator::ASAP(ASAPOp::SummaryAgg { + reduction: producer, + .. + }), + Operator::ASAP(ASAPOp::SummaryAgg { + reduction: consumer, + .. + }), + ) = (producer, consumer) + else { + return GroupingEdgeCompatibility::NotApplicable; + }; + match (producer, consumer) { + (p, c) if p == c => GroupingEdgeCompatibility::Identical, + (Reduction::PerEntity, Reduction::Reduce(_)) => { + GroupingEdgeCompatibility::ConsumerCoarsensProducer + } + (Reduction::Reduce(p), Reduction::Reduce(c)) + if !p.is_without() && !c.is_without() && c.iter().all(|key| p.contains(key)) => + { + GroupingEdgeCompatibility::ConsumerCoarsensProducer + } + _ => GroupingEdgeCompatibility::Incompatible, + } +} diff --git a/crates/types/src/lib.rs b/crates/types/src/lib.rs index 9fc450c7f..3f666d8a6 100644 --- a/crates/types/src/lib.rs +++ b/crates/types/src/lib.rs @@ -1,32 +1,20 @@ //! `asap-types` — shared vocabulary for the whole workspace. //! -//! Merges the former `asap-ir` crate (the pre-ASAP intent algebra, -//! workload/batch types, and DAG export) with the data-type-only modules of -//! the former `asap-sketch` crate (the post-ASAP sketch-bound IR types, -//! under [`post_asap`]). -//! -//! - [`pre_asap`] / [`types`] / [`workload`] / [`dag_export`] — the -//! pre-ASAP IR: language-agnostic query intent, independent of any -//! sketch decision. -//! - [`post_asap`] — the post-ASAP IR: sketch-bound types -//! ([`post_asap::sketch`], [`post_asap::expr`], [`post_asap::schema`]) -//! that commit to a concrete `SummaryKind`/`SummaryParams` realization. -//! No execution logic lives in this workspace (see issue #190) — a -//! downstream deployment crate is expected to supply that. -//! [`post_asap::query_time`] is the one exception, folder-separated from -//! the rest of `post_asap` on purpose: pure, sketch-object-agnostic -//! posterior error-bound math (issue #239) that a future real sketch -//! runtime's readout path can call directly — see that module's docs -//! for the planning-time/execution-time boundary and why it's unwired -//! today. +//! - [`ir`] — the unified operator IR (#511): one operator DAG before and +//! after ASAP optimization ([`ir::OperatorNode`]), arranged by #511 section +//! ([`ir::operator`], [`ir::scalar`], [`ir::schema`], [`ir::properties`]), +//! plus its passes (canonicalize, CSE, timing) and the wire export +//! ([`ir::export`]). No execution logic lives in this crate (issue #190). +//! - [`workload`] — planner inputs (#509): query and data workloads, the +//! lowered [`workload::parsed_workload`], and [`workload::resources`]. +//! - [`physical`] — #509 Stage 2 decision data: exact-operator schema helpers +//! and window-summary pane primitives. +//! - [`types`] / [`dag_export`] / [`cost`] — accuracy targets, the generic +//! DAG export, and cost annotations. pub mod cost; pub mod dag_export; -pub mod parsed_workload; -pub mod post_asap; -pub mod pre_asap; -pub mod resources; +pub mod ir; +pub mod physical; pub mod serde_f64; pub mod types; pub mod workload; - -pub mod ir; diff --git a/crates/types/src/physical/execution_data_state.rs b/crates/types/src/physical/execution_data_state.rs new file mode 100644 index 000000000..64620e01d --- /dev/null +++ b/crates/types/src/physical/execution_data_state.rs @@ -0,0 +1,21 @@ +//! Exact-operator schema helpers for summary planning. + +use thiserror::Error; + +use crate::ir::schema::Schema; +use crate::ir::SchemaDerivationError; + +/// `schema` as a summary-planning node output: fields and time axis kept, +/// unique keys dropped, closed. +pub fn lift_plain(schema: &Schema) -> Schema { + Schema::lifted(schema.fields.clone(), schema.time_index) +} + +/// Why an exact operator's output schema could not be derived. +#[derive(Debug, Error)] +pub enum ExactOperationSchemaError { + #[error("exact operator input carries summary state, not plain columns")] + NonPlainInput, + #[error("schema derivation failed: {0}")] + Schema(#[from] SchemaDerivationError), +} diff --git a/crates/types/src/physical/mod.rs b/crates/types/src/physical/mod.rs new file mode 100644 index 000000000..f599ad28c --- /dev/null +++ b/crates/types/src/physical/mod.rs @@ -0,0 +1,10 @@ +//! #509 Stage 2 decision data: exact-operator schema helpers for +//! materialization and the pane primitives of window summaries. + +pub mod execution_data_state; +pub mod summary_window; + +pub use execution_data_state::{lift_plain, ExactOperationSchemaError}; +pub use summary_window::{ + plan_pane_phase, validate_pane_coverage, PaneCoverageError, PaneLayout, WindowEdgeCoverage, +}; diff --git a/crates/types/src/post_asap/summary_window.rs b/crates/types/src/physical/summary_window.rs similarity index 71% rename from crates/types/src/post_asap/summary_window.rs rename to crates/types/src/physical/summary_window.rs index 0e344c1ed..724a8d595 100644 --- a/crates/types/src/post_asap/summary_window.rs +++ b/crates/types/src/physical/summary_window.rs @@ -1,29 +1,13 @@ -//! Planner-level summary-window primitives. +//! Planner-level summary-window pane primitives. //! -//! These values identify the abstract window framework selected during -//! candidate search. They do not identify a runtime library, process, -//! placement, shard layout, storage backend, or deployment instance; those -//! choices belong to downstream physical compilation. +//! These values describe pane layout and window-edge coverage. They do not +//! identify a runtime library, process, placement, shard layout, storage +//! backend, or deployment instance; those choices belong to downstream +//! physical compilation. use crate::workload::RepeatedDemand; use serde::{Deserialize, Serialize}; -/// Abstract framework used to organize incrementally maintained summary -/// state over time. -/// -/// The built-in variants name semantics that the planner can compare across -/// implementations with defined planning and accuracy behavior. -#[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum SummaryWindowFramework { - /// Disjoint, fixed-width windows. - Tumbling, - /// Overlapping logical windows, commonly realized from reusable panes. - Sliding, - /// Hierarchical buckets with exponentially increasing coverage. - ExponentialHistogram, -} - /// Concrete pane phase recorded in a catalog layout or inventory snapshot. /// Milliseconds are canonical throughout the shared contract. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] @@ -57,7 +41,7 @@ pub enum PaneCoverageError { }, } -/// Validate that a pane-only readout covers a query exactly. A mismatched +/// Validate that a pane-only evaluation covers a query exactly. A mismatched /// phase is sound only when the physical plan explicitly supplies an exact /// residual for the partial edge panes. pub fn validate_pane_coverage( @@ -123,31 +107,7 @@ mod tests { use super::*; #[test] - fn built_in_frameworks_round_trip() { - for framework in [ - SummaryWindowFramework::Tumbling, - SummaryWindowFramework::Sliding, - SummaryWindowFramework::ExponentialHistogram, - ] { - let encoded = serde_json::to_string(&framework).unwrap(); - assert_eq!( - serde_json::from_str::(&encoded).unwrap(), - framework - ); - } - } - - /// Opaque names cannot enter planning without defined window semantics. - #[test] - fn unimplemented_window_extensions_are_rejected() { - assert!(serde_json::from_value::( - serde_json::json!({"extension": "learned_window"}) - ) - .is_err()); - } - - #[test] - fn pane_only_readout_rejects_source_and_query_phase_mismatch() { + fn pane_only_evaluation_rejects_source_and_query_phase_mismatch() { let layout = PaneLayout { pane_width_ms: 60_000, pane_origin_ms: Some(26_000), diff --git a/crates/types/src/post_asap/cse.rs b/crates/types/src/post_asap/cse.rs deleted file mode 100644 index 55889008c..000000000 --- a/crates/types/src/post_asap/cse.rs +++ /dev/null @@ -1,442 +0,0 @@ -//! Structural sharing for a selected workload in one execution/data scope. -//! -//! This is not candidate selection or a cross-request cache. Callers opt into -//! common producer execution only after agreeing on lifecycle and data scope. -//! Typed equality includes schemas, guarantees and complete source expressions. - -use std::collections::HashMap; -use std::rc::Rc; - -use super::{SummaryExpr, SummaryNode}; - -/// Numeric PartialEq alone conflates signed zeros. The serialized check is -/// additional evidence, never a replacement for typed equality (JSON maps -/// nonfinite floats to null). Keep this rule local to structural sharing. -fn same_value(left: &T, right: &T) -> bool { - left == right - && match (serde_json::to_string(left), serde_json::to_string(right)) { - (Ok(left), Ok(right)) => left == right, - _ => false, - } -} - -/// Children have already been interned. Comparing their identities avoids -/// recursively expanding a shared DAG once for every path to each descendant. -fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { - use SummaryExpr::*; - let expression_equal = match (&left.expr, &right.expr) { - (KeepPreAsap(a), KeepPreAsap(b)) => Rc::ptr_eq(a, b) || same_value(a, b), - ( - BinaryOp { - lhs: al, - rhs: ar, - operator: ao, - timing: at, - }, - BinaryOp { - lhs: bl, - rhs: br, - operator: bo, - timing: bt, - }, - ) => Rc::ptr_eq(al, bl) && Rc::ptr_eq(ar, br) && ao == bo && at == bt, - ( - ValueOperation { - child: ac, - operation: ao, - timing: at, - }, - ValueOperation { - child: bc, - operation: bo, - timing: bt, - }, - ) => Rc::ptr_eq(ac, bc) && same_value(ao, bo) && at == bt, - ( - RelationalJoin { - left: al, - right: ar, - kind: ak, - pred: ap, - pruning: ax, - }, - RelationalJoin { - left: bl, - right: br, - kind: bk, - pred: bp, - pruning: bx, - }, - ) => { - Rc::ptr_eq(al, bl) - && Rc::ptr_eq(ar, br) - && ak == bk - && same_value(ap, bp) - && same_value(ax, bx) - } - ( - SummaryAgg { - child: ac, - family: af, - input: ai, - reduction: ar, - grouping: ag, - filter: afl, - }, - SummaryAgg { - child: bc, - family: bf, - input: bi, - reduction: br, - grouping: bg, - filter: bfl, - }, - ) => { - Rc::ptr_eq(ac, bc) - && af == bf - && same_value(ai, bi) - && ar == br - && ag == bg - && same_value(afl, bfl) - } - ( - SummaryJoin { - outer: ao, - inner: ai, - key: ak, - family: af, - }, - SummaryJoin { - outer: bo, - inner: bi, - key: bk, - family: bf, - }, - ) => Rc::ptr_eq(ao, bo) && Rc::ptr_eq(ai, bi) && ak == bk && af == bf, - ( - SummarySubtract { - left: al, - right: ar, - }, - SummarySubtract { - left: bl, - right: br, - }, - ) => Rc::ptr_eq(al, bl) && Rc::ptr_eq(ar, br), - ( - SummaryEstimate { - summary_input: ai, - query: aq, - }, - SummaryEstimate { - summary_input: bi, - query: bq, - }, - ) => Rc::ptr_eq(ai, bi) && same_value(aq, bq), - ( - SummaryDelete { - summary_input: ai, - key: ak, - }, - SummaryDelete { - summary_input: bi, - key: bk, - }, - ) => Rc::ptr_eq(ai, bi) && ak == bk, - ( - SummaryMerge { - children: a, - timing: at, - }, - SummaryMerge { - children: b, - timing: bt, - }, - ) => at == bt && a.len() == b.len() && a.iter().zip(b).all(|(a, b)| Rc::ptr_eq(a, b)), - // Keep this exhaustive on the left: new variants require a sharing rule. - ( - KeepPreAsap(_) - | BinaryOp { .. } - | ValueOperation { .. } - | RelationalJoin { .. } - | SummaryAgg { .. } - | SummaryJoin { .. } - | SummarySubtract { .. } - | SummaryEstimate { .. } - | SummaryDelete { .. } - | SummaryMerge { .. }, - _, - ) => false, - }; - expression_equal && left.schema == right.schema && same_value(&left.guarantee, &right.guarantee) -} - -/// Intern equal selected sub-DAGs across roots while preserving every root ID. -/// -/// Only structural equality is used: no grouping, parameter, accuracy or source -/// coercions are performed. All roots must belong to the same data snapshot or -/// maintenance scope. Downstream realization must still check physical -/// implementation compatibility. Use separate calls for independent executions. -/// -/// When the selected states are identical, this is the planner's -/// summary-capability rule (#509 Pass 2): one summary build node feeds every -/// readout it supports, e.g. one KLL for p50 and p99, or one UnivMon for -/// distinct count, entropy and L2. Candidate generation sizes a variant for -/// the strictest sibling consumer so differing accuracy targets can reach -/// identical states here. -pub fn share_common_summary_sub_dags( - roots: Vec<(Id, Rc)>, -) -> Vec<(Id, Rc)> { - fn visit( - node: &Rc, - seen: &mut HashMap>, - pool: &mut Vec>, - ) -> Rc { - let identity = Rc::as_ptr(node) as usize; - if let Some(node) = seen.get(&identity) { - return Rc::clone(node); - } - let mut result = node.as_ref().clone(); - match &mut result.expr { - SummaryExpr::KeepPreAsap(_) => {} - SummaryExpr::SummaryAgg { child, .. } => *child = visit(child, seen, pool), - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - *lhs = visit(lhs, seen, pool); - *rhs = visit(rhs, seen, pool); - } - - SummaryExpr::ValueOperation { child, .. } => *child = visit(child, seen, pool), - SummaryExpr::RelationalJoin { left, right, .. } => { - *left = visit(left, seen, pool); - *right = visit(right, seen, pool); - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - *outer = visit(outer, seen, pool); - *inner = visit(inner, seen, pool); - } - SummaryExpr::SummarySubtract { left, right } => { - *left = visit(left, seen, pool); - *right = visit(right, seen, pool); - } - SummaryExpr::SummaryEstimate { summary_input, .. } - | SummaryExpr::SummaryDelete { summary_input, .. } => { - *summary_input = visit(summary_input, seen, pool); - } - SummaryExpr::SummaryMerge { children, .. } => { - for child in children { - *child = visit(child, seen, pool); - } - } - } - let result = match pool.iter().find(|existing| same_node(existing, &result)) { - Some(existing) => Rc::clone(existing), - None => { - let result = Rc::new(result); - pool.push(Rc::clone(&result)); - result - } - }; - seen.insert(identity, Rc::clone(&result)); - result - } - let mut seen = HashMap::new(); - let mut pool = Vec::new(); - roots - .into_iter() - .map(|(id, root)| (id, visit(&root, &mut seen, &mut pool))) - .collect() -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::post_asap::{ResultGuarantee, Schema}; - use crate::pre_asap::{QueryExpr, ScalarValue}; - - fn leaf(value: f64) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::Literal(ScalarValue::Float64( - value, - )))), - schema: Schema::lifted(vec![], None), - guarantee: Some(ResultGuarantee::exact("fixture")), - }) - } - - // Equal separately constructed roots preserve both IDs but share identity. - #[test] - fn shares_equal_roots_and_preserves_ids() { - let roots = share_common_summary_sub_dags(vec![("a", leaf(1.0)), ("b", leaf(1.0))]); - assert_eq!(roots[0].0, "a"); - assert_eq!(roots[1].0, "b"); - assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); - } - - // A diamond is retained across the returned roots, not copied per consumer. - #[test] - fn shares_children_across_distinct_roots() { - let merge = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: crate::post_asap::ExecutionTiming::IngestionTime, - children: vec![leaf(1.0), leaf(2.0)], - }, - schema: Schema::lifted(vec![], None), - guarantee: None, - }); - let roots = share_common_summary_sub_dags(vec![(0, leaf(1.0)), (1, merge)]); - let SummaryExpr::SummaryMerge { children, .. } = &roots[1].1.expr else { - panic!() - }; - assert!(Rc::ptr_eq(&roots[0].1, &children[0])); - assert!(!Rc::ptr_eq(&children[0], &children[1])); - } - - // Unknown guarantees must not be replaced by an equal expression's exact guarantee. - #[test] - fn distinct_guarantees_and_values_are_not_shared() { - let mut unknown = leaf(1.0).as_ref().clone(); - unknown.guarantee = None; - let roots = share_common_summary_sub_dags(vec![ - (0, leaf(1.0)), - (1, Rc::new(unknown)), - (2, leaf(2.0)), - ]); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[2].1)); - assert!(roots[1].1.guarantee.is_none()); - } - - // Sharing must preserve IEEE signed zero, including inside exact expressions. - #[test] - fn signed_zero_is_not_coalesced() { - for values in [[0.0, -0.0], [-0.0, 0.0]] { - let roots = - share_common_summary_sub_dags(vec![(0, leaf(values[0])), (1, leaf(values[1]))]); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - for ((_, root), expected) in roots.iter().zip(values) { - let SummaryExpr::KeepPreAsap(expr) = &root.expr else { - panic!() - }; - let QueryExpr::Literal(ScalarValue::Float64(actual)) = expr.as_ref() else { - panic!() - }; - assert_eq!(actual.to_bits(), expected.to_bits()); - assert_eq!(1.0 / actual, 1.0 / expected); - } - } - } - - // Exact expression wrappers must retain signed zero too; JSON's null - // encoding of nonfinite floats must never become the equality decision. - #[test] - fn nested_values_and_nonfinite_values_remain_distinct() { - let wrapped = |value| { - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(Rc::new(QueryExpr::promql_scalar(value))), - ..leaf(1.0).as_ref().clone() - }) - }; - for (a, b) in [ - (0.0, -0.0), - (f64::INFINITY, f64::NEG_INFINITY), - (f64::NAN, f64::NAN), - ] { - let roots = share_common_summary_sub_dags(vec![(0, wrapped(a)), (1, wrapped(b))]); - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - } - let roots = share_common_summary_sub_dags(vec![ - (0, wrapped(f64::INFINITY)), - (1, wrapped(f64::INFINITY)), - ]); - assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); - } - - // Distinct quantile readouts share only a compatible typed sketch producer. - #[test] - fn quantile_roots_share_producer_but_not_readout_or_parameters() { - use crate::post_asap::{ - FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, - SketchStatistic, SummaryUpdate, - }; - use crate::pre_asap::{ColumnRef, Reduction}; - fn readout(q: f64, alpha: f64) -> Rc { - let producer = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: leaf(1.0), - family: FieldDataType::Sketch( - SketchKind::new( - SketchAlgorithm::DDSketch, - SketchParams::DDSketch { alpha }, - ), - GroupingStrategy::default(), - ), - input: SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::PerEntity, - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted(vec![], None), - guarantee: None, - }); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: producer, - query: SketchStatistic::Quantile { q }, - }, - schema: Schema::lifted(vec![], None), - guarantee: None, - }) - } - let roots = share_common_summary_sub_dags(vec![ - ("p95", readout(0.95, 0.01)), - ("p99", readout(0.99, 0.01)), - ("strict", readout(0.95, 0.001)), - ]); - let producer = |root: &Rc| match &root.expr { - SummaryExpr::SummaryEstimate { summary_input, .. } => Rc::clone(summary_input), - _ => panic!(), - }; - assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); - assert!(Rc::ptr_eq(&producer(&roots[0].1), &producer(&roots[1].1))); - assert!(!Rc::ptr_eq(&producer(&roots[0].1), &producer(&roots[2].1))); - } - - // Fifty unique input nodes must not require walking an expanded 2^24 DAG. - // The timeout is a coarse runaway guard, not a performance SLA. - #[test] - fn shared_diamond_does_not_expand_during_comparison() { - let (done, completion) = std::sync::mpsc::channel(); - let worker = std::thread::spawn(move || { - fn diamond() -> Rc { - let mut current = leaf(1.0); - for _ in 0..24 { - current = Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - timing: super::super::ExecutionTiming::QueryTime, - lhs: Rc::clone(¤t), - rhs: current, - operator: super::super::BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: crate::pre_asap::BinaryOpKind::Arithmetic( - crate::pre_asap::ArithmeticOpKind::Add, - ), - vector_match: None, - }, - }, - schema: super::super::Schema::lifted(vec![], None), - guarantee: None, - }); - } - current - } - let roots = share_common_summary_sub_dags(vec![(0, diamond()), (1, diamond())]); - assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); - done.send(()).unwrap(); - }); - completion - .recv_timeout(std::time::Duration::from_secs(5)) - .expect("comparison expanded the shared DAG"); - worker.join().unwrap(); - } -} diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs deleted file mode 100644 index 91373816b..000000000 --- a/crates/types/src/post_asap/execution_data_state.rs +++ /dev/null @@ -1,1341 +0,0 @@ -//! Execution-data-state contract for mixed exact/summary plans (issue #171). -//! -//! A post-ASAP DAG mixes two very different moments of execution: the -//! **update/ingest path** (rows arrive, maintained summary state is updated) -//! and **query evaluation** (maintained state is read out and a final result -//! is produced). A plan that places a query-time residual *underneath* a -//! maintained summary is not merely expensive — it is unexecutable, because -//! the maintenance loop has no readout values to feed into that summary. -//! [`SummaryExpr::ValueOperation`] represents such work without inventing a -//! node per function or use case. Its [`ExecutionTiming`] makes placement -//! explicit and independent of the semantic [`ValueOperation`]. -//! -//! [`ExecutionDataState`] is what a node's output *is*, at which data_state; -//! [`validate_execution_data_states`] checks every edge of a DAG against the -//! rules below at plan construction, returning a typed [`ExecutionDataStateError`] rather -//! than deferring to a runtime failure. -//! -//! ## Edge rules -//! -//! | Parent | Accepts from `child` | -//! |---|---| -//! | `SummaryAgg.child` | Rows or exact accumulator state at either phase. The initial construction phase follows the input; deployment assigns final phases. | -//! | `SummaryEstimate.summary_input` | Summary state at either phase (any family). Initial readout produces `QUERY_ROWS`. | -//! | `SummaryJoin.outer/inner` | `INGESTION_ROWS` or `INGESTION_SUMMARY`; never a read-time data_state. | -//! | `SummarySubtract`/`SummaryDelete` | `INGESTION_SUMMARY`. | -//! | `SummaryMerge` | Summary state at its explicit ingestion or read timing. | -//! | `ValueOperation.child` with `IngestionTime` | `INGESTION_ROWS`; explicit `FinalizeExactAccumulator` also accepts exact accumulator state. Produces `INGESTION_ROWS`. | -//! | `ValueOperation.child` with `QueryTime` | `QUERY_ROWS`. Produces `QUERY_ROWS`. | -//! -//! ## `KeepPreAsap` declares its data_state through the derivation -//! -//! A [`SummaryExpr::KeepPreAsap`] leaf is a raw pre-ASAP computation that a -//! runtime can execute at either time: as maintenance input beneath a -//! `SummaryAgg`/maintenance-time `ValueOperation`, or as a query-time fallback -//! beneath a read-time `ValueOperation` (or at the root). It carries no timing -//! field of its own -//! — every existing consumer pattern-matches the one-field shape — so its -//! data_state is *assigned* by [`validate_execution_data_states`] from the edge that -//! reaches it and reported in the returned [`ExecutionDataStateAssignment`]. What it may -//! not do is stay ambiguous inside one mixed plan: the same `Rc` -//! reached once as update input and once as query-time fallback is -//! [`ExecutionDataStateError::AmbiguousKeepPreAsap`], because no single execution of that -//! sub-DAG can serve both roles. - -use std::collections::HashMap; -use std::rc::Rc; - -use thiserror::Error; - -use super::expr::{ExactOperation, SummaryExpr, SummaryNode, ValueOperation}; -use crate::pre_asap::schema::FieldDataType; - -use crate::pre_asap::query_expr::{aggregate_output_schema, Predicate, QueryExprError}; -use crate::pre_asap::schema::Schema; - -/// When a post-ASAP value is produced. -#[derive( - Default, Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize, -)] -#[serde(rename_all = "snake_case")] -pub enum ExecutionTiming { - IngestionTime, - #[default] - QueryTime, -} - -impl ExecutionTiming { - pub fn is_query_time(&self) -> bool { - *self == Self::QueryTime - } - pub fn as_str(self) -> &'static str { - match self { - Self::IngestionTime => "ingestion_time", - Self::QueryTime => "query_time", - } - } -} - -/// The primitive representation carried by a post-ASAP edge. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] -pub enum DataPrimitive { - /// Directly usable values, including approximate summary readouts. - /// This does not imply original input data or an exact guarantee. - Raw, - SummaryState, -} - -impl DataPrimitive { - pub fn as_str(self) -> &'static str { - match self { - Self::Raw => "raw", - Self::SummaryState => "summary_state", - } - } -} - -/// The two-dimensional edge contract: when a value exists and which data -/// primitive it carries. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] -pub struct ExecutionDataState { - pub timing: ExecutionTiming, - pub primitive: DataPrimitive, -} - -impl ExecutionDataState { - pub const INGESTION_ROWS: Self = Self { - timing: ExecutionTiming::IngestionTime, - primitive: DataPrimitive::Raw, - }; - pub const INGESTION_SUMMARY: Self = Self { - timing: ExecutionTiming::IngestionTime, - primitive: DataPrimitive::SummaryState, - }; - pub const QUERY_ROWS: Self = Self { - timing: ExecutionTiming::QueryTime, - primitive: DataPrimitive::Raw, - }; -} - -impl std::fmt::Display for ExecutionDataState { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "{}/{}", self.timing.as_str(), self.primitive.as_str()) - } -} - -/// Which parent/edge a [`ExecutionDataStateError`] is about — the variant name of the -/// parent `SummaryExpr` plus its field, for a message a plan author can act -/// on. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ExecutionDataStateEdge { - SummaryAggChild, - SummaryEstimateInput, - SummaryJoinInput, - SummarySubtractInput, - SummaryDeleteInput, - SummaryMergeInput, - ValueOperationChild, -} - -impl ExecutionDataStateEdge { - fn describe(self) -> &'static str { - match self { - Self::SummaryAggChild => "SummaryAgg.child", - Self::SummaryEstimateInput => "SummaryEstimate.summary_input", - Self::SummaryJoinInput => "SummaryJoin.{outer,inner}", - Self::SummarySubtractInput => "SummarySubtract.{left,right}", - Self::SummaryDeleteInput => "SummaryDelete.summary_input", - Self::SummaryMergeInput => "SummaryMerge.children[]", - Self::ValueOperationChild => "ValueOperation.child", - } - } -} - -/// A plan-construction-time data_state violation. Typed (not a string) so a -/// strategy can degrade to a conservative fallback on the specific variant -/// it expects, and so tests can assert the *reason* a plan was rejected. -#[derive(Debug, Clone, PartialEq, Eq, Error)] -pub enum ExecutionDataStateError { - #[error("invalid maintained-population maintenance/readout contract")] - InvalidMaintainedPopulation, - /// A query-time value (`SummaryEstimate` / read-time `ValueOperation` output) - /// placed beneath a maintained summary — the one shape issue #171's - /// data_state split exists to make unrepresentable. - #[error( - "readout value under maintenance: {edge} received a {child} input, but a maintained \ - summary can only consume update-path values (or exact accumulator state)" - )] - ReadoutUnderMaintenance { - edge: &'static str, - child: ExecutionDataState, - }, - /// Any other edge whose child data_state the parent does not accept - /// (e.g. plain update rows fed straight into a `SummaryEstimate`, or a - /// sketch's opaque state fed into a read-time `ValueOperation`). - #[error("{edge} does not accept a {child} input")] - IllegalChildDataState { - edge: &'static str, - child: ExecutionDataState, - }, - /// A `SummaryAgg` whose child is summary state of a family other than an - /// exact accumulator — re-accumulating opaque sketch/sample/… state on - /// the update path has no defined semantics here. - #[error( - "SummaryAgg.child carries {family} summary state; only exact accumulator state can be \ - composed into another maintained summary" - )] - UnsupportedStateComposition { family: String }, - /// One shared `KeepPreAsap` node reached both as update-path raw input - /// and as a query-time fallback — see the module docs. - #[error( - "KeepPreAsap sub-DAG is data_state-ambiguous: reached as {first} and as {second} in the same \ - plan" - )] - AmbiguousKeepPreAsap { - first: ExecutionDataState, - second: ExecutionDataState, - }, - /// A maintenance-time value operation at the root of a plan: - /// nothing maintains state above it, so its output is never read. - #[error("A maintenance-time value operation cannot be a plan root: its update-path output feeds nothing")] - MaintenanceRowsAtRoot, - #[error("unsupported maintenance binary schema or operator")] - InvalidMaintenanceBinary, - #[error("checked division requires one valid guard on a read-time division operator")] - InvalidCheckedDivision, - /// An `ExactOperation` whose input columns are not all `Plain` at its - /// declared data_state. - #[error("exact operator consumes non-plain column {column:?} ({dtype})")] - NonPlainOperand { column: String, dtype: String }, - /// A reserved ASAP operator (`SummaryMerge`, `SummarySubtract`, - /// `SummaryDelete`, `SummaryJoin`, `Extension`) in an executable plan. - #[error("{operator} is a reserved operator with no execution contract yet")] - UnimplementedOperator { operator: &'static str }, -} - -/// The data_state assigned to every node of a validated plan, keyed by -/// `Rc` pointer identity — the explicit per-node "execution_data_state" a -/// runtime or a DAG export reads instead of re-deriving it. For every -/// non-`KeepPreAsap` node this equals [`produced_data_state`]; for a -/// `KeepPreAsap` leaf it is the data_state the reaching edge assigned. -#[derive(Debug, Clone, Default)] -pub struct ExecutionDataStateAssignment { - domains: HashMap<*const SummaryNode, ExecutionDataState>, -} - -impl ExecutionDataStateAssignment { - /// The data_state assigned to `node`, if it was part of the validated plan. - pub fn data_state_of(&self, node: &Rc) -> Option { - self.domains.get(&Rc::as_ptr(node)).copied() - } - - /// The data_state assigned to the node at `ptr` — for callers walking a plan - /// by reference rather than by `Rc`. - pub fn data_state_of_ptr(&self, ptr: *const SummaryNode) -> Option { - self.domains.get(&ptr).copied() - } -} - -/// Initial layout proposed by semantic realization, not a restriction on physical -/// operator placement. `PostAsapDAG::with_execution_phases` assigns the final -/// phase independently of payload kind. Returns `None` for -/// [`SummaryExpr::KeepPreAsap`], whose data_state is assigned by the edge reaching -/// it (see the module docs). -pub fn produced_data_state(expr: &SummaryExpr) -> Option { - Some(match expr { - SummaryExpr::KeepPreAsap(_) => return None, - SummaryExpr::BinaryOp { timing, .. } => ExecutionDataState { - timing: *timing, - primitive: DataPrimitive::Raw, - }, - SummaryExpr::RelationalJoin { .. } => ExecutionDataState::QUERY_ROWS, - SummaryExpr::SummaryAgg { child, .. } => ExecutionDataState { - timing: produced_data_state(&child.expr) - .map_or(ExecutionTiming::IngestionTime, |state| state.timing), - primitive: DataPrimitive::SummaryState, - }, - SummaryExpr::SummaryJoin { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } => ExecutionDataState::INGESTION_SUMMARY, - SummaryExpr::SummaryMerge { timing, .. } => ExecutionDataState { - timing: *timing, - primitive: DataPrimitive::SummaryState, - }, - SummaryExpr::SummaryEstimate { .. } => ExecutionDataState::QUERY_ROWS, - SummaryExpr::ValueOperation { timing, .. } => match timing { - ExecutionTiming::IngestionTime => ExecutionDataState::INGESTION_ROWS, - ExecutionTiming::QueryTime => ExecutionDataState::QUERY_ROWS, - }, - }) -} - -/// Is `family` the exact-accumulator family whose partial state *is* the -/// value — the one summary state a `SummaryAgg` may re-accumulate? -fn is_exact_accumulator_state(schema: &Schema) -> Result<(), ExecutionDataStateError> { - for field in &schema.fields { - match &field.dtype { - FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) => {} - other => { - return Err(ExecutionDataStateError::UnsupportedStateComposition { - family: format!("{other:?}"), - }) - } - } - } - Ok(()) -} - -/// Validate every edge of the DAG rooted at `root` against the module-level -/// rules, returning each node's assigned data_state on success. Shared -/// `Rc`s are visited once per reaching edge (the assignment is -/// per node, so a conflict between two edges is what -/// [`ExecutionDataStateError::AmbiguousKeepPreAsap`] detects). -pub fn validate_execution_data_states( - root: &Rc, -) -> Result { - // The root may be a readable value or bare maintained state (a - // deployment may hand an `ExactAggregate` accumulator straight to a - // consumer) — only an update-path-only root is meaningless. - let root_domain = match produced_data_state(&root.expr) { - None => ExecutionDataState::QUERY_ROWS, - Some(ExecutionDataState::INGESTION_ROWS) => { - return Err(ExecutionDataStateError::MaintenanceRowsAtRoot) - } - Some(data_state) => data_state, - }; - validate_execution_data_states_at(root, root_domain) -} - -/// [`validate_execution_data_states`] for a *sub*-plan whose root is known to -/// sit at `data_state` — e.g. a maintenance-time `ValueOperation` about to be placed beneath a -/// `SummaryAgg`, which would be rejected as a whole-plan root but is a -/// legal update-path input. Validates every edge beneath `root` exactly -/// as the whole-plan entry point does. -pub fn validate_execution_data_states_at( - root: &Rc, - data_state: ExecutionDataState, -) -> Result { - let mut assignment = ExecutionDataStateAssignment::default(); - visit(root, data_state, &mut assignment)?; - Ok(assignment) -} - -/// The source rows whose series a maintenance operand has one row for: a -/// finalized per-series Sum or Count of those rows, or aligned arithmetic of -/// operands over the same rows. Each emits exactly the series with a sample. -fn per_series_rows(node: &SummaryNode) -> Option<&crate::pre_asap::QueryExpr> { - use crate::post_asap::ExactKind; - match &node.expr { - SummaryExpr::ValueOperation { - child, - operation: ValueOperation::FinalizeExactAccumulator, - .. - } => match &child.expr { - SummaryExpr::SummaryAgg { - child, - family: FieldDataType::ExactAggregate(ExactKind::Sum | ExactKind::Count, _), - reduction: crate::pre_asap::query_expr::Reduction::PerEntity, - .. - } => match &child.expr { - SummaryExpr::KeepPreAsap(rows) => Some(rows.as_ref()), - _ => None, - }, - _ => None, - }, - SummaryExpr::BinaryOp { - lhs, - rhs, - timing: ExecutionTiming::IngestionTime, - .. - } => { - let rows = per_series_rows(lhs)?; - (per_series_rows(rhs) == Some(rows)).then_some(rows) - } - _ => None, - } -} - -/// Record `data_state` for `node` (detecting a conflicting earlier assignment -/// for a `KeepPreAsap`), then check and recurse into every child edge. -fn visit( - node: &Rc, - data_state: ExecutionDataState, - assignment: &mut ExecutionDataStateAssignment, -) -> Result<(), ExecutionDataStateError> { - let ptr = Rc::as_ptr(node); - if let Some(previous) = assignment.domains.get(&ptr) { - if *previous != data_state { - return Err(ExecutionDataStateError::AmbiguousKeepPreAsap { - first: *previous, - second: data_state, - }); - } - // Already validated through another edge with the same data_state. - return Ok(()); - } - assignment.domains.insert(ptr, data_state); - - match &node.expr { - SummaryExpr::KeepPreAsap(_) => Ok(()), - SummaryExpr::BinaryOp { - lhs, - rhs, - timing, - operator, - } => { - if (operator.checked_relative_division && operator.checked_finite_division) - || (operator.checked_relative_division || operator.checked_finite_division) - && (*timing != ExecutionTiming::QueryTime - || !matches!( - operator.kind, - crate::pre_asap::BinaryOpKind::Arithmetic( - crate::pre_asap::ArithmeticOpKind::Div - ) - )) - { - return Err(ExecutionDataStateError::InvalidCheckedDivision); - } - if *timing == ExecutionTiming::IngestionTime { - use crate::pre_asap::{BinaryOpKind, DataType}; - if operator.vector_match.is_some() - || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) - || lhs.schema != rhs.schema - || lhs.schema != node.schema - // The opaque identity is a key, not an extra maintenance value. - || node.schema.fields.iter().filter(|field| { - field.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY - }).count() > 1 - // Maintenance arithmetic pairs every row by identity, while - // Prometheus drops unmatched series; it is exact only when - // both operands provably produce the same series. - || node.schema.fields.iter().any(|field| { - field.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY - }) && per_series_rows(lhs).is_none_or(|rows| per_series_rows(rhs) != Some(rows)) - || !node.schema.fields.iter().all(|field| { - !field.nullable - && if field.name == crate::pre_asap::schema::PROMQL_SERIES_IDENTITY { - field.dtype == FieldDataType::Plain(DataType::Utf8) - } else { - matches!( - field.dtype, - FieldDataType::Plain(DataType::Float64 | DataType::Timestamp) - ) - } - }) - || node - .schema - .fields - .iter() - .filter(|field| { - matches!(field.dtype, FieldDataType::Plain(DataType::Float64)) - }) - .count() - != 1 - { - return Err(ExecutionDataStateError::InvalidMaintenanceBinary); - } - } - let expected = ExecutionDataState { - timing: *timing, - primitive: DataPrimitive::Raw, - }; - for input in [lhs, rhs] { - let state = produced_data_state(&input.expr).unwrap_or(expected); - if state != expected { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: "BinaryOp operand", - child: state, - }); - } - visit(input, state, assignment)?; - } - Ok(()) - } - - SummaryExpr::RelationalJoin { left, right, .. } => { - for input in [left, right] { - let state = - produced_data_state(&input.expr).unwrap_or(ExecutionDataState::QUERY_ROWS); - if state != ExecutionDataState::QUERY_ROWS { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: "RelationalJoin input", - child: state, - }); - } - visit(input, state, assignment)?; - } - Ok(()) - } - SummaryExpr::SummaryAgg { child, .. } => { - let child_domain = child_domain( - child, - ExecutionDataStateEdge::SummaryAggChild, - |avail| match avail { - ExecutionDataState::INGESTION_ROWS | ExecutionDataState::QUERY_ROWS => Ok(()), - state if state.primitive == DataPrimitive::SummaryState => { - is_exact_accumulator_state(&child.schema) - } - other => Err(ExecutionDataStateError::ReadoutUnderMaintenance { - edge: ExecutionDataStateEdge::SummaryAggChild.describe(), - child: other, - }), - }, - )?; - visit(child, child_domain, assignment) - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - for input in [outer, inner] { - let s = child_domain(input, ExecutionDataStateEdge::SummaryJoinInput, |avail| { - match avail { - ExecutionDataState::INGESTION_ROWS - | ExecutionDataState::INGESTION_SUMMARY => Ok(()), - other => Err(ExecutionDataStateError::ReadoutUnderMaintenance { - edge: ExecutionDataStateEdge::SummaryJoinInput.describe(), - child: other, - }), - } - })?; - visit(input, s, assignment)?; - } - Ok(()) - } - SummaryExpr::SummarySubtract { left, right } => { - for input in [left, right] { - let s = state_only(input, ExecutionDataStateEdge::SummarySubtractInput)?; - visit(input, s, assignment)?; - } - Ok(()) - } - SummaryExpr::SummaryDelete { summary_input, .. } => { - let s = state_only(summary_input, ExecutionDataStateEdge::SummaryDeleteInput)?; - visit(summary_input, s, assignment) - } - SummaryExpr::SummaryMerge { children, timing } => { - for input in children { - let s = child_domain(input, ExecutionDataStateEdge::SummaryMergeInput, |state| { - if state.primitive == DataPrimitive::SummaryState - && (*timing == ExecutionTiming::QueryTime || state.timing == *timing) - { - Ok(()) - } else { - Err(ExecutionDataStateError::IllegalChildDataState { - edge: ExecutionDataStateEdge::SummaryMergeInput.describe(), - child: state, - }) - } - })?; - visit(input, s, assignment)?; - } - Ok(()) - } - SummaryExpr::SummaryEstimate { summary_input, .. } => { - let s = state_only(summary_input, ExecutionDataStateEdge::SummaryEstimateInput)?; - visit(summary_input, s, assignment) - } - SummaryExpr::ValueOperation { - child, - operation, - timing, - } => { - // Population timing is a lifecycle decision: a retained population - // is maintained at ingestion time, an ephemeral one is rebuilt - // from raw input per query. Its input and readout contracts are - // structural and hold either way. - let valid_population = match operation { - ValueOperation::MaintainPopulation { population } => { - matches!(&child.expr, SummaryExpr::KeepPreAsap(input) if population.matches_input(input)) - } - ValueOperation::ReadPopulation { readout } => { - *timing == ExecutionTiming::QueryTime - && matches!(&child.expr, SummaryExpr::ValueOperation { operation: ValueOperation::MaintainPopulation { population }, .. } if population.supports(readout)) - } - _ => true, - }; - if !valid_population { - return Err(ExecutionDataStateError::InvalidMaintainedPopulation); - } - let required = match timing { - ExecutionTiming::IngestionTime => ExecutionDataState::INGESTION_ROWS, - ExecutionTiming::QueryTime => ExecutionDataState::QUERY_ROWS, - }; - let s = produced_data_state(&child.expr).unwrap_or(required); - let exact_readout = (*timing == ExecutionTiming::QueryTime - || matches!(operation, ValueOperation::FinalizeExactAccumulator)) - && s.primitive == DataPrimitive::SummaryState - && (*timing == ExecutionTiming::QueryTime || s.timing == *timing) - && is_exact_accumulator_state(&child.schema).is_ok(); - // A query-time readout may read a population retained at ingestion. - let population_readout = matches!(operation, ValueOperation::ReadPopulation { .. }) - && *timing == ExecutionTiming::QueryTime - && matches!( - &child.expr, - SummaryExpr::ValueOperation { - operation: ValueOperation::MaintainPopulation { .. }, - .. - } - ); - if s != required && !exact_readout && !population_readout { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: ExecutionDataStateEdge::ValueOperationChild.describe(), - child: s, - }); - } - check_plain_operands(operation, &child.schema)?; - visit(child, s, assignment) - } - } -} - -/// The data_state `child` takes as a direct input of `parent`, without -/// validating legality — `child`'s own produced data_state, or for a -/// `KeepPreAsap` leaf the data_state `parent`'s edge assigns it (update-path raw -/// input under maintenance-time operation edges, query-time fallback under a -/// a read-time operation, and — meaninglessly, but for a stable answer — maintenance rows -/// under a state-only edge). For DAG export and other reporting that needs -/// an explicit per-node data_state even on a plan that -/// [`validate_execution_data_states`] would reject. -pub fn assigned_child_data_state(parent: &SummaryExpr, child: &SummaryNode) -> ExecutionDataState { - if let Some(avail) = produced_data_state(&child.expr) { - return avail; - } - match parent { - SummaryExpr::ValueOperation { - timing: ExecutionTiming::QueryTime, - .. - } - | SummaryExpr::BinaryOp { - timing: ExecutionTiming::QueryTime, - .. - } => ExecutionDataState::QUERY_ROWS, - SummaryExpr::KeepPreAsap(_) - | SummaryExpr::BinaryOp { - timing: ExecutionTiming::IngestionTime, - .. - } - | SummaryExpr::RelationalJoin { .. } - | SummaryExpr::SummaryAgg { .. } - | SummaryExpr::SummaryJoin { .. } - | SummaryExpr::SummarySubtract { .. } - | SummaryExpr::SummaryDelete { .. } - | SummaryExpr::SummaryEstimate { .. } - | SummaryExpr::SummaryMerge { .. } - | SummaryExpr::ValueOperation { - timing: ExecutionTiming::IngestionTime, - .. - } => ExecutionDataState::INGESTION_ROWS, - } -} - -/// The data_state `child` takes on `edge`: its own produced data_state -/// (checked via `accept`), or — for a `KeepPreAsap` leaf — the data_state the -/// edge assigns it, derived from what that edge accepts. -fn child_domain( - child: &Rc, - edge: ExecutionDataStateEdge, - accept: impl Fn(ExecutionDataState) -> Result<(), ExecutionDataStateError>, -) -> Result { - match produced_data_state(&child.expr) { - Some(avail) => { - accept(avail)?; - Ok(avail) - } - None => { - // A raw pre-ASAP sub-DAG executes at whichever data_state its consumer - // needs: update-path input for maintenance-time operation edges, - // query-time fallback for a read-time edge. State-only edges - // can't consume plain rows at all. - let assigned = match edge { - ExecutionDataStateEdge::SummaryAggChild - | ExecutionDataStateEdge::SummaryJoinInput - | ExecutionDataStateEdge::ValueOperationChild => ExecutionDataState::INGESTION_ROWS, - ExecutionDataStateEdge::SummaryEstimateInput - | ExecutionDataStateEdge::SummarySubtractInput - | ExecutionDataStateEdge::SummaryDeleteInput - | ExecutionDataStateEdge::SummaryMergeInput => { - return Err(ExecutionDataStateError::IllegalChildDataState { - edge: edge.describe(), - child: ExecutionDataState::INGESTION_ROWS, - }) - } - }; - accept(assigned)?; - Ok(assigned) - } - } -} - -fn state_only( - child: &Rc, - edge: ExecutionDataStateEdge, -) -> Result { - child_domain(child, edge, |avail| match avail { - state - if state.primitive == DataPrimitive::SummaryState - && (state.timing == ExecutionTiming::IngestionTime - || matches!(edge, ExecutionDataStateEdge::SummaryEstimateInput)) => - { - Ok(()) - } - other => Err(ExecutionDataStateError::IllegalChildDataState { - edge: edge.describe(), - child: other, - }), - }) -} - -/// The exact operator must consume only `Plain` columns of its input: for -/// an `Aggregate` payload, every grouping key and every measure's input -/// column. -fn check_plain_operands( - op: &ValueOperation, - input: &Schema, -) -> Result<(), ExecutionDataStateError> { - if matches!( - op, - ValueOperation::Sort { .. } - | ValueOperation::Limit { .. } - | ValueOperation::Project { .. } - | ValueOperation::Filter { .. } - | ValueOperation::FinalizeExactAccumulator - ) { - return check_plain_or_exact_values(input); - } - let ValueOperation::Exact(op) = op else { - return check_all_plain(input); - }; - let ExactOperation::Aggregate { - reduction, - measures, - filters, - .. - } = op; - let mut referenced: Vec = reduction - .group_keys() - .map(|keys| keys.keys().to_vec()) - .unwrap_or_default(); - for m in measures { - referenced.extend(m.input_cols()); - } - for Predicate(f) in filters.iter().flatten() { - referenced.extend(f.columns_referenced().into_iter().copied()); - } - // With no explicit input column (the PromQL sample-value convention) - // the operator reads every non-key column, so all must be plain. - let implicit = measures.iter().any(|m| m.input_cols().is_empty()); - for (i, field) in input.fields.iter().enumerate() { - if !(implicit || referenced.contains(&i)) { - continue; - } - if !matches!(field.dtype, FieldDataType::Plain(_)) { - return Err(ExecutionDataStateError::NonPlainOperand { - column: field.name.clone(), - dtype: format!("{:?}", field.dtype), - }); - } - } - Ok(()) -} - -fn check_plain_or_exact_values(input: &Schema) -> Result<(), ExecutionDataStateError> { - for field in &input.fields { - if !matches!( - field.dtype, - FieldDataType::Plain(_) | FieldDataType::ExactAggregate(..) - ) { - return Err(ExecutionDataStateError::NonPlainOperand { - column: field.name.clone(), - dtype: format!("{:?}", field.dtype), - }); - } - } - Ok(()) -} - -fn check_all_plain(input: &Schema) -> Result<(), ExecutionDataStateError> { - for field in &input.fields { - if !matches!(field.dtype, FieldDataType::Plain(_)) { - return Err(ExecutionDataStateError::NonPlainOperand { - column: field.name.clone(), - dtype: format!("{:?}", field.dtype), - }); - } - } - Ok(()) -} - -/// `schema` with its reuse metadata dropped, or `None` if any field carries -/// summary state — the shape an exact operator reads. -pub fn plain_schema(schema: &Schema) -> Option { - schema - .is_all_plain() - .then(|| Schema::lifted(schema.fields.clone(), schema.time_index)) -} - -/// `schema` as a summary-planning node output: fields and time axis kept, -/// unique keys dropped, closed. -pub fn lift_plain(schema: &Schema) -> Schema { - Schema::lifted(schema.fields.clone(), schema.time_index) -} - -/// Output schema of `op` applied to a child whose edge carries `input` — -/// the same canonical derivation the pre-ASAP `Aggregate` node uses, so an -/// exact `ValueOperation` never disagrees with the pre-ASAP -/// target it was lowered from. `Err` when the child carries non-plain -/// state the operator cannot read. -pub fn exact_operation_output_schema( - op: &ExactOperation, - input: &Schema, -) -> Result { - let plain = plain_schema(input).ok_or(ExactOperationSchemaError::NonPlainInput)?; - let ExactOperation::Aggregate { - reduction, - measures, - output_names, - .. - } = op; - let out = aggregate_output_schema(&plain, reduction, measures, output_names)?; - Ok(lift_plain(&out)) -} - -/// Why [`exact_operation_output_schema`] could not derive a schema. -#[derive(Debug, Error)] -pub enum ExactOperationSchemaError { - #[error("exact operator input carries summary state, not plain columns")] - NonPlainInput, - #[error("schema derivation failed: {0}")] - Schema(#[from] QueryExprError), -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::post_asap::{ExactKind, ExactParams, GroupingStrategy, SketchStatistic}; - use crate::pre_asap::agg_intent::AggIntent; - use crate::pre_asap::expr_ir::ColumnRef; - use crate::pre_asap::query_expr::{QueryExpr, Reduction, Source}; - use crate::pre_asap::schema::{DataType, Field}; - - /// Both execution phases use raw values, distinct from maintained state. - #[test] - fn raw_primitive_labels() { - assert_eq!( - ExecutionDataState::INGESTION_ROWS.primitive, - DataPrimitive::Raw - ); - assert_eq!(ExecutionDataState::QUERY_ROWS.primitive, DataPrimitive::Raw); - assert_eq!( - ExecutionDataState::INGESTION_ROWS.to_string(), - "ingestion_time/raw" - ); - assert_eq!(ExecutionDataState::QUERY_ROWS.to_string(), "query_time/raw"); - assert_eq!(DataPrimitive::SummaryState.as_str(), "summary_state"); - } - - fn scan() -> Rc { - Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - Field::plain("zone", DataType::Utf8, true), - ], - 0, - vec![], - ), - }) - } - - fn keep() -> Rc { - let s = scan(); - let schema = lift_plain(&s.output_schema().unwrap()); - Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(s), - schema, - guarantee: None, - }) - } - - fn plain(names: &[&str]) -> Schema { - Schema::lifted( - names - .iter() - .map(|n| Field { - name: (*n).into(), - dtype: FieldDataType::Plain(DataType::Float64), - nullable: false, - table: None, - }) - .collect(), - None, - ) - } - - fn agg(child: Rc, family: FieldDataType) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: crate::post_asap::SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted( - vec![Field { - name: "state".into(), - dtype: family, - nullable: false, - table: None, - }], - None, - ), - guarantee: None, - }) - } - - fn kll() -> FieldDataType { - use crate::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; - FieldDataType::Sketch( - SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), - GroupingStrategy::default(), - ) - } - - fn estimate(child: Rc) -> Rc { - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryEstimate { - summary_input: child, - query: SketchStatistic::Quantile { q: 0.99 }, - }, - schema: plain(&["quantile_0_99"]), - guarantee: None, - }) - } - - fn max_op() -> ExactOperation { - ExactOperation::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::Max { col: None }], - output_names: vec![], - having: None, - filters: vec![], - } - } - - #[test] - fn keep_pre_asap_under_summary_agg_is_update_input() { - let leaf = keep(); - let root = agg(Rc::clone(&leaf), kll()); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&leaf), - Some(ExecutionDataState::INGESTION_ROWS) - ); - assert_eq!( - assignment.data_state_of(&root), - Some(ExecutionDataState::INGESTION_SUMMARY) - ); - } - - // Typed derived updates retain one opaque series identity only when both - // operands cover the same series; arbitrary labels are never admitted. - #[test] - fn maintenance_binary_accepts_only_well_typed_series_identity() { - use crate::pre_asap::{ArithmeticOpKind, BinaryOpKind}; - let identity = crate::pre_asap::schema::PROMQL_SERIES_IDENTITY; - // A finalized per-series Sum of `metric`'s rows. - let operand = |metric: &str, schema: &Schema| { - let rows = Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { - metric: metric.into(), - }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - }); - let mut state = schema.clone(); - state.fields[0].dtype = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let sum = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child: Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(rows), - schema: schema.clone(), - guarantee: None, - }), - family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), - input: crate::post_asap::SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::PerEntity, - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: state, - guarantee: None, - }); - Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: sum, - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::IngestionTime, - }, - schema: schema.clone(), - guarantee: None, - }) - }; - let validate = |schema: Schema, rhs: &str| { - let binary = Rc::new(SummaryNode { - expr: SummaryExpr::BinaryOp { - lhs: operand("m", &schema), - rhs: operand(rhs, &schema), - timing: ExecutionTiming::IngestionTime, - operator: crate::post_asap::BinaryOperator { - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - vector_match: None, - checked_relative_division: false, - checked_finite_division: false, - }, - }, - schema, - guarantee: None, - }); - validate_execution_data_states(&estimate(agg(binary, kll()))).map(|_| ()) - }; - let mut schema = plain(&["value"]); - schema.fields.push(Field { - table: None, - name: "ts".into(), - dtype: FieldDataType::Plain(DataType::Timestamp), - nullable: false, - }); - schema.time_index = Some(1); - assert!(validate(schema.clone(), "m").is_ok()); - assert!( - validate(schema.clone(), "n").is_ok(), - "no identity to align" - ); - schema.fields.push(Field { - table: None, - name: identity.into(), - dtype: FieldDataType::Plain(DataType::Utf8), - nullable: false, - }); - assert!(validate(schema.clone(), "m").is_ok()); - assert_eq!( - validate(schema.clone(), "n"), - Err(ExecutionDataStateError::InvalidMaintenanceBinary), - "different selectors may cover different series" - ); - for mutation in 0..4 { - let mut invalid = schema.clone(); - match mutation { - 0 => invalid.fields[2].nullable = true, - 1 => invalid.fields[2].dtype = FieldDataType::Plain(DataType::Timestamp), - 2 => invalid.fields.push(invalid.fields[2].clone()), - _ => invalid.fields[2].name = "label".into(), - } - assert_eq!( - validate(invalid, "m"), - Err(ExecutionDataStateError::InvalidMaintenanceBinary) - ); - } - } - - #[test] - fn exact_accumulator_state_may_feed_another_summary_agg() { - let inner = agg( - keep(), - FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), - ); - let root = estimate(agg(inner, kll())); - assert!(validate_execution_data_states(&root).is_ok()); - } - - #[test] - fn readout_can_feed_summary_construction_at_query_time() { - let inner = estimate(agg(keep(), kll())); - let summary = agg(inner, kll()); - let root = estimate(summary.clone()); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&summary).unwrap().timing, - ExecutionTiming::QueryTime - ); - } - - #[test] - fn query_time_operation_over_readout_is_legal_and_root_is_readout() { - let inner = estimate(agg(keep(), kll())); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&root), - Some(ExecutionDataState::QUERY_ROWS) - ); - } - - #[test] - fn non_exact_operator_uses_the_same_read_domain_contract() { - let inner = estimate(agg(keep(), kll())); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Extension { - name: "approximate_calibration".into(), - }, - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["calibrated"]), - guarantee: None, - }); - - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&root), - Some(ExecutionDataState::QUERY_ROWS) - ); - } - - #[test] - fn query_time_values_can_feed_query_time_summary_construction() { - let inner = estimate(agg(keep(), kll())); - let post = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let root = agg(post, kll()); - let root = estimate(root); - validate_execution_data_states(&root).unwrap(); - } - - #[test] - fn function_under_summary_agg_is_legal_but_not_at_root() { - let operation = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: keep(), - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::IngestionTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - assert_eq!( - validate_execution_data_states(&operation).err(), - Some(ExecutionDataStateError::MaintenanceRowsAtRoot) - ); - let root = estimate(agg(Rc::clone(&operation), kll())); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&operation), - Some(ExecutionDataState::INGESTION_ROWS) - ); - } - - #[test] - fn function_over_readout_is_rejected() { - let inner = estimate(agg(keep(), kll())); - let operation = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: inner, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::IngestionTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let root = agg(operation, kll()); - assert!(matches!( - validate_execution_data_states(&root), - Err(ExecutionDataStateError::IllegalChildDataState { - edge: "ValueOperation.child", - child: ExecutionDataState::QUERY_ROWS - }) - )); - } - - #[test] - fn execution_phase_wire_names_are_ingestion_and_query_time() { - for (phase, name) in [ - (ExecutionTiming::IngestionTime, "ingestion_time"), - (ExecutionTiming::QueryTime, "query_time"), - ] { - assert_eq!(phase.as_str(), name); - assert_eq!(serde_json::to_value(phase).unwrap(), name); - assert_eq!( - serde_json::from_value::(serde_json::json!(name)).unwrap(), - phase - ); - } - assert!(serde_json::from_str::("\"maintenance_time\"").is_err()); - assert!(serde_json::from_str::("\"MaintenanceTime\"").is_err()); - } - - #[test] - fn summary_merge_runs_at_ingestion_or_query_time() { - for timing in [ExecutionTiming::IngestionTime, ExecutionTiming::QueryTime] { - let input = agg(keep(), kll()); - let merged = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - children: vec![input.clone()], - timing, - }, - schema: input.schema.clone(), - guarantee: None, - }); - let root = estimate(merged.clone()); - let assignment = validate_execution_data_states(&root).unwrap(); - assert_eq!( - assignment.data_state_of(&merged), - Some(ExecutionDataState { - timing, - primitive: DataPrimitive::SummaryState, - }) - ); - let exported = crate::post_asap::compile_post_asap_dag(&root).unwrap(); - assert!(exported.nodes.iter().any(|node| matches!(node.payload, - crate::post_asap::PostAsapOperatorPayload::SummaryMerge - if node.output_state.timing == timing))); - } - } - - #[test] - fn ingestion_merge_cannot_depend_on_query_execution() { - let input = agg(keep(), kll()); - let query_merge = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - children: vec![input.clone()], - timing: ExecutionTiming::QueryTime, - }, - schema: input.schema.clone(), - guarantee: None, - }); - let ingestion_merge = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - children: vec![query_merge], - timing: ExecutionTiming::IngestionTime, - }, - schema: input.schema.clone(), - guarantee: None, - }); - assert!(validate_execution_data_states(&ingestion_merge).is_err()); - } - - #[test] - fn a_shared_keep_pre_asap_reached_in_two_domains_is_ambiguous() { - // One raw sub-DAG used both as update input (under a SummaryAgg) and - // as a query-time fallback (under an ExactRead) — no single - // execution can serve both, so the plan is rejected. - let shared = keep(); - let maintained = estimate(agg(Rc::clone(&shared), kll())); - let post_over_raw = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: Rc::clone(&shared), - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::SummaryMerge { - timing: ExecutionTiming::IngestionTime, - children: vec![ - Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: maintained, - operation: ValueOperation::Exact(max_op()), - timing: ExecutionTiming::QueryTime, - }, - schema: plain(&["max"]), - guarantee: None, - }), - post_over_raw, - ], - }, - schema: plain(&["max"]), - guarantee: None, - }); - // SummaryMerge only accepts state, so this fails earlier for a - // different reason; probe the ambiguity through a direct visit. - let mut assignment = ExecutionDataStateAssignment::default(); - visit(&shared, ExecutionDataState::INGESTION_ROWS, &mut assignment).unwrap(); - assert_eq!( - visit(&shared, ExecutionDataState::QUERY_ROWS, &mut assignment), - Err(ExecutionDataStateError::AmbiguousKeepPreAsap { - first: ExecutionDataState::INGESTION_ROWS, - second: ExecutionDataState::QUERY_ROWS, - }) - ); - assert!(validate_execution_data_states(&root).is_err()); - } - - // Both paired operands must be plain; an unrelated state column is not an input. - #[test] - fn pearson_corr_checks_both_operand_states() { - let operation = ValueOperation::Exact(ExactOperation::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::PearsonCorr { left: 0, right: 1 }], - output_names: vec![], - having: None, - filters: vec![], - }); - for operand in [0, 1] { - let mut input = plain(&["x", "y", "unused"]); - input.fields[operand].dtype = kll(); - assert!(matches!( - check_plain_operands(&operation, &input), - Err(ExecutionDataStateError::NonPlainOperand { .. }) - )); - } - let mut input = plain(&["x", "y", "unused"]); - input.fields[2].dtype = kll(); - check_plain_operands(&operation, &input).unwrap(); - } - - #[test] - fn exact_operator_schema_matches_pre_asap_aggregate_derivation() { - let child_schema = lift_plain(&scan().output_schema().unwrap()); - let op = ExactOperation::Aggregate { - reduction: Reduction::by(vec![2]), - measures: vec![AggIntent::Max { col: None }], - output_names: vec![], - having: None, - filters: vec![], - }; - let out = exact_operation_output_schema(&op, &child_schema).unwrap(); - let names: Vec<_> = out.fields.iter().map(|f| f.name.as_str()).collect(); - assert_eq!(names, vec!["zone", "max"]); - assert!(out - .fields - .iter() - .all(|f| matches!(f.dtype, FieldDataType::Plain(_)))); - } - - #[test] - fn exact_operator_rejects_non_plain_input() { - let state = agg(keep(), kll()); - assert!(matches!( - exact_operation_output_schema(&max_op(), &state.schema), - Err(ExactOperationSchemaError::NonPlainInput) - )); - } -} diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs deleted file mode 100644 index 7de1e81db..000000000 --- a/crates/types/src/post_asap/expr.rs +++ /dev/null @@ -1,285 +0,0 @@ -use super::ExecutionTiming; -use std::rc::Rc; - -use super::guarantee::ResultGuarantee; -use super::sketch::{GroupingStrategy, SketchStatistic, SummaryUpdate}; -use crate::pre_asap::agg_intent::AggIntent; -use crate::pre_asap::query_expr::Predicate; -use crate::pre_asap::schema::{FieldDataType, Schema}; -use crate::pre_asap::{ - BinaryOpKind, ColumnRef, GroupKeys, JoinKind, ProjectItem, QueryExpr, Reduction, SortKey, - VectorMatch, -}; - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[non_exhaustive] -pub enum ExactOperation { - Aggregate { - reduction: Reduction, - measures: Vec, - output_names: Vec, - /// Per-measure row predicates parallel to `measures`, positional - /// against the child's output rows — the same contract as - /// `QueryExpr::Aggregate.filters` (issue #466). - #[serde(default)] - filters: Vec>, - having: Option, - }, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[non_exhaustive] -pub enum ValueOperation { - /// Maintain the full declared population, including membership changes, - /// so removing a TopK member can promote another. - MaintainPopulation { - population: super::maintained_population::MaintainedPopulation, - }, - /// Read an aggregate or TopK prefix from the maintained population. - ReadPopulation { - readout: super::maintained_population::PopulationStatistic, - }, - Exact(ExactOperation), - /// Read an exact accumulator's state as its finalized scalar value. - /// - /// Exact accumulators do not need an estimator, but the explicit node - /// marks the maintenance-to-read boundary before query-time operators - /// such as PromQL binary arithmetic, sorting, and limiting. - FinalizeExactAccumulator, - /// Query-time column projection. SQL lowering retains the SELECT list as - /// a `Project` above its aggregate, so the post-ASAP DAG must preserve - /// its expressions, aliases, and optional derived-table qualifier while - /// allowing the aggregate child to be planned independently. - Project { - cols: Vec, - qualifier: Option, - }, - /// Query-time row filtering. The predicate remains positional against - /// the child's output schema and is evaluated only after any summary - /// state below it has been read out to rows. - Filter { - pred: Predicate, - }, - /// Query-time ordering of the child's value rows. This is deliberately - /// distinct from frequency-sketch heavy-hitter readout: PromQL `topk` - /// ranks the values produced by its child at the evaluation timestamp. - Sort { - keys: Vec, - partition_by: GroupKeys, - }, - /// Query-time row selection, normally composed over [`Self::Sort`] for - /// PromQL `topk`/`bottomk` and SQL `ORDER BY … LIMIT`. - Limit { - n: usize, - offset: usize, - /// Apply the offset and limit independently to each group. - partition_by: GroupKeys, - }, - Extension { - name: String, - }, -} - -/// Whether a candidate-membership sidecar is proven to contain every true -/// top-k key or is an explicitly approximate optimization. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(tag = "kind", rename_all = "snake_case")] -pub enum CandidateCompleteness { - Certified { guarantee: ResultGuarantee }, - BestEffort { guarantee: Option }, -} - -// ── Post-ASAP DAG node ─────────────────────────────────────────────────────── - -/// A node in the post-ASAP DAG: wraps the expression and its derived output -/// schema so every edge carries a typed schema. `Schema` may contain -/// summary-state-typed columns (`FieldDataType`'s non-`Plain` variants); -/// the pre-ASAP `Schema` cannot. -#[derive(Debug, Clone, PartialEq)] -pub struct SummaryNode { - pub expr: SummaryExpr, - /// Output schema of `expr` — the schema of the data flowing on the edge - /// leading *from* this node to its parent(s). - pub schema: Schema, - /// The machine-readable accuracy guarantee of the *value* this node - /// produces (issue #172) — `Some` on every finalized, caller-visible - /// value: a `SummaryEstimate` readout, an `ExactAggregate`-family - /// `SummaryAgg` (its state *is* the value), or a `KeepPreAsap` sub-DAG - /// (executed exactly). `None` on raw summary state — a sketch-family - /// `SummaryAgg`, `SummaryMerge`, `SummarySubtract`, `SummaryDelete`, - /// `SummaryJoin` — whose guarantee only exists once something reads it - /// out; and `None` on a readout of a family the plugged-in - /// `AccuracyModel` has no local guarantee for (`Sample`/`Wavelet`/ - /// `StatModel`), which a fail-closed consumer must treat as "unknown", - /// never as exact. - pub guarantee: Option, -} - -// ── Post-ASAP sketch-bound IR ──────────────────────────────────────────────── - -/// Sketch-bound IR produced by post-ASAP binding and final selection. Binding -/// rules selectively replace logical aggregates and joins in the pre-ASAP -/// `QueryExpr` with summary-bound counterparts. Final selection can retain -/// supported read-time value operations around independently planned children; -/// other unsupported sub-DAGs pass through as `KeepPreAsap(Rc)`. -/// -/// Traversing from the root node yields a DAG; shared sub-expressions appear -/// as multiple `Rc` references to the same `SummaryNode`. -#[derive(Debug, Clone, PartialEq)] -pub enum SummaryExpr { - /// A pre-ASAP sub-DAG kept as-is because it has no selected implementation - /// or supported residual decomposition. Output schema is the inner node's - /// schema, lifted to `Schema` with all fields as - /// `FieldDataType::Plain`. - KeepPreAsap(Rc), - - /// A PromQL binary operation whose operands were planned independently. - /// This keeps realizable summary/readout leaves visible instead of - /// hiding the complete expression inside `KeepPreAsap`. - BinaryOp { - timing: ExecutionTiming, - lhs: Rc, - rhs: Rc, - operator: BinaryOperator, - }, - - /// Plain-row semantics composed with a post-ASAP child. Timing is an - /// independent physical choice, not part of the operation's identity. - ValueOperation { - child: Rc, - operation: ValueOperation, - timing: super::execution_data_state::ExecutionTiming, - }, - - /// Read-time relational join over two row-producing children. This is - /// distinct from [`SummaryJoin`](Self::SummaryJoin), which combines - /// summary states for join estimation during maintenance. - RelationalJoin { - left: Rc, - right: Rc, - kind: JoinKind, - pred: Predicate, - /// Optional proof for candidate pruning; ranking remains a separate operation. - pruning: Option, - }, - - /// Summary aggregation. Post-ASAP binding chose `family` — which - /// summary family (exact accumulator, sketch, sample, wavelet, or - /// statistical model) and its `(kind, params)` — from the catalog for - /// `AggIntent` under `DeploymentConstraints`. - /// Output schema: grouping columns (verbatim) + one field carrying - /// partial summary state per group, typed `family`. - SummaryAgg { - child: Rc, - /// Which summary family realizes this aggregation, and that - /// family's own `(kind, params)`. Never `FieldDataType::Plain` - /// — this node always produces summary state, not a plain value. - family: FieldDataType, - /// Optional multidimensional item identity and the observation/update - /// weight fed into each state update. Subpopulation semantics remain - /// on `reduction`; physical sharing remains on `grouping`. - input: SummaryUpdate, - /// How this aggregation's output rows relate to `child`'s — the - /// same [`Reduction`] the pre-ASAP `Aggregate` node it was bound - /// from carried (issue #165), reused verbatim rather than - /// flattened to a bare `Vec`. `Reduction::Reduce(by)` - /// with an empty `by` is a genuine full reduction (merge every - /// candidate into one group); `Reduction::PerEntity` has no - /// grouping concept at all (never merge across entities) — the - /// two collapsed to the same ambiguous `by: []` before this field - /// existed (issue #163). - reduction: Reduction, - /// How this aggregation's summary state is physically instantiated - /// across `reduction`'s subpopulations — one independent instance - /// per `by` key (today's only behavior, and this field's default), - /// or one shared Hydra-family structure serving all of them (issue - /// #256). Lives here, next to `reduction`, for planning, and is also - /// encoded in sketch-valued `family`/output-schema state so merges - /// can reject incompatible layouts. `reduction` is the field that - /// carries the `by` keys this axis's legality depends on (a - /// `SharedMultiSubpopulation` choice only makes sense when - /// `reduction` actually has a subpopulation concept — see - /// `asap_aware_mapping::grouping`'s module docs for the legality - /// rules). Every existing producer of a `SummaryAgg` sets this to - /// `GroupingStrategy::PerSubpopulationInstance` (its `Default`), - /// so no existing behavior changes. - grouping: GroupingStrategy, - /// Row predicate gating this summary's updates (issue #466): only - /// rows where it is `TRUE` update the state; grouping keys are - /// still read from every row. Positional against `child`'s output. - /// A field rather than a `Filter` child so summaries that differ - /// only in predicate can still share one child. No binding rule - /// sets it yet — a filtered pre-ASAP measure stays `KeepPreAsap` — - /// so every producer today writes `None`. - filter: Option, - }, - - /// Summary-aware join (KMV / theta for join-cardinality; join-sample for - /// sampling). Emitted only when a `Bind*OnJoin` rule fires. - /// Output schema: one field typed `family`, read by a downstream - /// `SummaryEstimate`. - SummaryJoin { - outer: Rc, - inner: Rc, - key: ColumnRef, - /// Never `FieldDataType::Plain` — see [`SummaryAgg::family`](SummaryExpr::SummaryAgg). - family: FieldDataType, - }, - - /// Subtract one summary from another. Valid only for families with a - /// linear-inverse property (CMS, theta, count-based). Catalog flag - /// `subtractable` must be true for the family. - /// Output schema: one field (same family + params as inputs). - SummarySubtract { - left: Rc, - right: Rc, - }, - - /// Delete a key from a summary (CMS update with −1, deletable Bloom - /// filter). Catalog flag `deletable` must be true. Output schema = - /// input schema unchanged in type (same field type as input). - SummaryDelete { - summary_input: Rc, - key: ColumnRef, - }, - - /// Read out a query result from a built summary. The summary-state field - /// type does *not* propagate downstream of an estimate — the output - /// schema is a regular row-shaped schema (Float64 for quantile, Int64 - /// for count/cardinality, `[(key, count)]` for top-k). - SummaryEstimate { - summary_input: Rc, - query: SketchStatistic, - }, - - /// ⊕ — union of summaries across stages / shards. Distinct from the - /// pre-ASAP `Concat` because summary union has type constraints: all - /// inputs must agree on `family` (kind + params) and the catalog flag - /// `mergeable` must be true. Inserted by a deployment's own stage - /// allocator (not modeled in this crate) on cut edges. - /// Output schema: one field (same family + params as inputs). - SummaryMerge { - children: Vec>, - timing: ExecutionTiming, - }, -} - -/// All semantics owned by a post-ASAP binary operator. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -pub struct BinaryOperator { - /// Execute division only for finite operands, a nonzero divisor, and a - /// normal finite result; otherwise use exact execution. Required by the - /// relative-value division certificate, including floating-point range. - #[serde(default)] - pub checked_relative_division: bool, - /// Conditional exact rewrites (such as temporal average from sum/count) - /// require finite operands and quotient. Zero/subnormal results are valid; - /// overflow must fall back to the original query rather than emit infinity. - #[serde(default)] - pub checked_finite_division: bool, - pub kind: BinaryOpKind, - /// `None` is the only currently supported vector/vector matching mode. - /// The field is retained so execution never has to recover semantics by - /// re-parsing PromQL. - pub vector_match: Option, -} diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs deleted file mode 100644 index f71aaca0f..000000000 --- a/crates/types/src/post_asap/mod.rs +++ /dev/null @@ -1,82 +0,0 @@ -//! The post-ASAP IR: summary-bound types, distinct from -//! [`crate::pre_asap`]'s pre-ASAP IR. -//! -//! Where [`crate::pre_asap`] carries *intent* only ("compute a -//! quantile to ε accuracy"), this module is the summary-bound IR: the -//! summary family, kind/algorithm, and parameters are committed (one -//! `(Kind, Params)` pair per family — [`sketch::ExactKind`]/[`sketch::ExactParams`], -//! [`sketch::SamplingKind`]/[`sketch::SamplingParams`], -//! [`sketch::WaveletKind`]/[`sketch::WaveletParams`], -//! [`sketch::StatModelKind`]/[`sketch::StatModelParams`]), and -//! [`expr::SummaryNode`] / [`expr::SummaryExpr`] describe the summary -//! computation. The `Sketch` family is the one exception to that -//! one-pair-per-family shape: it nests a third level, [`sketch::SketchKind`] -//! (quantile/cardinality/frequency/top-k), which itself carries the -//! committed [`sketch::SketchAlgorithm`] and [`sketch::SketchParams`] — -//! `FieldDataType::Sketch(SketchKind, GroupingStrategy)`, not a flat -//! `(kind, params)` pair -//! — because `Sketch` is the one family with more than one algorithm per -//! purpose today; no other family needs that extra level yet. -//! -//! A second, orthogonal axis lives here too: [`sketch::GroupingStrategy`] -//! (issue #256) — *how many* physical instances of a chosen family/kind -//! exist across a grouped aggregate's `by` subpopulations -//! (`PerSubpopulationInstance`, today's only behavior, vs. -//! `SharedMultiSubpopulation`/Hydra — see [`sketch::HydraKind`]/ -//! [`sketch::HydraParams`]), carried on [`expr::SummaryExpr::SummaryAgg`] -//! alongside `reduction` and on sketch-valued edge types -//! — see `asap_aware_mapping::grouping`'s module docs for why. - -pub mod cse; -pub mod execution_data_state; -pub mod expr; -pub mod guarantee; -pub mod maintained_population; -pub mod post_asap_dag; -pub mod query_time; -pub mod sketch; -pub mod summary_maintenance; -pub mod summary_maintenance_lifecycle; -pub mod summary_window; - -pub use crate::pre_asap::schema::{Field, FieldDataType, Schema}; -pub use cse::share_common_summary_sub_dags; -pub use execution_data_state::{ - assigned_child_data_state, exact_operation_output_schema, produced_data_state, - validate_execution_data_states, validate_execution_data_states_at, DataPrimitive, - ExactOperationSchemaError, ExecutionDataState, ExecutionDataStateAssignment, - ExecutionDataStateError, ExecutionTiming, -}; -pub use expr::{ - BinaryOperator, CandidateCompleteness, ExactOperation, SummaryExpr, SummaryNode, ValueOperation, -}; -pub use guarantee::{ - AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, GuaranteeSource, ProbabilityExpr, - ResultGuarantee, -}; -pub use post_asap_dag::{ - compile_post_asap_dag, compile_post_asap_dag_with_node_ids, EdgeRole, - GroupingEdgeCompatibility, PostAsapDAG, PostAsapDAGCompilation, PostAsapDAGDocument, - PostAsapDAGEdge, PostAsapDAGNode, PostAsapDAGValidationError, PostAsapNodeId, - PostAsapNodeIdentityMap, PostAsapOperatorPayload, WindowEdgeCompatibility, - POST_ASAP_DAG_WIRE_VERSION, -}; -pub use query_time::{ - classic_cms_sizing, cms_posterior_error_bound, count_sketch_posterior_error_bound, - cu_sketch_posterior_error_bound, traditional_a_priori_bound, -}; -pub use sketch::{ - default_hydra_params, hydra_kind_for, EntityIdentity, ExactKind, ExactParams, GroupingStrategy, - HydraKind, HydraParams, NonNegativeWeightProof, SamplingKind, SamplingParams, SketchAlgorithm, - SketchCategory, SketchKind, SketchParams, SketchStatistic, StatModelKind, StatModelParams, - SummaryInputExpr, SummaryUpdate, WaveletKind, WaveletParams, WeightDomain, -}; -pub use summary_maintenance::SummaryMaintenanceMode; -pub use summary_maintenance_lifecycle::{ - EvaluationSchedule, OutputRepresentation, SummaryMaintenanceLifecycle, - SummaryMaintenanceLifecycleGuarantee, -}; -pub use summary_window::{ - plan_pane_phase, validate_pane_coverage, PaneCoverageError, PaneLayout, SummaryWindowFramework, - WindowEdgeCoverage, -}; diff --git a/crates/types/src/post_asap/post_asap_dag.rs b/crates/types/src/post_asap/post_asap_dag.rs deleted file mode 100644 index d489c055e..000000000 --- a/crates/types/src/post_asap/post_asap_dag.rs +++ /dev/null @@ -1,867 +0,0 @@ -//! Runtime-neutral post-ASAP DAG contract shared by precompute and query engines. - -use std::collections::HashMap; -use std::rc::Rc; - -use super::{ - validate_execution_data_states, ExecutionDataState, ExecutionDataStateError, ResultGuarantee, - Schema, SummaryExpr, SummaryNode, -}; -use super::{ - BinaryOperator, CandidateCompleteness, ExecutionTiming, FieldDataType, GroupingStrategy, - SketchStatistic, SummaryUpdate, ValueOperation, -}; -use crate::pre_asap::{ColumnRef, JoinKind, Predicate, QueryExpr, Reduction}; -use thiserror::Error; - -pub const POST_ASAP_DAG_WIRE_VERSION: u32 = 6; - -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub enum EdgeRole { - Input, - Left, - Right, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub enum GroupingEdgeCompatibility { - Identical, - ConsumerCoarsensProducer, - Incompatible, - NotApplicable, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub enum WindowEdgeCompatibility { - /// Physical lowering must prove equal pane/query phase or install an - /// exact boundary residual. The logical DAG alone cannot make that claim. - #[serde(rename = "RequiresAlignedPanePhaseOrExactBoundaryResidual")] - RequiresAlignedPanePhaseOrExactWindowEdgeResidual, - NotApplicable, -} - -/// Stable identity of a node within one exported post-ASAP semantic DAG. -#[derive( - Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, serde::Serialize, serde::Deserialize, -)] -#[serde(transparent)] -pub struct PostAsapNodeId(pub u32); - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] -pub enum PostAsapOperatorPayload { - Fallback { - expression: QueryExpr, - }, - Binary { - operator: BinaryOperator, - }, - Value { - operation: ValueOperation, - }, - RelationalJoin { - join_kind: JoinKind, - pred: Predicate, - pruning: Option, - }, - SummaryAgg { - family: FieldDataType, - input: SummaryUpdate, - reduction: Reduction, - grouping: GroupingStrategy, - /// See `SummaryExpr::SummaryAgg::filter`. Wire version 6 added it; - /// a version-5 reader would otherwise take a filtered summary as - /// unfiltered. - filter: Option, - }, - SummaryJoin { - key: ColumnRef, - family: FieldDataType, - }, - SummarySubtract, - SummaryDelete { - key: ColumnRef, - }, - SummaryEstimate { - query: SketchStatistic, - }, - SummaryMerge, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDAGNode { - pub id: PostAsapNodeId, - /// The payload variant is the sole operator identity (`payload.kind` in JSON). - pub payload: PostAsapOperatorPayload, - /// Phase is a placement choice for every operator, independent of payload kind. - pub output_state: ExecutionDataState, - pub output_schema: Schema, - pub guarantee: Option, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDAGEdge { - pub producer: PostAsapNodeId, - pub consumer: PostAsapNodeId, - pub role: EdgeRole, - pub intermediate_schema: Schema, - pub data_state: ExecutionDataState, - pub grouping: GroupingEdgeCompatibility, - pub window: WindowEdgeCompatibility, -} - -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDAG { - pub nodes: Vec, - pub edges: Vec, - /// Semantic workload root. Physical query/precompute sinks are selected - /// downstream by the control plane. - pub root: PostAsapNodeId, -} - -/// Versioned transport envelope for a post-ASAP semantic DAG. -/// -/// Process boundaries exchange this envelope and call [`Self::validate`]. -#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct PostAsapDAGDocument { - pub schema_version: u32, - pub dag: PostAsapDAG, -} - -#[derive(Debug, Clone, PartialEq, Eq, Error)] -pub enum PostAsapDAGValidationError { - #[error("phase assignment must name every DAG node exactly once")] - IncompletePhaseAssignment, - #[error("ingestion node {consumer:?} depends on query node {producer:?}")] - QueryDependencyInIngestion { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, - }, - #[error("unsupported post-ASAP DAG schema version {0}")] - UnsupportedVersion(u32), - #[error("duplicate post-ASAP node id {0:?}")] - DuplicateNodeId(PostAsapNodeId), - #[error("post-ASAP DAG root {0:?} does not name a node")] - MissingRoot(PostAsapNodeId), - #[error("edge endpoint {0:?} does not name a node")] - MissingEdgeEndpoint(PostAsapNodeId), - #[error("edge {producer:?}->{consumer:?} schema differs from producer output")] - EdgeSchemaMismatch { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, - }, - #[error("edge {producer:?}->{consumer:?} data state differs from producer output")] - EdgeDataStateMismatch { - producer: PostAsapNodeId, - consumer: PostAsapNodeId, - }, - #[error("post-ASAP DAG contains a cycle")] - Cycle, - #[error("post-ASAP node {0:?} is not reachable from the root")] - UnreachableNode(PostAsapNodeId), - #[error("summary aggregate node {node:?} output schema does not contain its declared family")] - SummaryFamilySchemaMismatch { node: PostAsapNodeId }, - #[error( - "summary aggregate node {node:?} declares grouping inconsistent with its sketch state" - )] - SummaryGroupingMismatch { node: PostAsapNodeId }, -} - -impl PostAsapDAGDocument { - pub fn new(dag: PostAsapDAG) -> Self { - Self { - schema_version: POST_ASAP_DAG_WIRE_VERSION, - dag, - } - } - - pub fn validate(&self) -> Result<(), PostAsapDAGValidationError> { - if self.schema_version != POST_ASAP_DAG_WIRE_VERSION { - return Err(PostAsapDAGValidationError::UnsupportedVersion( - self.schema_version, - )); - } - self.dag.validate() - } -} - -impl PostAsapDAG { - /// Assign execution phases without changing operator semantics. Phase choices - /// do not prove deployment support: callers must bind concrete implementations - /// and storage boundaries before installing this plan. - pub fn with_execution_phases( - &self, - phases: &std::collections::BTreeMap, - ) -> Result { - self.validate()?; - if phases.len() != self.nodes.len() - || self.nodes.iter().any(|node| !phases.contains_key(&node.id)) - { - return Err(PostAsapDAGValidationError::IncompletePhaseAssignment); - } - let mut dag = self.clone(); - for node in &mut dag.nodes { - node.output_state.timing = phases[&node.id]; - } - let states: HashMap<_, _> = dag.nodes.iter().map(|n| (n.id, n.output_state)).collect(); - for edge in &mut dag.edges { - edge.data_state = states[&edge.producer]; - } - dag.validate()?; - Ok(dag) - } - - pub fn validate(&self) -> Result<(), PostAsapDAGValidationError> { - use std::collections::{HashMap, HashSet}; - let mut nodes = HashMap::new(); - for node in &self.nodes { - if nodes.insert(node.id, node).is_some() { - return Err(PostAsapDAGValidationError::DuplicateNodeId(node.id)); - } - if let PostAsapOperatorPayload::SummaryAgg { - family, grouping, .. - } = &node.payload - { - let mut found_family = false; - for field in &node.output_schema.fields { - if &field.dtype == family { - found_family = true; - } - if let FieldDataType::Sketch(_, schema_grouping) = &field.dtype { - if schema_grouping != grouping { - return Err(PostAsapDAGValidationError::SummaryGroupingMismatch { - node: node.id, - }); - } - } - } - if !found_family { - return Err(PostAsapDAGValidationError::SummaryFamilySchemaMismatch { - node: node.id, - }); - } - } - } - if !nodes.contains_key(&self.root) { - return Err(PostAsapDAGValidationError::MissingRoot(self.root)); - } - let mut children: HashMap> = HashMap::new(); - for edge in &self.edges { - let producer = nodes.get(&edge.producer).ok_or( - PostAsapDAGValidationError::MissingEdgeEndpoint(edge.producer), - )?; - if !nodes.contains_key(&edge.consumer) { - return Err(PostAsapDAGValidationError::MissingEdgeEndpoint( - edge.consumer, - )); - } - if producer.output_state.timing == ExecutionTiming::QueryTime - && nodes[&edge.consumer].output_state.timing == ExecutionTiming::IngestionTime - { - return Err(PostAsapDAGValidationError::QueryDependencyInIngestion { - producer: edge.producer, - consumer: edge.consumer, - }); - } - if edge.intermediate_schema != producer.output_schema { - return Err(PostAsapDAGValidationError::EdgeSchemaMismatch { - producer: edge.producer, - consumer: edge.consumer, - }); - } - if edge.data_state != producer.output_state { - return Err(PostAsapDAGValidationError::EdgeDataStateMismatch { - producer: edge.producer, - consumer: edge.consumer, - }); - } - children - .entry(edge.consumer) - .or_default() - .push(edge.producer); - } - fn visit( - id: PostAsapNodeId, - children: &HashMap>, - visiting: &mut HashSet, - visited: &mut HashSet, - ) -> bool { - if visited.contains(&id) { - return true; - } - if !visiting.insert(id) { - return false; - } - if children - .get(&id) - .into_iter() - .flatten() - .any(|child| !visit(*child, children, visiting, visited)) - { - return false; - } - visiting.remove(&id); - visited.insert(id); - true - } - if !visit( - self.root, - &children, - &mut HashSet::new(), - &mut HashSet::new(), - ) { - return Err(PostAsapDAGValidationError::Cycle); - } - let mut reachable = HashSet::new(); - fn mark( - id: PostAsapNodeId, - children: &HashMap>, - reachable: &mut HashSet, - ) { - if !reachable.insert(id) { - return; - } - for child in children.get(&id).into_iter().flatten() { - mark(*child, children, reachable); - } - } - mark(self.root, &children, &mut reachable); - if let Some(id) = nodes.keys().find(|id| !reachable.contains(id)) { - return Err(PostAsapDAGValidationError::UnreachableNode(*id)); - } - Ok(()) - } -} - -/// Compiler-local identity assignment. It deliberately retains `Rc` handles -/// and is not serialized; deployed artifacts persist the post-ASAP node ID -/// together with their physical materialization/query IDs. -#[derive(Debug, Clone)] -pub struct PostAsapNodeIdentityMap { - nodes_by_id: Vec>, -} - -impl PostAsapNodeIdentityMap { - pub fn node_id(&self, node: &Rc) -> Option { - self.nodes_by_id - .iter() - .position(|candidate| Rc::ptr_eq(candidate, node)) - .map(|id| PostAsapNodeId(id as u32)) - } - - pub fn summary_node(&self, id: PostAsapNodeId) -> Option<&Rc> { - self.nodes_by_id.get(id.0 as usize) - } -} - -#[derive(Debug, Clone)] -pub struct PostAsapDAGCompilation { - pub dag: PostAsapDAG, - pub node_ids: PostAsapNodeIdentityMap, -} - -pub fn compile_post_asap_dag( - root: &Rc, -) -> Result { - Ok(compile_post_asap_dag_with_node_ids(root)?.dag) -} - -pub fn compile_post_asap_dag_with_node_ids( - root: &Rc, -) -> Result { - let assignment = validate_execution_data_states(root)?; - let mut nodes = Vec::new(); - let mut edges = Vec::new(); - let mut ids = HashMap::new(); - let mut nodes_by_id = Vec::new(); - - fn visit( - node: &Rc, - assignment: &super::ExecutionDataStateAssignment, - ids: &mut HashMap<*const SummaryNode, PostAsapNodeId>, - nodes: &mut Vec, - edges: &mut Vec, - nodes_by_id: &mut Vec>, - ) -> PostAsapNodeId { - if let Some(id) = ids.get(&Rc::as_ptr(node)) { - return *id; - } - let children: Vec<(&Rc, EdgeRole)> = match &node.expr { - SummaryExpr::KeepPreAsap(_) => vec![], - SummaryExpr::BinaryOp { lhs, rhs, .. } => { - vec![(lhs, EdgeRole::Left), (rhs, EdgeRole::Right)] - } - - SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { - vec![(child, EdgeRole::Input)] - } - SummaryExpr::RelationalJoin { left, right, .. } => { - vec![(left, EdgeRole::Left), (right, EdgeRole::Right)] - } - SummaryExpr::SummaryJoin { outer, inner, .. } => { - vec![(outer, EdgeRole::Left), (inner, EdgeRole::Right)] - } - SummaryExpr::SummarySubtract { left, right } => { - vec![(left, EdgeRole::Left), (right, EdgeRole::Right)] - } - SummaryExpr::SummaryDelete { summary_input, .. } - | SummaryExpr::SummaryEstimate { summary_input, .. } => { - vec![(summary_input, EdgeRole::Input)] - } - SummaryExpr::SummaryMerge { children, .. } => { - children.iter().map(|c| (c, EdgeRole::Input)).collect() - } - }; - let child_ids: Vec<_> = children - .iter() - .map(|(c, r)| (visit(c, assignment, ids, nodes, edges, nodes_by_id), *c, *r)) - .collect(); - let id = PostAsapNodeId(nodes.len() as u32); - let state = assignment - .data_state_of(node) - .expect("validated node has state"); - let payload = match &node.expr { - SummaryExpr::KeepPreAsap(expression) => PostAsapOperatorPayload::Fallback { - expression: (**expression).clone(), - }, - SummaryExpr::BinaryOp { operator, .. } => PostAsapOperatorPayload::Binary { - operator: operator.clone(), - }, - - SummaryExpr::ValueOperation { operation, .. } => PostAsapOperatorPayload::Value { - operation: operation.clone(), - }, - SummaryExpr::RelationalJoin { - kind, - pred, - pruning, - .. - } => PostAsapOperatorPayload::RelationalJoin { - join_kind: kind.clone(), - pred: pred.clone(), - pruning: pruning.clone(), - }, - SummaryExpr::SummaryAgg { - family, - input, - reduction, - grouping, - filter, - .. - } => PostAsapOperatorPayload::SummaryAgg { - family: family.clone(), - input: input.clone(), - reduction: reduction.clone(), - grouping: grouping.clone(), - filter: filter.clone(), - }, - SummaryExpr::SummaryJoin { key, family, .. } => PostAsapOperatorPayload::SummaryJoin { - key: key.clone(), - family: family.clone(), - }, - SummaryExpr::SummarySubtract { .. } => PostAsapOperatorPayload::SummarySubtract, - SummaryExpr::SummaryDelete { key, .. } => { - PostAsapOperatorPayload::SummaryDelete { key: key.clone() } - } - SummaryExpr::SummaryEstimate { query, .. } => { - PostAsapOperatorPayload::SummaryEstimate { - query: query.clone(), - } - } - SummaryExpr::SummaryMerge { .. } => PostAsapOperatorPayload::SummaryMerge, - }; - nodes.push(PostAsapDAGNode { - id, - payload, - output_state: state, - output_schema: node.schema.clone(), - guarantee: node.guarantee.clone(), - }); - nodes_by_id.push(Rc::clone(node)); - ids.insert(Rc::as_ptr(node), id); - for (producer, child, role) in child_ids { - let maintenance_dependency = nodes[producer.0 as usize].output_state.timing - == ExecutionTiming::IngestionTime - && nodes[id.0 as usize].output_state.timing == ExecutionTiming::IngestionTime; - let grouping = match (&child.expr, &node.expr) { - ( - SummaryExpr::SummaryAgg { - reduction: producer, - .. - }, - SummaryExpr::SummaryAgg { - reduction: consumer, - .. - }, - ) if producer == consumer => GroupingEdgeCompatibility::Identical, - ( - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::PerEntity, - .. - }, - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::Reduce(_), - .. - }, - ) => GroupingEdgeCompatibility::ConsumerCoarsensProducer, - ( - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::Reduce(producer), - .. - }, - SummaryExpr::SummaryAgg { - reduction: crate::pre_asap::Reduction::Reduce(consumer), - .. - }, - ) if !producer.is_without() - && !consumer.is_without() - && consumer.iter().all(|key| producer.contains(key)) => - { - GroupingEdgeCompatibility::ConsumerCoarsensProducer - } - (SummaryExpr::SummaryAgg { .. }, SummaryExpr::SummaryAgg { .. }) => { - GroupingEdgeCompatibility::Incompatible - } - _ => GroupingEdgeCompatibility::NotApplicable, - }; - edges.push(PostAsapDAGEdge { - producer, - consumer: id, - role, - intermediate_schema: child.schema.clone(), - // The whole-DAG validator owns contextual state assignment, - // especially for shared KeepPreAsap leaves. Export that - // authoritative result instead of independently deriving the - // edge state a second time. - data_state: assignment - .data_state_of(child) - .expect("validated child has data state"), - grouping, - window: if maintenance_dependency { - WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual - } else { - WindowEdgeCompatibility::NotApplicable - }, - }); - } - id - } - - let root = visit( - root, - &assignment, - &mut ids, - &mut nodes, - &mut edges, - &mut nodes_by_id, - ); - let dag = PostAsapDAG { nodes, edges, root }; - dag.validate() - .expect("compiler emits a valid post-ASAP DAG"); - Ok(PostAsapDAGCompilation { - dag, - node_ids: PostAsapNodeIdentityMap { nodes_by_id }, - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::post_asap::{ - ExactKind, ExactParams, ExecutionTiming, FieldDataType, GroupingStrategy, SummaryUpdate, - ValueOperation, - }; - use crate::pre_asap::schema::{Field, Schema}; - use crate::pre_asap::{ColumnRef, DataType, QueryExpr, Reduction, Source}; - use std::collections::BTreeMap; - - #[test] - fn every_physical_payload_can_be_assigned_either_phase() { - use crate::post_asap::DataPrimitive; - use crate::pre_asap::{ArithmeticOpKind, BinaryOpKind, JoinKind, Predicate, ScalarValue}; - let family = FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let predicate = Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))); - let payloads = vec![ - PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Literal(ScalarValue::Int64(1)), - }, - PostAsapOperatorPayload::Binary { - operator: BinaryOperator { - checked_relative_division: false, - checked_finite_division: false, - kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), - vector_match: None, - }, - }, - PostAsapOperatorPayload::Value { - operation: ValueOperation::Limit { - n: 1, - offset: 0, - partition_by: Default::default(), - }, - }, - PostAsapOperatorPayload::RelationalJoin { - join_kind: JoinKind::Semi, - pred: predicate, - pruning: None, - }, - PostAsapOperatorPayload::SummaryAgg { - family: family.clone(), - input: SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - PostAsapOperatorPayload::SummaryJoin { - key: ColumnRef::SampleValue, - family: family.clone(), - }, - PostAsapOperatorPayload::SummarySubtract, - PostAsapOperatorPayload::SummaryDelete { - key: ColumnRef::SampleValue, - }, - PostAsapOperatorPayload::SummaryEstimate { - query: SketchStatistic::Cardinality, - }, - PostAsapOperatorPayload::SummaryMerge, - ]; - for payload in payloads { - // This checks physical identity and placement, not kernel availability. - let primitive = match &payload { - PostAsapOperatorPayload::Fallback { .. } - | PostAsapOperatorPayload::Binary { .. } - | PostAsapOperatorPayload::Value { .. } - | PostAsapOperatorPayload::RelationalJoin { .. } - | PostAsapOperatorPayload::SummaryEstimate { .. } => DataPrimitive::Raw, - PostAsapOperatorPayload::SummaryAgg { .. } - | PostAsapOperatorPayload::SummaryJoin { .. } - | PostAsapOperatorPayload::SummarySubtract - | PostAsapOperatorPayload::SummaryDelete { .. } - | PostAsapOperatorPayload::SummaryMerge => DataPrimitive::SummaryState, - }; - let dag = PostAsapDAG { - root: PostAsapNodeId(0), - edges: vec![], - nodes: vec![PostAsapDAGNode { - id: PostAsapNodeId(0), - payload: payload.clone(), - output_state: ExecutionDataState { - timing: ExecutionTiming::QueryTime, - primitive, - }, - output_schema: Schema::lifted( - vec![Field { - name: "value".into(), - dtype: family.clone(), - nullable: false, - table: None, - }], - None, - ), - guarantee: None, - }], - }; - for phase in [ExecutionTiming::IngestionTime, ExecutionTiming::QueryTime] { - let placed = dag - .with_execution_phases(&BTreeMap::from([(dag.root, phase)])) - .unwrap(); - assert_eq!(placed.nodes[0].payload, payload); - assert_eq!(placed.nodes[0].output_state.timing, phase); - let wire = serde_json::to_value(&placed).unwrap(); - assert!(wire["nodes"][0]["payload"].get("timing").is_none()); - assert_eq!(serde_json::from_value::(wire).unwrap(), placed); - } - assert!(dag.with_execution_phases(&BTreeMap::new()).is_err()); - } - } - - #[test] - fn phase_assignment_updates_edges_and_rejects_query_dependencies_in_ingestion() { - use crate::pre_asap::ScalarValue; - let schema = Schema::lifted(vec![], None); - let nodes = [0, 1] - .into_iter() - .map(|id| PostAsapDAGNode { - id: PostAsapNodeId(id), - payload: PostAsapOperatorPayload::Fallback { - expression: QueryExpr::Literal(ScalarValue::Int64(1)), - }, - output_state: ExecutionDataState::QUERY_ROWS, - output_schema: schema.clone(), - guarantee: None, - }) - .collect(); - let dag = PostAsapDAG { - nodes, - root: PostAsapNodeId(1), - edges: vec![PostAsapDAGEdge { - producer: PostAsapNodeId(0), - consumer: PostAsapNodeId(1), - role: EdgeRole::Input, - intermediate_schema: schema, - data_state: ExecutionDataState::QUERY_ROWS, - grouping: GroupingEdgeCompatibility::NotApplicable, - window: WindowEdgeCompatibility::NotApplicable, - }], - }; - let placed = dag - .with_execution_phases(&BTreeMap::from([ - (PostAsapNodeId(0), ExecutionTiming::IngestionTime), - (PostAsapNodeId(1), ExecutionTiming::QueryTime), - ])) - .unwrap(); - assert_eq!( - placed.edges[0].data_state.timing, - ExecutionTiming::IngestionTime - ); - assert_eq!(dag.edges[0].data_state.timing, ExecutionTiming::QueryTime); - assert!(matches!( - dag.with_execution_phases(&BTreeMap::from([ - (PostAsapNodeId(0), ExecutionTiming::QueryTime), - (PostAsapNodeId(1), ExecutionTiming::IngestionTime), - ])), - Err(PostAsapDAGValidationError::QueryDependencyInIngestion { .. }) - )); - } - - #[test] - fn exports_summary_over_summary_as_typed_precompute_edges() { - let scan = Rc::new(QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), - }); - let raw = Rc::new(SummaryNode { - expr: SummaryExpr::KeepPreAsap(scan), - schema: Schema::lifted( - vec![Field { - name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), - nullable: false, - table: None, - }], - None, - ), - guarantee: None, - }); - let make_agg = |child: Rc, kind, params| { - let family = FieldDataType::ExactAggregate(kind, params); - Rc::new(SummaryNode { - expr: SummaryExpr::SummaryAgg { - child, - family: family.clone(), - input: SummaryUpdate::column(ColumnRef::SampleValue), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - }, - schema: Schema::lifted( - vec![Field { - name: "value".into(), - dtype: family, - nullable: false, - table: None, - }], - None, - ), - guarantee: None, - }) - }; - let inner = make_agg(raw, ExactKind::Sum, ExactParams::Sum); - let outer = make_agg(Rc::clone(&inner), ExactKind::Sum, ExactParams::Sum); - let root = Rc::new(SummaryNode { - expr: SummaryExpr::ValueOperation { - child: outer, - operation: ValueOperation::FinalizeExactAccumulator, - timing: ExecutionTiming::QueryTime, - }, - schema: Schema::lifted( - vec![Field { - name: "value".into(), - dtype: FieldDataType::Plain(DataType::Float64), - nullable: false, - table: None, - }], - None, - ), - guarantee: None, - }); - - let compiled = compile_post_asap_dag_with_node_ids(&root).unwrap(); - assert_eq!(compiled.node_ids.node_id(&root), Some(PostAsapNodeId(3))); - assert!(Rc::ptr_eq( - compiled.node_ids.summary_node(PostAsapNodeId(1)).unwrap(), - &inner - )); - let dag = compiled.dag; - assert_eq!(dag.root, PostAsapNodeId(3)); - assert_eq!( - dag.nodes[1].output_state, - ExecutionDataState::INGESTION_SUMMARY - ); - assert_eq!( - dag.nodes[2].output_state, - ExecutionDataState::INGESTION_SUMMARY - ); - let dependency = dag - .edges - .iter() - .find(|e| e.producer == PostAsapNodeId(1) && e.consumer == PostAsapNodeId(2)) - .unwrap(); - assert_eq!(dependency.data_state, ExecutionDataState::INGESTION_SUMMARY); - assert_eq!(dependency.grouping, GroupingEdgeCompatibility::Identical); - assert_eq!( - dependency.window, - WindowEdgeCompatibility::RequiresAlignedPanePhaseOrExactWindowEdgeResidual - ); - assert!(matches!( - dependency.intermediate_schema.fields[0].dtype, - FieldDataType::ExactAggregate(ExactKind::Sum, _) - )); - let encoded = serde_json::to_string(&dag).expect("serialize post-ASAP DAG"); - let decoded: PostAsapDAG = - serde_json::from_str(&encoded).expect("deserialize post-ASAP DAG"); - assert_eq!(decoded, dag); - let document = PostAsapDAGDocument::new(decoded); - document.validate().unwrap(); - let mut invalid = serde_json::to_value(&document).unwrap(); - invalid["dag"]["nodes"][0]["operator"] = serde_json::json!("Binary"); - assert!(serde_json::from_value::(invalid).is_err()); - assert!(document.dag.nodes.iter().all(|node| { - let wire = serde_json::to_value(node).unwrap(); - wire.get("operator").is_none() && wire["payload"]["kind"].is_string() - })); - let mut old_version = document.clone(); - old_version.schema_version = 1; - assert_eq!( - old_version.validate(), - Err(PostAsapDAGValidationError::UnsupportedVersion(1)) - ); - let mut unknown = serde_json::to_value(&document).unwrap(); - unknown["unexpected"] = serde_json::json!(true); - assert!(serde_json::from_value::(unknown).is_err()); - assert!(matches!( - dag.nodes[2].payload, - PostAsapOperatorPayload::SummaryAgg { - family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), - reduction: Reduction::Reduce(_), - .. - } - )); - } - - #[test] - fn post_asap_node_ids_serialize_in_deterministic_binding_order() { - let mut bindings = BTreeMap::new(); - bindings.insert(PostAsapNodeId(10), "materialization-10"); - bindings.insert(PostAsapNodeId(2), "query-2"); - assert_eq!( - serde_json::to_string(&bindings).unwrap(), - r#"{"2":"query-2","10":"materialization-10"}"# - ); - } -} diff --git a/crates/types/src/post_asap/query_time/error_estimation.rs b/crates/types/src/post_asap/query_time/error_estimation.rs deleted file mode 100644 index 663d2da12..000000000 --- a/crates/types/src/post_asap/query_time/error_estimation.rs +++ /dev/null @@ -1,643 +0,0 @@ -//! Posterior (query-time) sketch error estimation — Chen, Wu, Yang, Jiang, -//! Liu, ["Precise Error Estimation for Sketch-based Flow -//! Measurement"](https://zaoxing.github.io/papers/2021/IMC21_ErrorEstimation.pdf) -//! (IMC '21). Tracks issue #239. -//! -//! ## What this is -//! -//! The traditional CMS/Count-Sketch/CU-Sketch guarantee is an *a priori*, -//! worst-case bound derived before any data is seen: e.g. classic CMS sizing -//! (`w = ⌈e/ε⌉` counters/row, `r = ⌈ln(1/δ)⌉` rows — exactly -//! `crates/asap-aware-mapping/src/implementation.rs`'s `cms_width`/`cms_depth`) -//! guarantees `Pr[error > ε·|F|₁] < δ` obliviously to the real data -//! distribution, by construction assuming the adversarial worst case (§3, -//! §3.3 of the paper). -//! -//! The paper's insight: once a sketch has actually ingested data, its *real* -//! counter values already encode a tighter, still-rigorous bound — no need -//! to fall back to the worst case. [`cms_posterior_error_bound`] implements -//! that estimator (§3.1's Algorithm 1): given one sketch row's `w` counter -//! values and the target confidence `1−δ`, it returns the -//! `⌊w·δ^(1/r)⌋`-th largest counter in that row as the error bound, valid -//! with confidence `1−δ` for *any* flow's estimate (not just the one -//! queried) — proved in §3.2 (Theorems 3.1/3.2/3.4) to closely approximate -//! the true ground-truth error bound (bias `O(1/√w)`, Eq. 5), and in §3.3 -//! (Eq. 6) to always be at least as tight as the traditional a priori bound -//! for the same `(w, r)`. [`traditional_a_priori_bound`] and -//! [`classic_cms_sizing`] implement that traditional bound/sizing for -//! comparison — see [`cms_posterior_bound_is_at_most_traditional_bound`] -//! below for the checked property. -//! -//! Appendix A.1 generalizes the technique: [`cu_sketch_posterior_error_bound`] -//! (Algorithm 2) is — per the paper's own text — *identical* to -//! [`cms_posterior_error_bound`], since CU-Sketch shares CM-Sketch's -//! "minimum counter across rows" query paradigm. -//! [`count_sketch_posterior_error_bound`] (Algorithm 3) is Count-Sketch's -//! variant: because Count-Sketch estimates a flow's size as the *median* of -//! its `r` signed counters rather than the minimum, the estimator instead -//! searches for the smallest fractile `p₀` whose two-sided binomial-tail -//! probability clears `δ` (Theorem A.1, `r = 2k+1` odd only — see that -//! function's docs for why even `r` is out of scope here). -//! -//! ## What this is *not* — no runtime sketch exists yet to wire this into -//! -//! This issue names two possible integration points: (1) runtime/readout-time -//! accuracy reporting from a sketch's *actual* counters, and (2) tighter -//! plan-time sizing. As of this module landing, **this repository has no -//! vendored CMS/CountSketch/CU-Sketch runtime and no counter-array data -//! structure anywhere** — confirmed by inspecting -//! `crates/types/src/post_asap/sketch.rs` (`SketchAlgorithm`, `SketchParams`, -//! this module's neighbors) and `crates/asap-aware-mapping/src/implementation.rs` -//! (`default_size_params`, `cms_width`, `cms_depth`): both are purely -//! planning-time sizing metadata. There is no `A[row][col]` counter matrix -//! anywhere in the workspace for these functions to be handed at query -//! time. So integration point (1) — reporting an *actual* query's posterior -//! error from real counters at readout — has nothing to wire into today. -//! -//! The functions here are deliberately **sketch-object-agnostic**: they take -//! plain counter slices (`&[u64]` / `&[i64]`) and numeric parameters, not a -//! concrete sketch type, specifically so that the moment a real CMS/ -//! Count-Sketch/CU-Sketch runtime lands in this workspace, its readout path -//! can call these functions directly on its real counter arrays with zero -//! changes needed here. That wiring is out of scope for this module — see -//! issue #239. -//! -//! Integration point (2) — tighter *plan-time* sizing under an expected-case -//! (non-adversarial) assumption — is wired for real, since it touches code -//! that already exists: see -//! `crates/asap-aware-mapping/src/implementation.rs::posterior_aware_size_params`. - -// ── Posterior (query-time) estimators ─────────────────────────────────────── - -/// Count-Min Sketch posterior error estimator — §3.1, Algorithm 1. -/// -/// `row` is one sketch row's `w` raw counter values (any of the sketch's `r` -/// rows — the algorithm is defined per-row and every row gives an -/// independent, equally valid estimate). `rows` is the sketch's total row -/// count `r` (used only to convert the target confidence into the -/// per-row fractile `p = δ^(1/r)`, per the union-bound argument in §3.1: -/// "the probability δ that all r corresponding counters are not bounded by -/// g(δ) is p^r"). `delta` is the target failure probability `δ` (so the -/// returned bound holds with confidence `1−δ`). -/// -/// Returns the `⌊w·p⌋`-th largest value in `row` (1-indexed, clamped to -/// `[1, w]`), matching the paper's own descriptive prose in §3.1 ("report -/// the ⌊wp⌋-th largest counter as our estimation for g(δ)"). Note: the -/// paper's Algorithm 1 pseudocode box instead prints `⌈wp⌉` (ceiling) — the -/// paper is internally inconsistent between its prose and its pseudocode -/// box on this rounding direction. This implementation follows the prose -/// (and issue #239's own paraphrase of it); the two conventions differ by -/// at most one rank, well within the paper's own `O(1/√w)` bias bound -/// (Eq. 5), so the choice does not affect any of this module's correctness -/// properties. -/// -/// Returns `None` for degenerate inputs: an empty row, zero rows, or `delta` -/// outside `(0, 1]`. -pub fn cms_posterior_error_bound(row: &[u64], rows: u32, delta: f64) -> Option { - let idx = posterior_rank(row.len(), rows, delta)?; - Some(kth_largest(row, idx)) -} - -/// CU-Sketch posterior error estimator — Appendix A.1, Algorithm 2. -/// -/// Per the paper: "the algorithm for CU-Sketch is exactly the same as that -/// for CM-Sketch due to their similar properties in generating sketch: they -/// share a common query paradigm that returns the minimum counter value -/// among the `r` rows as the estimated flow size." This is a distinctly -/// named entry point (matching the paper's own Algorithm 2 naming) that -/// delegates to [`cms_posterior_error_bound`] rather than duplicating its -/// logic — see that function's docs for the full parameter/behavior -/// contract, which applies verbatim here. -pub fn cu_sketch_posterior_error_bound(row: &[u64], rows: u32, delta: f64) -> Option { - cms_posterior_error_bound(row, rows, delta) -} - -/// Count-Sketch posterior error estimator — Appendix A.1, Algorithm 3. -/// -/// Count-Sketch counters are signed (each row's hash also picks a random -/// sign, so a flow's contribution can subtract as well as add — "balanced/ -/// zero-mean-error" per `SketchAlgorithm::CountSketch`'s own doc), and a flow's -/// size is estimated as the *median*, not the minimum, of its `r` per-row -/// counters. That breaks Algorithm 1/2's simple `p = δ^(1/r)` derivation (a -/// union bound over "any row is bad"): the median needs *more than half* the -/// rows to be bad, so Algorithm 3 instead searches for the smallest -/// fractile `p₀ ∈ {2/w, 3/w, …}` whose two-sided binomial-tail probability — -/// the chance that more than half of `r` independent `Bernoulli(p₀/2)` -/// trials succeed — first reaches `δ`, then reports the `⌈w·p₀⌉`-th largest -/// *absolute* counter value at that `p₀` (`row` here is `A[1][1..w]`, -/// unsigned by absolute value, matching Algorithm 3's -/// `SortToDescendingOrder(|A[1][1]| … |A[1][w]|)`). -/// -/// Theorem A.1 states the optimal bound only for `r = 2k+1` (odd); the -/// paper's own footnote for the even case ("Similar equation for an even -/// r") does not spell out the formula. Per issue #239's instruction to -/// "only implement what you can read and verify" rather than guess, this -/// function requires odd `rows` and returns `None` for even `rows` — a -/// documented scope boundary, not an oversight. -/// -/// Returns `None` for: an empty `row`, `rows == 0`, even `rows`, or `delta` -/// outside `(0, 1]`. -pub fn count_sketch_posterior_error_bound(row: &[i64], rows: u32, delta: f64) -> Option { - let w = row.len(); - if w == 0 || rows == 0 || rows.is_multiple_of(2) || !(delta > 0.0 && delta <= 1.0) { - return None; - } - let r = rows as u64; - // Majority threshold: for r = 2k+1, this is k+1 — the smallest j for - // which "j of r rows" is a strict majority. Also correct (by the same - // "more than half" reading) for the even-r case this function declines - // to handle, but we never reach here with even r. - let j_start = r.div_ceil(2); - - // `C(r, j)` for every `j` in `j_start..=r`, computed once via the - // incremental recurrence below (O(r) total) instead of letting - // `binomial_tail` recompute `binomial_coeff` (itself O(r)) from - // scratch for every `j` on every fractile tried — the fractile search - // just below evaluates the tail up to `log2(w)` times, so hoisting - // this out turns each evaluation's cost from O(r²) into O(r). - let coeffs = binomial_coeffs_from(r, j_start); - - // `2 * binomial_tail(p0/2)` is monotonically non-decreasing in `p0` - // (raising each row's bad-probability can only raise the chance that a - // majority of rows are bad), so the smallest qualifying fractile - // `p0 = (i+1)/w` is a binary search over `i`, not a linear scan — - // O(log w) tail evaluations instead of O(w). - let satisfies = |i: usize| -> bool { - let p0 = ((i + 1) as f64 / w as f64).min(1.0); - 2.0 * binomial_tail_with_coeffs(&coeffs, r, j_start, p0 / 2.0) >= delta - }; - let mut lo = 1usize; - let mut hi = w; - while lo < hi { - let mid = lo + (hi - lo) / 2; - if satisfies(mid) { - hi = mid; - } else { - lo = mid + 1; - } - } - // `lo == hi`: either `satisfies(lo)`, or `lo == w` and the search - // never found one — matching the original linear scan's fallback, - // which left `chosen_p0` at its pre-loop-initialized `1.0` in that - // case (the same value `p0(w)` evaluates to below). - let chosen_p0 = ((lo + 1) as f64 / w as f64).min(1.0); - - let idx = ((w as f64) * chosen_p0).ceil() as usize; - let idx = idx.clamp(1, w); - let abs_row: Vec = row.iter().map(|v| v.unsigned_abs()).collect(); - Some(kth_largest(&abs_row, idx)) -} - -// ── Traditional a priori bound (§3.3), for comparison ────────────────────── - -/// `⌈x⌉` clamped to `[lo, hi]`; NaN / non-positive `x` saturate to `hi` (a -/// degenerate accuracy target means "as accurate as this family goes"). -/// Byte-for-byte the same policy as `implementation.rs`'s private -/// `saturating_ceil` — duplicated here, not imported, for the same -/// layering reason [`classic_cms_sizing`] itself is duplicated rather than -/// calling `implementation.rs` directly. -fn saturating_ceil(x: f64, lo: u32, hi: u32) -> u32 { - if !x.is_finite() || x <= 0.0 { - return hi; - } - (x.ceil() as u32).clamp(lo, hi) -} - -/// The classic CMS `(ε, δ)` sizing formula (§3.3, restated just before -/// Eq. 6; identical — formula *and* clamping — to -/// `crates/asap-aware-mapping/src/implementation.rs`'s private -/// `cms_width`/`cms_depth`, reimplemented here — deliberately, not by -/// accident — because `asap-types` sits below `asap-aware-mapping` in the -/// workspace's dependency layering and cannot import from it): `w = ⌈e/ε⌉` -/// counters/row, clamped to `[2, 2²⁶]`; `r = ⌈ln(1/δ)⌉` rows, clamped to -/// `[1, 32]`. -/// -/// Returns `(width, depth)`. Matching `implementation.rs`'s own clamp semantics -/// exactly (not just its in-range formula) matters here specifically: -/// this function's whole purpose is giving comparison/test code (and, -/// per issue #250, a future replan) the traditional bound to compare -/// this module's posterior bound against — an un-clamped reimplementation -/// would silently diverge from `implementation.rs`'s real sizing outside a -/// narrow "nothing saturates" range of `(eps, delta)`, exactly the range -/// most tests default to, making the divergence easy to miss. -pub fn classic_cms_sizing(eps: f64, delta: f64) -> (u32, u32) { - let width = saturating_ceil(std::f64::consts::E / eps, 2, 1 << 26); - let depth = saturating_ceil((1.0 / delta).ln(), 1, 32); - (width, depth) -} - -/// The traditional a priori bound value itself: `ε·|F|₁`, the classic CMS -/// guarantee's right-hand side (§3.3: "the original CM bound … guarantees -/// `Pr[𝕏ₑᵢ > ε|F|₁] < δ`"). Exists so a posterior bound (in the same -/// counter units as `total_flow_size`) can be compared directly against the -/// traditional worst-case bound it is meant to improve on — see -/// [`cms_posterior_bound_is_at_most_traditional_bound`]'s test below. -pub fn traditional_a_priori_bound(total_flow_size: u64, eps: f64) -> f64 { - eps * total_flow_size as f64 -} - -// ── Shared internals ───────────────────────────────────────────────────────── - -/// `⌊w·δ^(1/r)⌋`, clamped to `[1, w]` (1-indexed rank into a -/// descending-sorted row of `w` counters). `None` for degenerate inputs. -fn posterior_rank(w: usize, rows: u32, delta: f64) -> Option { - if w == 0 || rows == 0 || !(delta > 0.0 && delta <= 1.0) { - return None; - } - let p = delta.powf(1.0 / rows as f64); - let idx = ((w as f64) * p).floor() as i64; - Some((idx.max(1) as usize).min(w)) -} - -/// The `k`-th largest value in `values` (1-indexed): sorts a copy in -/// descending order and returns `sorted[k-1]`. `k` is expected already -/// clamped to `[1, values.len()]` by the caller. -/// The `k`-th largest value in `values` (1-indexed: `k=1` is the max). -/// `select_nth_unstable_by` partitions in O(w) average instead of fully -/// sorting in O(w log w) — this only ever needs one rank, not a total -/// order, and both call sites (this module's per-query readout math) are -/// documented as meant to run on a future runtime's hot readout path. -fn kth_largest(values: &[u64], k: usize) -> u64 { - let mut buf: Vec = values.to_vec(); - let idx = k - 1; - let (_, &mut kth, _) = buf.select_nth_unstable_by(idx, |a, b| b.cmp(a)); - kth -} - -/// `Σ_{j=j_start}^{r} C(r, j) · p^j · (1-p)^(r-j)` — the upper binomial tail -/// probability, given `coeffs[i] = C(r, j_start + i)` already computed by -/// [`binomial_coeffs_from`]. Split out from the coefficient computation so -/// a caller trying several `p` values against the same `(r, j_start)` (as -/// [`count_sketch_posterior_error_bound`]'s fractile search does) pays for -/// the coefficients once, not once per `p`. -fn binomial_tail_with_coeffs(coeffs: &[f64], r: u64, j_start: u64, p: f64) -> f64 { - let p = p.clamp(0.0, 1.0); - coeffs - .iter() - .enumerate() - .map(|(offset, &c)| { - let j = j_start + offset as u64; - c * p.powi(j as i32) * (1.0 - p).powi((r - j) as i32) - }) - .sum() -} - -/// `C(r, j)` for every `j` in `j_start..=r`, in one O(r) pass via the -/// incremental recurrence `C(r,j) = C(r,j-1) · (r-j+1)/j` — instead of -/// calling [`binomial_coeff`] (itself O(r)) fresh for every `j`, which is -/// what made evaluating a single tail probability O(r²). -fn binomial_coeffs_from(r: u64, j_start: u64) -> Vec { - let mut coeffs = Vec::with_capacity((r - j_start + 1) as usize); - let mut c = binomial_coeff(r, j_start); - coeffs.push(c); - for j in (j_start + 1)..=r { - c = c * (r - j + 1) as f64 / j as f64; - coeffs.push(c); - } - coeffs -} - -/// `C(n, k)`, computed as an iterative running product in `f64` (avoids -/// factorial overflow; exact for the small `n` realistic sketch depths use, -/// and only ever used as a probability-mass weight so `f64` rounding is -/// immaterial). -fn binomial_coeff(n: u64, k: u64) -> f64 { - let k = k.min(n - k); - let mut result = 1.0_f64; - for i in 0..k { - result = result * (n - i) as f64 / (i + 1) as f64; - } - result -} - -#[cfg(test)] -mod tests { - use super::*; - - // ── Algorithm 1 (CM-Sketch) — hand-computable examples ────────────────── - - #[test] - fn cms_single_row_confidence_one_picks_the_largest_counter() { - // rows=1, delta=1 ⇒ p = 1^(1/1) = 1 ⇒ index = floor(w*1) = w ⇒ the - // w-th largest = the *smallest* counter (delta=1 means "no - // confidence required", so the loosest possible bound is fine — - // this pins the boundary behavior, not a meaningful confidence). - let row = [10u64, 4, 7, 1]; - assert_eq!(cms_posterior_error_bound(&row, 1, 1.0), Some(1)); - } - - #[test] - fn cms_hand_computed_example() { - // r=2, delta=0.25 ⇒ p = 0.25^(1/2) = 0.5. w=10 ⇒ index = floor(10*0.5) = 5. - // Descending-sorted row: [10,9,8,7,6,5,4,3,2,1] ⇒ 5th largest = 6. - let row: Vec = (1..=10).collect(); // [1..10] - assert_eq!(cms_posterior_error_bound(&row, 2, 0.25), Some(6)); - } - - #[test] - fn cms_rank_clamped_to_at_least_one() { - // Very small p (large r, small delta) floors to 0 — must clamp to - // rank 1 (the single largest counter), not panic / index -1. - let row = [5u64, 100, 1]; - let bound = cms_posterior_error_bound(&row, 50, 1e-9).unwrap(); - assert_eq!(bound, 100); // the largest counter - } - - #[test] - fn cms_all_zero_counters_bound_is_zero() { - let row = [0u64; 8]; - assert_eq!(cms_posterior_error_bound(&row, 3, 0.05), Some(0)); - } - - #[test] - fn cms_degenerate_inputs_return_none() { - assert_eq!(cms_posterior_error_bound(&[], 3, 0.05), None); - assert_eq!(cms_posterior_error_bound(&[1, 2, 3], 0, 0.05), None); - assert_eq!(cms_posterior_error_bound(&[1, 2, 3], 3, 0.0), None); - assert_eq!(cms_posterior_error_bound(&[1, 2, 3], 3, -0.1), None); - assert_eq!(cms_posterior_error_bound(&[1, 2, 3], 3, 1.5), None); - } - - #[test] - fn cms_monotonic_in_width_tighter_bound_as_w_grows() { - // Hold the *total* collided mass fixed and spread it over more - // counters: a wider sketch means less collision per counter for - // the same workload, so the reported bound should not increase. - // (Using row = 1..=w instead — growing *with* w — would grow the - // total mass too, which is a different, unrelated effect; the - // fixed-total-uniform-spread row below isolates width alone.) - let total = 20_160u64; // divisible by every width below - let mut previous = u64::MAX; - for w in [10usize, 20, 40, 80, 160] { - let row = vec![total / w as u64; w]; - let bound = cms_posterior_error_bound(&row, 4, 0.1).unwrap(); - assert!( - bound <= previous, - "bound grew from {previous} to {bound} as w increased to {w}" - ); - previous = bound; - } - } - - #[test] - fn cms_monotonic_in_rows_tighter_bound_as_r_grows() { - // Algorithm 1 computes its bound from *one* row's counters; `rows` - // (r) only feeds into p = delta^(1/r). For delta < 1, delta^(1/r) - // increases toward 1 as r grows (more rows ⇒ the union bound over - // "any row is bad" needs a looser per-row p to keep the same - // overall delta), which raises the rank index ⌊w·p⌋ — i.e. r is a - // confidence dial that monotonically raises the rank (and, on a - // descending-sorted row, that means an equal-or-tighter reported - // bound). We assert the rank itself grows monotonically with r, - // which is the exact, documented relationship. - let row: Vec = (1..=100).rev().collect(); - let mut previous_idx = 0usize; - for r in [1u32, 2, 4, 8, 16, 32] { - let idx = posterior_rank(row.len(), r, 0.1).unwrap(); - assert!( - idx >= previous_idx, - "rank shrank from {previous_idx} to {idx} as r grew to {r}" - ); - previous_idx = idx; - } - } - - // ── Algorithm 2 (CU-Sketch) — identical to Algorithm 1 ───────────────── - - #[test] - fn cu_sketch_matches_cms_exactly() { - let row: Vec = (1..=50).collect(); - for (r, delta) in [(1u32, 0.5), (3, 0.1), (7, 0.01)] { - assert_eq!( - cu_sketch_posterior_error_bound(&row, r, delta), - cms_posterior_error_bound(&row, r, delta) - ); - } - } - - // ── Algorithm 3 (Count-Sketch) ─────────────────────────────────────────── - - #[test] - fn count_sketch_rejects_even_rows() { - let row = [3i64, -5, 2, 8]; - assert_eq!(count_sketch_posterior_error_bound(&row, 4, 0.05), None); - } - - #[test] - fn count_sketch_uses_absolute_values() { - // Symmetric-magnitude row: signs shouldn't change the bound. - let row_pos: Vec = (1..=21).collect(); - let row_neg: Vec = (1..=21).map(|v| -v).collect(); - let a = count_sketch_posterior_error_bound(&row_pos, 5, 0.05); - let b = count_sketch_posterior_error_bound(&row_neg, 5, 0.05); - assert!(a.is_some()); - assert_eq!(a, b); - } - - #[test] - fn count_sketch_degenerate_inputs_return_none() { - assert_eq!(count_sketch_posterior_error_bound(&[], 3, 0.05), None); - assert_eq!( - count_sketch_posterior_error_bound(&[1, 2, 3], 0, 0.05), - None - ); - assert_eq!(count_sketch_posterior_error_bound(&[1, 2, 3], 3, 0.0), None); - assert_eq!(count_sketch_posterior_error_bound(&[1, 2, 3], 3, 1.5), None); - } - - #[test] - fn count_sketch_single_row_extreme_confidence() { - // r=1 (odd), delta=1: the loosest possible ask — should resolve to - // *some* valid rank in range without panicking. - let row: Vec = vec![9, -4, 7, -1, 3]; - let bound = count_sketch_posterior_error_bound(&row, 1, 1.0).unwrap(); - assert!(row.iter().any(|v| v.unsigned_abs() == bound)); - } - - #[test] - fn count_sketch_monotonic_in_width() { - // Same fixed-total-uniform-spread construction as the CMS - // monotonicity test above — isolates width's effect from total - // mass's. - let total = 20_160i64; - let mut previous = u64::MAX; - for w in [10usize, 20, 40, 80, 160] { - let row = vec![total / w as i64; w]; - let bound = count_sketch_posterior_error_bound(&row, 5, 0.1).unwrap(); - assert!( - bound <= previous, - "bound grew from {previous} to {bound} as w increased to {w}" - ); - previous = bound; - } - } - - // ── Traditional bound / classic sizing ────────────────────────────────── - - #[test] - fn classic_cms_sizing_matches_implementation_rs_formula() { - // Same worked example as - // `asap-aware-mapping::replacement::tests::epsilon_delta_sizes_cms_depth` - // (eps=0.001, delta=0.001 ⇒ width=2719, depth=7), pinned here too so - // the two independent (layering-forced) reimplementations can't - // silently drift apart undetected. - assert_eq!(classic_cms_sizing(0.001, 0.001), (2719, 7)); - assert_eq!(classic_cms_sizing(0.01, 0.01), (272, 5)); - } - - #[test] - fn classic_cms_sizing_degenerate_inputs_saturate_like_implementation_rs() { - // Degenerate width/depth saturate to their hi clamp (2^26 / 32), - // matching `implementation.rs`'s `saturating_ceil` exactly — not `0` (see - // the correctness fix on `classic_cms_sizing`'s doc comment: an - // earlier version returned `(0, 0)` here, silently diverging from - // `implementation.rs`'s real degenerate-input behavior). - assert_eq!(classic_cms_sizing(0.0, 0.01), (1 << 26, 5)); - assert_eq!(classic_cms_sizing(0.01, 0.0), (272, 32)); - assert_eq!(classic_cms_sizing(0.01, 1.0), (272, 32)); - assert_eq!(classic_cms_sizing(f64::NAN, 0.01), (1 << 26, 5)); - } - - /// Pinned extreme-range values — the clamp actually engaging, not just - /// the in-range formula — computed by hand against the same - /// `[2, 2²⁶]` / `[1, 32]` bounds `implementation.rs`'s `cms_width`/`cms_depth` - /// use, so a future edit that reintroduces the un-clamped bug (an - /// earlier version of this function silently diverged from - /// `implementation.rs` outside the narrow range most other tests exercise) - /// gets caught here. - #[test] - fn classic_cms_sizing_clamps_extreme_ranges_like_implementation_rs() { - // eps=1e-10: raw width e/eps ≈ 2.7e10, hi-clamped to 2^26. - assert_eq!(classic_cms_sizing(1e-10, 0.01), (1 << 26, 5)); - // eps=3.0: raw width e/3 < 1, lo-clamped to 2. - assert_eq!(classic_cms_sizing(3.0, 0.01), (2, 5)); - // delta=1e-20: raw depth ln(1e20) ≈ 46.05 → ⌈⌉ 47, hi-clamped to 32. - assert_eq!(classic_cms_sizing(0.01, 1e-20), (272, 32)); - } - - #[test] - fn traditional_bound_is_linear_in_total_and_eps() { - assert_eq!(traditional_a_priori_bound(1_000_000, 0.01), 10_000.0); - assert_eq!(traditional_a_priori_bound(0, 0.01), 0.0); - } - - // ── The correctness property this issue is actually about: the ──────── - // ── posterior bound (Algorithm 1) is always ≤ the traditional a ─────── - // ── priori bound (Eq. 6), for the same worst-case (w, r) sizing. ────── - - /// A tiny deterministic PRNG (xorshift64*) — enough to generate varied - /// synthetic counter distributions without adding a `rand` dependency - /// to this crate. - struct Xorshift64(u64); - impl Xorshift64 { - fn next(&mut self) -> u64 { - let mut x = self.0; - x ^= x << 13; - x ^= x >> 7; - x ^= x << 17; - self.0 = x; - x - } - /// A weight in `[1, max]`, skewed low (many small draws, occasional - /// large one) by squaring a uniform fraction — cheap stand-in for a - /// heavy-tailed/Zipf-like distribution shape. - fn skewed_weight(&mut self, max: u64) -> u64 { - let u = (self.next() % 1_000_000) as f64 / 1_000_000.0; - let skewed = u * u; // biases toward 0 - 1 + (skewed * max as f64) as u64 - } - } - - /// Distributes `total` across `w` non-negative counters honoring a CMS - /// row's real structure: each unit of `total` lands in exactly one of - /// the `w` counters (a counter is a sum of colliding flow weights), so - /// `Σ counters == total` always — the property Eq. 6's proof relies on. - fn synthetic_row(rng: &mut Xorshift64, w: usize, total: u64, skewed: bool) -> Vec { - let mut counters = vec![0u64; w]; - let mut remaining = total; - while remaining > 0 { - let bucket = (rng.next() as usize) % w; - let chunk = if skewed { - rng.skewed_weight(remaining.min(1000)) - } else { - 1 + rng.next() % remaining.clamp(1, 37) - } - .min(remaining); - counters[bucket] += chunk; - remaining -= chunk; - } - counters - } - - #[test] - fn cms_posterior_bound_is_at_most_traditional_bound() { - // The paper's Eq. 6 claim: at the standard worst-case sizing - // (w=classic width, r=classic depth for a target (eps,delta)), the - // posterior bound computed from real counters never exceeds the - // traditional a priori bound eps*|F|1 — regardless of how the - // total mass is actually distributed across counters. - let mut rng = Xorshift64(0x243F6A8885A308D3); - for (eps, delta) in [(0.05, 0.05), (0.02, 0.1), (0.1, 0.01), (0.03, 0.2)] { - let (w, r) = classic_cms_sizing(eps, delta); - let traditional = traditional_a_priori_bound(1_000_000, eps); - for skewed in [false, true] { - for trial in 0..20 { - rng.0 = rng.0.wrapping_add(0x9E3779B97F4A7C15).wrapping_add(trial); - let row = synthetic_row(&mut rng, w as usize, 1_000_000, skewed); - let posterior = - cms_posterior_error_bound(&row, r, delta).expect("valid inputs"); - assert!( - (posterior as f64) <= traditional + 1e-6, - "posterior bound {posterior} exceeded traditional bound \ - {traditional} for eps={eps} delta={delta} w={w} r={r} \ - skewed={skewed} trial={trial}" - ); - } - } - } - } - - #[test] - fn cms_posterior_bound_at_most_traditional_bound_worst_case_single_counter() { - // Degenerate worst case: all mass in one counter (maximum possible - // collision) — the pigeonhole argument underlying Eq. 6 still must - // hold: a single counter holding the entire total is itself only - // ever picked as the estimate when the rank lands on it, and - // whenever it isn't, the picked counter is 0 <= traditional bound. - for (eps, delta) in [(0.05, 0.05), (0.01, 0.01)] { - let (w, r) = classic_cms_sizing(eps, delta); - let traditional = traditional_a_priori_bound(1_000_000, eps); - let mut row = vec![0u64; w as usize]; - row[0] = 1_000_000; - let posterior = cms_posterior_error_bound(&row, r, delta).expect("valid inputs"); - assert!( - (posterior as f64) <= traditional + 1e-6, - "posterior bound {posterior} exceeded traditional bound {traditional}" - ); - } - } - - #[test] - fn kth_largest_pigeonhole_property() { - // The elementary fact Eq. 6's proof leans on: for w non-negative - // counters summing to T, the k-th largest is at most T/k. This is - // the general-purpose invariant behind the paper-specific test - // above, checked directly and unconditionally (no sketch sizing - // formula involved). - let mut rng = Xorshift64(0xD1B54A32D192ED03); - for _ in 0..50 { - let w = 5 + (rng.next() % 50) as usize; - let total = 1 + rng.next() % 100_000; - let row = synthetic_row(&mut rng, w, total, true); - let sum: u64 = row.iter().sum(); - assert_eq!(sum, total); - for k in 1..=w { - let kth = kth_largest(&row, k); - assert!( - (kth as f64) * (k as f64) <= (total as f64) + 1e-6, - "k={k}-th largest {kth} violates k*kth <= total ({total})" - ); - } - } - } -} diff --git a/crates/types/src/post_asap/query_time/mod.rs b/crates/types/src/post_asap/query_time/mod.rs deleted file mode 100644 index 44e979117..000000000 --- a/crates/types/src/post_asap/query_time/mod.rs +++ /dev/null @@ -1,43 +0,0 @@ -//! Code meant to run at **query execution time**, not planning time — -//! folder-separated from `post_asap`'s other modules ([`super::expr`], -//! [`super::schema`], [`super::sketch`]) on purpose. -//! -//! `asap-types`' crate doc states the invariant this workspace otherwise -//! holds to without exception: "no execution logic lives in this -//! workspace (issue #190) — a downstream deployment crate is expected to -//! supply that." Everything else under [`super`] is *planning*-time IR — -//! types the planner (`asap-aware-mapping`) constructs and commits to -//! before a query ever runs, describing *what* summary will be built, not -//! *computing over* one that has run. -//! -//! [`error_estimation`] is the one deliberate exception, and this -//! submodule exists so that exception is visible in the directory listing -//! itself, not just in prose: it holds pure, sketch-object-agnostic *math* -//! (given a sketch's real counter values, compute a tighter posterior -//! error bound) that only makes sense to invoke *after* a real sketch has -//! ingested data and is being read out — a downstream runtime's job, not -//! this crate's. Nothing in `asap-types` or `asap-aware-mapping` calls -//! into this module today; it is a library waiting for a runtime that -//! doesn't exist yet in this workspace (see [`error_estimation`]'s own -//! docs for exactly what's blocked and why). -//! -//! Contrast with `asap-aware-mapping::replacement::posterior_aware_size_params` -//! (issue #239, PR #248): that function is real, wired *planning*-time -//! code — it lives in the planning crate, not here, and does not call -//! into this module. It borrows the same underlying intuition (a -//! non-adversarial workload can use a smaller sketch than the worst case) -//! but as an explicit, caller-supplied planning-time assumption, not a -//! query-time measurement — the two are independent by construction. See -//! issue #250 for the still-unimplemented idea of actually connecting -//! them: recording this module's query-time observations over time and -//! feeding them into a future replan. - -pub mod error_estimation; - -// Glob, not a named list: this submodule exists only to hold -// `error_estimation` today, so re-exporting everything it makes public -// keeps this hop in sync automatically — a new `pub fn` there needs no -// matching edit here, only in `post_asap::mod`'s own list below (the -// actual curated short-path public surface, kept explicit like `expr`'s -// and `sketch`'s re-exports). -pub use error_estimation::*; diff --git a/crates/types/src/post_asap/summary_maintenance.rs b/crates/types/src/post_asap/summary_maintenance.rs deleted file mode 100644 index d1e50d7e5..000000000 --- a/crates/types/src/post_asap/summary_maintenance.rs +++ /dev/null @@ -1,38 +0,0 @@ -//! Planner-level construction mode for a materialized summary. -//! -//! A [`super::SummaryNode`] is a logical summary expression and deliberately -//! does not carry this choice: the same candidate may be built directly for -//! one workload or maintained incrementally for another. Planner search -//! attaches the selected mode to its lifecycle guarantee; downstream physical -//! compilation chooses its concrete implementation. - -/// How a summary deployment obtains its state, independent of implementation. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum SummaryMaintenanceMode { - /// Rebuild the summary from its complete input when the deployment needs - /// a value. No update stream is required. - DirectBuild, - /// Create the state once and apply input changes as they arrive. - Incremental, -} - -impl SummaryMaintenanceMode { - pub const fn as_str(self) -> &'static str { - match self { - Self::DirectBuild => "direct_build", - Self::Incremental => "incremental", - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn maintenance_modes_have_stable_export_names() { - assert_eq!(SummaryMaintenanceMode::DirectBuild.as_str(), "direct_build"); - assert_eq!(SummaryMaintenanceMode::Incremental.as_str(), "incremental"); - } -} diff --git a/crates/types/src/post_asap/summary_maintenance_lifecycle.rs b/crates/types/src/post_asap/summary_maintenance_lifecycle.rs deleted file mode 100644 index 798d861f6..000000000 --- a/crates/types/src/post_asap/summary_maintenance_lifecycle.rs +++ /dev/null @@ -1,74 +0,0 @@ -//! Planner-level summary-maintenance lifecycle vocabulary. -//! -//! A **summary-maintenance lifecycle** describes when one materialized summary -//! state is created, retained or shared, updated, and retired. It does not -//! describe the broader data lifecycle (collection, transport, and storage), -//! and it is not implied by a logical `SummaryAgg`. Physical planning compares -//! alternatives using the expected number and timing of reads, the source-data -//! arrival/update rate, state-operation costs, and runtime capabilities. - -use super::SummaryMaintenanceMode; -use crate::workload::{DurationMs, TimestampMs}; - -/// When an operator is evaluated. This is independent of whether it owns -/// state and how long that state is retained. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum EvaluationSchedule { - OneShot, - PerUpdate, - OnRead, -} - -/// The form in which this deployment exposes its result to its consumer. The -/// consumer is the next operator in the execution plan that reads the -/// summary's output; for example, `Estimate` is the consumer in -/// `SummaryAgg -> Estimate`. The exposed result is ordinary rows, reusable -/// summary state, or a finalized value. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum OutputRepresentation { - PlainRows, - SummaryState, - FinalizedValue, -} - -/// Abstract policy for when one materialized summary state is created, -/// retained or shared, updated as data arrives, and retired. -/// -/// This is not the lifecycle of the source data or query. Query recurrence -/// provides the expected number and timing of reads; data arrival provides the -/// expected state-update demand. The planner combines those quantities with -/// costs and runtime capabilities to compare these policies. -#[derive(Debug, Clone, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] -#[serde(tag = "kind", rename_all = "snake_case")] -pub enum SummaryMaintenanceLifecycle { - Ephemeral, - Prepared { - #[serde(rename = "activate_at_ms")] - activate_at: TimestampMs, - #[serde(rename = "retire_at_ms")] - retire_at: TimestampMs, - }, - Shared { - #[serde(rename = "retention_ms")] - retention: DurationMs, - }, - ContinuouslyMaintained, -} - -/// The lifecycle commitment emitted for one materialized summary deployment. -/// -/// This names the summary-maintenance promise explicitly so consumers do not -/// confuse it with guarantees about the broader data lifecycle. Accuracy is a -/// separate [`super::ResultGuarantee`]. -#[derive(Debug, Clone, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] -#[serde(deny_unknown_fields)] -pub struct SummaryMaintenanceLifecycleGuarantee { - #[serde(rename = "lifecycle")] - pub summary_maintenance_lifecycle: SummaryMaintenanceLifecycle, - #[serde(rename = "maintenance_mode")] - pub summary_maintenance_mode: SummaryMaintenanceMode, - pub evaluation_schedule: EvaluationSchedule, - pub output_representation: OutputRepresentation, -} diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs deleted file mode 100644 index b9e6a653a..000000000 --- a/crates/types/src/pre_asap/canonicalize.rs +++ /dev/null @@ -1,782 +0,0 @@ -//! Shared post-lowering canonicalization of the resolved [`QueryExpr`]. -//! -//! Both language front ends funnel through [`resolve_root`](super::resolve::resolve_root), -//! which runs this pass over the resolved DAG. Its job is to erase -//! *structural* differences between semantically identical queries so a -//! post-ASAP binding rule matching on the intent algebra sees one canonical -//! spelling regardless of source language (issue #34). -//! -//! ## Heavy-hitter promotion -//! -//! An additive-ranked "order by the aggregate, take the top k" is a -//! heavy-hitter represented by [`AggIntent::TopK`]. Front ends may -//! emit it as an ordinary `Limit { Sort { … Aggregate } }`; this pass promotes -//! that shape to the canonical -//! -//! ```text -//! Aggregate { reduction: Reduce(), measures: [TopK{k}], -//! child: Aggregate { measures: [Count | Sum], … } } -//! ``` -//! -//! Count supplies unit weights and Sum supplies value weights. Because the -//! match is positional, aliases do not affect it. Other ranked expressions -//! retain Sort + Limit. - -use std::rc::Rc; - -use super::agg_intent::{topk, AggIntent}; -use super::expr_ir::{CompareOpKind, ScalarValue}; -use super::query_expr::{Predicate, QueryExpr, Reduction, SortKey, WindowFuncKind}; -use crate::types::AccuracyTarget; - -/// Rewrite `expr` into its canonical form (bottom-up). Idempotent: a DAG that -/// is already canonical is returned unchanged. -pub fn canonicalize(mut expr: QueryExpr) -> QueryExpr { - canon(&mut expr); - expr -} - -fn canon(expr: &mut QueryExpr) { - // A `Concat` asserting a caller-proven `discriminator_unique_key` (issue - // #228) had that key's `ColumnId`s resolved, in `resolve.rs`, against - // exactly the first branch's output schema *as it stood before this - // pass ran*. `try_promote_additive_top_ranking`/`try_rewrite_rownumber_topk` - // below can restructure that branch (anywhere within it — not only at - // its own top level, since the same recursive walk can rewrite a node - // nested under a pass-through wrapper too) into a shape with a - // different output schema, which would leave those `ColumnId`s - // pointing at the wrong column, or out of bounds, of the - // post-canonicalize schema. Snapshot the schema the discriminator key - // was actually resolved against, right here, before recursing into the - // children — this is the exact DAG state `resolve.rs` saw. - let discriminator_branch_schema_before = match expr { - QueryExpr::Concat { - children, - discriminator_unique_key: Some(_), - } => children.first().and_then(|c| c.output_schema().ok()), - _ => None, - }; - - // Bottom-up: canonicalize every child before matching at this node, so an - // inner heavy-hitter is promoted before an enclosing rewrite inspects it. - for child in children_mut(expr) { - canon(child); - } - - // If the first branch's output schema moved out from under the asserted - // key, the key can no longer be trusted — drop it (never re-derive it by - // guessing at name/position: the two rewrites above don't preserve - // column identity in a way that's safe to infer). A wrong `unique_keys` - // claim is a wrong query answer, not a missed optimization — see - // `ConcatDiscriminatorKey`'s soundness doc — so this errs conservatively: - // any difference at all (not just a column-count/type change) drops the - // key, including the schema becoming undecidable in either direction. - if let QueryExpr::Concat { - children, - discriminator_unique_key: key @ Some(_), - } = expr - { - let discriminator_branch_schema_after = - children.first().and_then(|c| c.output_schema().ok()); - if discriminator_branch_schema_before != discriminator_branch_schema_after { - *key = None; - } - } - - // Local rewrites chain: a `ROW_NUMBER()`-partitioned top-k rewrites to a - // `Limit{Sort}`, which the heavy-hitter rule may then promote to an - // `Aggregate([TopK])`. Each rule strictly simplifies the node, so applying - // them to a fixpoint terminates. - while let Some(rewritten) = - try_rewrite_rownumber_topk(expr).or_else(|| try_promote_additive_top_ranking(expr)) - { - *expr = rewritten; - } -} - -/// A `&mut QueryExpr` out of a child `Rc` — clone-on-write via -/// [`Rc::make_mut`]: free (no clone) while `r` is uniquely owned, which is -/// the overwhelmingly common case (a DAG `canonicalize` was just handed by -/// value); falls back to cloning just *this* node (its own fields — the -/// grandchildren stay shared `Rc`s, not deep-copied) only when some other -/// owner still holds the same `Rc`, e.g. a caller that kept its own clone -/// around (`once.clone()` in `is_idempotent` below — `QueryExpr::clone()` is -/// now a cheap `Rc`-bump, not a deep copy, so that clone shares structure -/// with `once` until a rewrite here needs to touch it). `Rc::get_mut` would -/// panic on exactly that case; `make_mut` degrades to a shallow copy instead -/// of requiring sole ownership as a precondition. Once a workload-level CSE -/// pass runs (issue #212, #222) and canonicalize sees an already-shared -/// sub-DAG from a *different* query, this is also the mechanism that keeps -/// canonicalizing one query from silently corrupting another's view of the -/// same shared node. -fn rc_mut(r: &mut Rc) -> &mut QueryExpr { - Rc::make_mut(r) -} - -/// Mutable references to the direct **operator** `QueryExpr` children of a -/// node — `canon`'s own top-down/bottom-up walk only ever visits the -/// relational skeleton, never descending into a scalar position (`Filter.pred`, -/// `ProjectItem.expr`, …): none of the three rewrite rules rewrite anything -/// inside a scalar sub-DAG, so there's nothing to gain by recursing into one, -/// and every scalar variant (issue #205) hits the catch-all below. -fn children_mut(expr: &mut QueryExpr) -> Vec<&mut QueryExpr> { - use QueryExpr::*; - match expr { - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220), not the relational skeleton — same "no children to recurse - // into" treatment as the scalar variants below. - Scan { .. } | EvalTimestamp | CurrentTimestamp | PromqlScalarBridge(_) => vec![], - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => vec![rc_mut(c)], - PromqlRelabel { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } - | PromqlSeriesSample { child, .. } - | PromqlInfoEnrich { child, .. } - | Sort { child, .. } - | Limit { child, .. } => vec![rc_mut(child)], - Concat { children, .. } => children.iter_mut().collect(), - Join { left, right, .. } | SetOp { left, right, .. } => { - vec![rc_mut(left), rc_mut(right)] - } - BinaryOp { lhs, rhs, .. } => vec![rc_mut(lhs), rc_mut(rhs)], - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => vec![], - } -} - -/// Recognise an additive-ranked -/// `Limit { Sort { [Project] Aggregate([Count | Sum]) } }` and rewrite it to -/// the canonical heavy-hitter `Aggregate([TopK])` over the explicit inner -/// aggregate. Returns `None` when the shape does not match. -fn try_promote_additive_top_ranking(expr: &QueryExpr) -> Option { - // Limit k, no offset (an OFFSET means "not the top k"). - let QueryExpr::Limit { - n: k, - offset: 0, - child, - } = expr - else { - return None; - }; - // A single ordering key on a column. - let QueryExpr::Sort { - keys, - partition_by, - child: sort_child, - } = child.as_ref() - else { - return None; - }; - let [SortKey { - expr: QueryExpr::Column(sort_col), - ascending, - .. - }] = keys.as_slice() - else { - return None; - }; - - // The ordered relation is an `Aggregate`, optionally behind a passthrough - // projection (a bare-column SELECT list). Map the sort key through the - // projection to the aggregate's own output column. - let (agg_expr, ranked_col) = match sort_child.as_ref() { - QueryExpr::Project { cols, child, .. } => { - let QueryExpr::Column(underlying) = &cols.get(*sort_col)?.expr else { - return None; - }; - (child.as_ref(), *underlying) - } - other => (other, *sort_col), - }; - - // Exactly one aggregate, ranked by *its* output column — the measure sits at - // index `by.len()` (after the group keys). A `PerEntity` reduction has no - // `by` to rank a measure against — this shape can't be heavy-hitter - // promoted, so it's a non-match rather than an error. - let QueryExpr::Aggregate { - reduction, - measures, - filters, - child: aggregate_child, - .. - } = agg_expr - else { - return None; - }; - let Reduction::Reduce(by) = reduction else { - return None; - }; - let [ranked_agg] = measures.as_slice() else { - return None; - }; - // A heavy-hitter sketch ranks the raw update stream; a filtered measure - // only counts part of it, and no binding rule applies the filter (#466). - if filters.iter().any(Option::is_some) { - return None; - } - if ranked_col != by.len() { - return None; - } - // The heavy-hitter decision — descending, over a measure with a realised - // heavy-hitter sketch — is the shared rule both front ends' promotions - // consult (issue #38). So an ascending additive-ranked limit - // (`ORDER BY COUNT(*) ASC LIMIT k` = bottom-k) stays generic, exactly as - // PromQL `bottomk(k, count_over_time(…))` does. - if !topk::Ranking::from_aggregate(ranked_agg).is_supported(!ascending) { - return None; - } - // A direct Sum is a stream of additive observation weights. A Sum over a - // derived child such as Rate/Increase is different: a heavy-hitter sketch - // may propose candidate membership, but PromQL still requires exact - // reset-aware/extrapolated values to rerank those candidates. The current - // post-ASAP IR has no candidate-sidecar + exact-rerank node, so keep that - // shape as Sort + Limit instead of treating a sketch estimate as final. - if matches!(ranked_agg, AggIntent::Sum { .. }) - && matches!(aggregate_child.as_ref(), QueryExpr::Aggregate { .. }) - { - return None; - } - // Count ranks unit updates; a direct Sum ranks weighted updates. - let accuracy = match ranked_agg { - AggIntent::Count { accuracy } => accuracy.clone(), - AggIntent::Sum { .. } => AccuracyTarget::Exact, - _ => unreachable!("additive ranking gate admitted a non-additive measure"), - }; - - // Outer heavy-hitter `TopK`, grouped by the ranking's partition (empty for a - // global `ORDER BY … LIMIT k`; the `by` labels for a partitioned `topk by`), - // over the unchanged inner additive aggregate. - Some(QueryExpr::Aggregate { - reduction: Reduction::by(partition_by.to_vec()), - measures: vec![AggIntent::TopK { k: *k, accuracy }], - output_names: Vec::new(), - filters: Vec::new(), - having: None, - child: Rc::new(agg_expr.clone()), - }) -} - -/// Recognise the SQL partitioned-top-k idiom — `WHERE rn <= k` over a -/// `ROW_NUMBER() OVER (PARTITION BY p ORDER BY o)` — and rewrite it to the -/// generic partitioned top-k `Limit{k} { Sort{ o, partition_by: p } }` (issue -/// #24). The count-ranked case is then promoted to a heavy-hitter `TopK` by -/// [`try_promote_additive_top_ranking`], so a SQL `ROW_NUMBER` top-k and the PromQL -/// `topk by (…)` it mirrors converge on the same canonical shape. -fn try_rewrite_rownumber_topk(expr: &QueryExpr) -> Option { - // Filter { pred: `Column(rn) <= k` }. - let QueryExpr::Filter { pred, child } = expr else { - return None; - }; - let Predicate(pred_expr) = pred; - let QueryExpr::Compare { left, op, right } = pred_expr.as_ref() else { - return None; - }; - // `rn <= k` (top-k). `rn < k` would be off-by-one; require `<=`. - if *op != CompareOpKind::Le { - return None; - } - let (QueryExpr::Column(rn_col), QueryExpr::Literal(ScalarValue::Int64(k))) = - (left.as_ref(), right.as_ref()) - else { - return None; - }; - if *k < 0 { - return None; - } - - // Optionally strip a passthrough projection (the derived table's SELECT that - // re-exposes the aggregate columns + rn), mapping the rn column through it. - let (wf_expr, rn_in_wf) = match child.as_ref() { - QueryExpr::Project { cols, child, .. } => { - let QueryExpr::Column(underlying) = &cols.get(*rn_col)?.expr else { - return None; - }; - (child.as_ref(), *underlying) - } - other => (other, *rn_col), - }; - - // The filtered column must be a `ROW_NUMBER()` window output — the single - // column the SQLWindowFunc appends after its input, i.e. the last one. - let QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - partition_by, - order_by, - child: inner, - .. - } = wf_expr - else { - return None; - }; - if order_by.is_empty() { - return None; - } - let inner_cols = inner.output_schema().ok()?.fields.len(); - if rn_in_wf != inner_cols { - return None; // the predicate ranks some other column, not the row number - } - - // Generic partitioned top-k. The window's ORDER BY keys are relative to its - // input (`inner`), so they transfer directly to a `Sort` over `inner`. - Some(QueryExpr::Limit { - n: *k as usize, - offset: 0, - child: Rc::new(QueryExpr::Sort { - keys: order_by.clone(), - partition_by: partition_by.clone(), - child: Rc::new(inner.as_ref().clone()), - }), - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::query_expr::{ - GroupKeys, ProjectItem, Source, WindowFrame, WindowFrameBound, WindowFrameOffset, - WindowFrameUnits, - }; - use crate::pre_asap::schema::{DataType, Field, Schema}; - use crate::types::AccuracyTarget; - - fn scan() -> QueryExpr { - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("service", DataType::Utf8, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - } - } - - /// `Aggregate{ by: [1], [Count] }` over the scan — output cols `[service, count]`. - fn count_by_service() -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(vec![1]), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - } - } - - fn desc(col: usize) -> Vec { - vec![SortKey { - expr: QueryExpr::Column(col), - ascending: false, - nulls_first: false, - }] - } - - fn limit(n: usize, offset: usize, child: QueryExpr) -> QueryExpr { - QueryExpr::Limit { - n, - offset, - child: Rc::new(child), - } - } - - fn sort(keys: Vec, child: QueryExpr) -> QueryExpr { - QueryExpr::Sort { - keys, - partition_by: GroupKeys::by(vec![]), - child: Rc::new(child), - } - } - - fn is_topk_over_count(qe: &QueryExpr) -> bool { - matches!(qe, - QueryExpr::Aggregate { measures, child, .. } - if matches!(measures.as_slice(), [AggIntent::TopK { k: 5, .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Count { .. }]))) - } - - #[test] - fn promotes_count_ranked_limit_sort() { - // Limit 5 { Sort DESC by count-col (1) { Aggregate[Count] by [1] } }. - let q = limit(5, 0, sort(desc(1), count_by_service())); - assert!(is_topk_over_count(&canonicalize(q))); - } - - // A heavy-hitter sketch ranks every row; a count that only counts some - // rows (#466) is not that, so the generic Sort + Limit stays. - #[test] - fn does_not_promote_a_filtered_count_ranking() { - let mut filtered = count_by_service(); - let QueryExpr::Aggregate { filters, .. } = &mut filtered else { - unreachable!() - }; - *filters = vec![Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), - })))]; - let q = limit(5, 0, sort(desc(1), filtered)); - let canonical = canonicalize(q.clone()); - assert!(!is_topk_over_count(&canonical)); - assert_eq!(canonical, q); - } - - #[test] - fn promotes_through_a_passthrough_projection() { - // …with a `SELECT service, count` projection between the Sort and the Agg. - let proj = QueryExpr::Project { - cols: vec![ - ProjectItem { - alias: None, - expr: QueryExpr::Column(0), - }, - ProjectItem { - alias: Some("c".into()), - expr: QueryExpr::Column(1), - }, - ], - qualifier: None, - child: Rc::new(count_by_service()), - }; - let q = limit(5, 0, sort(desc(1), proj)); - assert!(is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn is_idempotent() { - let q = limit(5, 0, sort(desc(1), count_by_service())); - let once = canonicalize(q); - let twice = canonicalize(once.clone()); - assert_eq!(once, twice, "canonicalize must be idempotent"); - } - - // ── Concat's discriminator_unique_key vs. canonicalize (issue #228 review) ── - // - // `resolve.rs` resolves `discriminator_unique_key`'s `ColumnId`s against - // the first branch's *pre-canonicalize* output schema. If canonicalize - // then restructures that branch (heavy-hitter promotion, the - // `ROW_NUMBER()` top-k rewrite), those `ColumnId`s can end up pointing at - // the wrong column — or out of bounds — of the new schema. The two tests - // below pin the fix: the key is dropped whenever the branch's schema - // actually changed, and survives untouched otherwise. Never guessed at. - - #[test] - fn concat_discriminator_key_survives_canonicalize_when_first_branch_is_unaffected() { - // A plain `Aggregate` first branch matches neither rewrite trigger, - // so its schema is identical before and after canonicalize. - let q = QueryExpr::concat_with_discriminator( - vec![count_by_service(), count_by_service()], - /* discriminator */ 0, - /* inner_key */ vec![1], - ); - let QueryExpr::Concat { - discriminator_unique_key, - .. - } = canonicalize(q) - else { - panic!("expected Concat"); - }; - assert!( - discriminator_unique_key.is_some(), - "an untouched first branch's discriminator key must survive canonicalize" - ); - } - - #[test] - fn concat_discriminator_key_is_dropped_when_first_branch_gets_rewritten() { - // The first branch is exactly the heavy-hitter promotion trigger — - // `Limit{Sort{Aggregate([Count])}}`, with an empty (global) - // `partition_by` — so canonicalize rewrites it in place to - // `Aggregate{TopK}`, whose own output is a single column, not the - // original two (`[service, count]`). A discriminator key resolved - // against the original 2-column shape (`discriminator` = `service` - // at index 0, `inner_key` = `count` at index 1) must not silently - // survive pointing at the new 1-column schema. - let promotable_branch = limit(5, 0, sort(desc(1), count_by_service())); - let q = QueryExpr::concat_with_discriminator( - vec![promotable_branch, count_by_service()], - /* discriminator */ 0, - /* inner_key */ vec![1], - ); - let QueryExpr::Concat { - children, - discriminator_unique_key, - } = canonicalize(q) - else { - panic!("expected Concat"); - }; - assert!( - is_topk_over_count(&children[0]), - "the first branch is still promoted normally" - ); - assert!( - discriminator_unique_key.is_none(), - "a stale discriminator key must be dropped, never silently kept wrong" - ); - } - - #[test] - fn does_not_promote_ascending_sort() { - // Ascending = bottom-k: the Top-K operator's ranking rule - // rejects it (needs descending), so it stays a generic Sort+Limit — the - // same call PromQL `bottomk` makes (issue #38). - let asc = vec![SortKey { - expr: QueryExpr::Column(1), - ascending: true, - nulls_first: false, - }]; - let q = limit(5, 0, sort(asc, count_by_service())); - assert!(!is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn does_not_promote_with_offset() { - let q = limit(5, 2, sort(desc(1), count_by_service())); - assert!(!is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn does_not_promote_ranking_by_a_group_key() { - // DESC by col 0 (the `service` group key), not the count → not a - // frequency heavy-hitter. - let q = limit(5, 0, sort(desc(0), count_by_service())); - assert!(!is_topk_over_count(&canonicalize(q))); - } - - #[test] - fn promotes_sum_ranked_limit_sort_as_weighted_heavy_hitter() { - let sum = QueryExpr::Aggregate { - reduction: Reduction::by(vec![1]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let q = limit(5, 0, sort(desc(1), sum)); - let out = canonicalize(q); - let QueryExpr::Aggregate { - measures, child, .. - } = out - else { - panic!("expected weighted TopK aggregate"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 5, .. }] - )); - assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Sum { .. }])) - ); - } - - #[test] - fn keeps_sum_over_counter_reduction_as_exact_value_ranking() { - for counter in [AggIntent::Rate, AggIntent::Increase] { - let derived = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![counter], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let sum = QueryExpr::Aggregate { - reduction: Reduction::by(vec![1]), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(derived), - }; - let out = canonicalize(limit(5, 0, sort(desc(1), sum))); - assert!(matches!(out, QueryExpr::Limit { child, .. } - if matches!(child.as_ref(), QueryExpr::Sort { child, .. } - if matches!(child.as_ref(), QueryExpr::Aggregate { measures, child, .. } - if matches!(measures.as_slice(), [AggIntent::Sum { .. }]) - && matches!(child.as_ref(), QueryExpr::Aggregate { .. }))))); - } - } - - // ── ROW_NUMBER() partitioned top-k (issue #24) ────────────────────────── - - /// A scan with `[ts, service, region, value]`. - fn scan4() -> QueryExpr { - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("service", DataType::Utf8, false), - Field::plain("region", DataType::Utf8, false), - Field::plain("value", DataType::Float64, false), - ], - 0, - vec![], - ), - } - } - - /// `Aggregate{ by: [1,2] (service, region), [agg] }` — output `[service, - /// region, ]` (3 cols), so a ROW_NUMBER over it appends `rn` at index 3. - fn grouped(agg: AggIntent) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(vec![1, 2]), - measures: vec![agg], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan4()), - } - } - - /// `ROW_NUMBER` ignores its frame clause, so the top-k rewrite doesn't care - /// what's in it; any concrete frame works as fixture data. - fn rownumber_frame() -> WindowFrame { - WindowFrame { - units: WindowFrameUnits::Rows, - start_bound: WindowFrameBound::Preceding(WindowFrameOffset::Scalar(ScalarValue::Null)), - end_bound: WindowFrameBound::Following(WindowFrameOffset::Scalar(ScalarValue::Null)), - } - } - - /// `Filter{ rn(3) <= 5 } { SQLWindowFunc{ RowNumber, PARTITION BY region(2), - /// ORDER BY col(2) DESC } { agg } }`. - fn rownumber_topk(agg: QueryExpr) -> QueryExpr { - let wf = QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::by(vec![2]), // region - order_by: vec![SortKey { - expr: QueryExpr::Column(2), // the aggregate output column - ascending: false, - nulls_first: true, - }], - frame: Some(rownumber_frame()), - output_name: "rn".into(), - child: Rc::new(agg), - }; - QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), // rn = the appended window column - op: CompareOpKind::Le, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(5))), - })), - child: Rc::new(wf), - } - } - - #[test] - fn rownumber_count_topk_becomes_a_partitioned_heavy_hitter() { - // Count-ranked ROW_NUMBER top-k → outer TopK grouped by the partition - // (region, col 2) over the explicit inner Count. - let q = rownumber_topk(grouped(AggIntent::Count { - accuracy: AccuracyTarget::Exact, - })); - let out = canonicalize(q); - let QueryExpr::Aggregate { - reduction, - measures, - child, - .. - } = &out - else { - panic!("expected outer Aggregate([TopK]), got {out:?}"); - }; - let Reduction::Reduce(by) = reduction else { - panic!("expected a Reduce grouping, got {reduction:?}"); - }; - assert!(matches!( - measures.as_slice(), - [AggIntent::TopK { k: 5, .. }] - )); - assert_eq!(**by, vec![2], "outer TopK partitioned by region"); - assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Count { .. }])) - ); - } - - #[test] - fn rownumber_avg_topk_becomes_a_partitioned_sort_limit() { - // Avg-ranked (not a frequency heavy-hitter) → generic partitioned - // top-k: Limit{5}{ Sort{ partition_by: [region] } }. - let q = rownumber_topk(grouped(AggIntent::Avg { col: None })); - let out = canonicalize(q); - let QueryExpr::Limit { n, child, .. } = &out else { - panic!("expected a Limit, got {out:?}"); - }; - assert_eq!(*n, 5); - let QueryExpr::Sort { - partition_by, - child, - .. - } = child.as_ref() - else { - panic!("expected a Sort under the Limit"); - }; - assert_eq!(**partition_by, vec![2], "partitioned by region"); - assert!( - matches!(child.as_ref(), QueryExpr::Aggregate { measures, .. } - if matches!(measures.as_slice(), [AggIntent::Avg { .. }])) - ); - } - - #[test] - fn filter_on_a_non_rownumber_column_is_left_alone() { - // `WHERE service_len <= 5` (col 0, not the rn window column) must not be - // mistaken for a top-k. - let wf = QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::by(vec![2]), - order_by: vec![SortKey { - expr: QueryExpr::Column(2), - ascending: false, - nulls_first: true, - }], - frame: Some(rownumber_frame()), - output_name: "rn".into(), - child: Rc::new(grouped(AggIntent::Count { - accuracy: AccuracyTarget::Exact, - })), - }; - let q = QueryExpr::Filter { - pred: Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(0)), // NOT the rn column (index 3) - op: CompareOpKind::Le, - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(5))), - })), - child: Rc::new(wf), - }; - assert!( - matches!(canonicalize(q), QueryExpr::Filter { .. }), - "left as a Filter" - ); - } -} diff --git a/crates/types/src/pre_asap/cse.rs b/crates/types/src/pre_asap/cse.rs deleted file mode 100644 index 0a00475e6..000000000 --- a/crates/types/src/pre_asap/cse.rs +++ /dev/null @@ -1,1111 +0,0 @@ -//! Pre-ASAP structural common-subexpression elimination: bottom-up -//! hash-consing over an already-`resolve_root`'d [`QueryExpr`] DAG (issue -//! #212, #222, #223). -//! -//! CSE only runs on an already-bound, already-canonicalized DAG — -//! structural matching is meaningless before canonicalization has converged -//! semantically-equivalent queries onto one shape (`docs/develop_docs/pre-asap-ir.md` -//! design principle 3; `median(latency)` and `approx_percentile_cont(latency, -//! 0.5)` already lower to an identical `AggIntent::Quantile` today, per -//! `sql_lowering.rs`'s `median_is_the_same_intent_as_an_explicit_half_percentile` -//! test). [`share_common_sub_dags`] is the single entry point, run once per -//! workload batch (or once per query — see "Single-query CSE" below) *after* -//! `resolve_root`, *before* the pre-ASAP → post-ASAP replacement/search pass -//! (`asap_aware_mapping::replacement`). -//! -//! ## Algorithm: classic hash-consing / value-numbering -//! -//! Bottom-up: every child is interned before its parent, so a parent's -//! candidacy for sharing naturally incorporates whether its own children were -//! themselves shared — two parents whose children were independently -//! deduplicated down to the same `Rc`s are structurally identical iff their -//! own fields also match, without re-walking the sub-DAGs. -//! -//! Only the **relational skeleton** participates — the same set of "operator" -//! children [`canonicalize`](super::canonicalize)'s `children_mut` walks -//! (`Filter`/`Project`/`Aggregate`/`Concat`/`Join`/`BinaryOp`/…). A scalar -//! subexpression reachable only through a wrapper position (`Predicate`, -//! `ProjectItem.expr`, `Aggregate.having`, `SQLWindowFunc.args`, …) stays -//! embedded as opaque data on its owning operator node, compared by -//! `QueryExpr`'s derived `PartialEq` along with the rest of that node's -//! fields, rather than separately hash-consed — the same scope -//! `canonicalize.rs` settled on ("none of the rewrite rules touch a scalar -//! sub-DAG, so there's nothing to gain by recursing into one"). Widening this -//! to scalar positions is future work, not attempted here. -//! -//! ## Correctness: hash is a filter, `PartialEq` is the decision -//! -//! This is the one non-negotiable rule. A **false positive** here — two -//! sub-DAGs wrongly judged shareable — is a wrong query answer, not a missed -//! optimization: two different queries would read each other's data. -//! [`structural_hash`] (`DefaultHasher`/SipHash over a canonical -//! serialization, no collision-freedom guarantee) may only narrow the -//! candidate set within one bucket; [`InternTable::intern`]'s `PartialEq` -//! check on that bucket is what actually decides sharing, every time, no -//! exceptions for "the hash probably didn't collide." -//! -//! This also means the pass is safe by construction against the case #212 -//! flagged as a real historical bug (issue #115): `AggIntent::Quantile` -//! carries its input column and its `AccuracyTarget`, both `PartialEq` -//! fields, so `Quantile(x, 0.99, ε=0.01)` and `Quantile(x, 0.99, ε=0.001)` — -//! or `Quantile(x, ..)` vs `Quantile(y, ..)` — are never merged. This is -//! intentionally conservative: it only recognizes *exact* structural -//! matches, not "a stricter-accuracy summary could also answer a looser -//! request." That subsumption question already has a documented, -//! deliberately-unfilled home (`asap_aware_mapping::Matcher`) — -//! CSE here does not attempt it. -//! -//! ## Legality: gated by `Schema::unique_keys` -//! -//! Structural equality alone is necessary but not sufficient. Per -//! [`Schema::unique_keys`](super::schema::Schema::unique_keys)'s own doc: "a -//! producer's output can only be safely shared across consumers when its row -//! identity is provably stable across reads." A candidate node with no -//! provable unique key (`Schema::has_unique_key()` false, or `output_schema` -//! not even defined for that node, e.g. a `Concat`/`SetOp` branch whose union -//! drops `unique_keys`, or an ungrouped/global `Aggregate`, whose empty `by` -//! also reports no unique key today) is **never** hoisted, even when it is -//! structurally identical to something already interned — it is always -//! inserted fresh, matching the rule the (now-deleted) prior CSE attempt -//! already encoded and the doc comment on `Aggregate`'s `child` field -//! ("`unique_keys` feeds CSE's producer-sharing legality check"). -//! -//! ## Single-query CSE falls out for free -//! -//! A repeated sub-expression within *one* query (e.g. the same grouped -//! `Aggregate` referenced twice on two `BinaryOp` branches) is deduplicated -//! by the exact same bottom-up interning — a workload of size one still -//! interns bottom-up within that one DAG. No separate mechanism is needed; -//! see the `single_query_shares_its_own_repeated_sub_dag` test below. -//! -//! ## Landing plan (issue #223) -//! -//! This module is stage 1 of a 4-stage plan. Stage 2 -//! (`asap_aware_mapping::replacement::search_workload_with`, which runs -//! [`share_common_sub_dags`] itself before searching) is a real caller, -//! wired at the same time so this never becomes unwired dead code again -//! (the original `asap-plan::cse::dedupe_subtrees` was deleted in #192 for -//! exactly that). Stage 3 — [`dag_export`](crate::dag_export) computing its -//! per-node `hash` by calling this module's [`structural_hash`] directly, -//! instead of a parallel reimplementation — is also done, so -//! `tools/dag-viewer`'s "shared sub-DAG" highlighting now flags exactly the -//! candidate pairs this module's own `InternTable` would bucket together -//! (still only a hash match, not a guarantee of -//! `share_common_sub_dags`-actual sharing — see `dag_export`'s module doc). -//! Stage 4 (issue #237) is implemented in -//! `asap_aware_mapping::cost_model::CostModel::cse_share_decision`, called -//! from `asap_aware_mapping::replacement::CandidateLogicalASAPDAGs::cost_sorted` (via that -//! module's own `cse_preference`) — a real, Volcano/Cascades-style cost -//! comparison over what this module detects, not a fixed rule. See -//! `docs/design_docs/cost-model.md`. This module's own -//! unconditional "share whenever legal" behavior is unchanged: detection -//! stays cost-agnostic by construction (this crate cannot depend on -//! `asap-aware-mapping`'s `CostModel`), and the cost-aware decision is -//! applied downstream, after detection, over what this module finds. - -use std::collections::HashMap; -use std::hash::{Hash, Hasher}; -use std::rc::Rc; - -use super::query_expr::QueryExpr; - -/// Bottom-up hash-consing table: structurally-equal, sharing-legal -/// [`QueryExpr`] nodes collapse onto one `Rc`. -/// -/// `buckets` is keyed by [`structural_hash`] — a coarse candidate filter -/// only (see the module-level "Correctness" section). Every entry within one -/// bucket is a full node kept around for the `PartialEq` comparison that -/// actually decides a match; a hash collision between structurally different -/// nodes just means a (harmless) linear scan of a few extra candidates. -struct InternTable { - buckets: HashMap>>, - /// Memoizes [`structural_hash`] per already-hashed `Rc` pointer, shared - /// across every [`intern`](Self::intern) call for the table's whole - /// lifetime — see [`structural_hash`]'s own doc on why this matters: - /// without it, hashing an `N`-node bottom-up pass costs `O(N)` work - /// *per node* (every already-interned descendant gets re-walked), not - /// `O(1)` amortized. - hash_cache: HashCache, -} - -impl InternTable { - fn new() -> Self { - Self { - buckets: HashMap::new(), - hash_cache: HashMap::new(), - } - } - - /// Intern one already-children-rebuilt node: look it up by - /// [`structural_hash`], confirm with `PartialEq`, and — only when - /// sharing is legal (see "Legality" above) — return the existing `Rc` - /// instead of allocating a new one. - fn intern(&mut self, node: QueryExpr) -> Rc { - let hash = structural_hash(&node, &mut self.hash_cache); - // A node with no provable unique key is never *returned* as a match - // for something else — it may still go on to occupy a fresh slot in - // the bucket (harmless; it just never gets found by a later - // `PartialEq` scan that also requires `reusable`). - let reusable = node - .output_schema() - .is_ok_and(|schema| schema.has_unique_key()); - let bucket = self.buckets.entry(hash).or_default(); - if reusable { - if let Some(existing) = bucket.iter().find(|candidate| candidate.as_ref() == &node) { - return Rc::clone(existing); - } - } - let rc = Rc::new(node); - bucket.push(Rc::clone(&rc)); - rc - } -} - -/// [`structural_hash`]'s memoization cache: maps an already-hashed node's -/// `Rc` pointer to its computed hash. Not tied to any one `QueryExpr` — a -/// fresh, empty cache is correct to start with anywhere; what matters is -/// letting it *persist* across every node in one bottom-up pass (as -/// [`InternTable`] does via its own `hash_cache` field), rather than -/// starting a new one per call. -/// -/// `pub` (not `pub(crate)`) so `asap_aware_mapping`'s workload-search MEMO -/// engine (`replacement::is_duplicate_rewrite`) can reuse this exact -/// candidate-narrowing filter for its own dedup, instead of maintaining a -/// parallel reimplementation — the same "one real hash, reused everywhere -/// it's needed" rationale [`structural_hash`]'s own doc gives for -/// [`dag_export`](crate::dag_export)'s `pub(crate)` reuse. -pub type HashCache = HashMap<*const QueryExpr, u64>; - -/// Coarse structural hash used only to bucket [`InternTable::intern`]'s -/// candidate search — never the actual sharing decision (`PartialEq` is). -/// -/// `QueryExpr` carries `f64`s (`Literal(ScalarValue::Float64)`, `AggIntent::Quantile.q`, …), so it -/// cannot derive `std::hash::Hash`. Serializing to a canonical JSON string -/// and hashing that sidesteps the `f64` problem — but only for `node`'s own -/// tag and non-child fields, *not* its children's full values: each -/// `Rc`-backed child's contribution is its own [`structural_hash`], looked -/// up in `cache` if already computed there (memoized by `Rc` pointer -/// identity) rather than recursed into again. -/// -/// This is the DAG-aware fix a naive "just serialize the whole sub-DAG" -/// hash would get wrong: after [`share_common_sub_dags`] (or even before -/// it — a front end can emit internal `Rc` sharing directly, e.g. a -/// repeated subexpression within one query), `node` generally has internal -/// sharing. A full-sub-DAG serialization re-serializes — re-walks — -/// any descendant `node` already shares internally once per parent that -/// references it; called once per node in a bottom-up pass (as -/// [`InternTable::intern`] and [`dag_export`](crate::dag_export) both do), -/// that costs `O(sub-DAG size)` *per node* instead of `O(1)` amortized — -/// quadratic-or-worse for a deep chain, compounding further with any real -/// internal sharing. Memoizing each child's hash by pointer identity in -/// `cache` (persisted across the whole pass by the caller, not reset per -/// node) makes each node's own contribution `O(1)` beyond its children's -/// already-known hashes, giving `O(N)` total for `N` nodes — matching -/// [`dag_node_count`]'s own shared-node counting fix (issue #212/#223/#237's stage -/// 4) in spirit, applied to hashing instead of counting. -/// -/// `pub` (not private) so [`dag_export`](crate::dag_export) can call -/// this exact function for its exported nodes' `hash` field instead of -/// maintaining its own parallel reimplementation — issue #223 stage 3. That -/// makes `tools/dag-viewer`'s "shared sub-DAG" highlighting reflect this -/// module's real hashing, not a lookalike computed a different way; see the -/// module doc's "Landing plan" section. A NaN/infinite `f64` makes JSON -/// serialization fail; falling back to a fixed hash just puts every such -/// node in one (larger, still `PartialEq`-disambiguated) bucket. Made `pub` -/// (rather than staying `pub(crate)`) for one more reuse across the crate -/// boundary: `asap_aware_mapping`'s workload-search MEMO engine -/// (`replacement::is_duplicate_rewrite`) needs the identical -/// candidate-narrowing filter this module's own [`InternTable::intern`] -/// already uses, so it doesn't have to reinvent (and risk drifting from) it. -/// -/// Exhaustive over every `QueryExpr` variant, matching [`rebuild_children`] -/// in which fields count as an operator child (must stay in sync — a new -/// variant fails to compile in both places until both are extended). -pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { - use QueryExpr::*; - - fn child_hash(child: &Rc, cache: &mut HashCache) -> u64 { - let ptr = Rc::as_ptr(child); - if let Some(&h) = cache.get(&ptr) { - return h; - } - let h = structural_hash(child, cache); - cache.insert(ptr, h); - h - } - - /// Hash `own_fields` (this node's own tag and non-child scalar - /// fields — anything JSON-serializable and small, i.e. never a - /// `QueryExpr` sub-DAG) via the same canonical-JSON-string trick the - /// whole-sub-DAG version used, just applied to `O(1)` fields instead - /// of `O(sub-DAG size)`. - fn hash_own_fields(hasher: &mut impl Hasher, own_fields: &impl serde::Serialize) { - serde_json::to_string(own_fields) - .unwrap_or_default() - .hash(hasher); - } - - let mut hasher = std::collections::hash_map::DefaultHasher::new(); - match node { - Scan { - source, - predicates, - schema, - } => hash_own_fields(&mut hasher, &("Scan", source, predicates, schema)), - PromqlVectorFromScalar(c) => { - "PromqlVectorFromScalar".hash(&mut hasher); - child_hash(c, cache).hash(&mut hasher); - } - PromqlScalarFromVector(c) => { - "PromqlScalarFromVector".hash(&mut hasher); - child_hash(c, cache).hash(&mut hasher); - } - PromqlRelabel { dst, value, child } => { - hash_own_fields(&mut hasher, &("PromqlRelabel", dst, value)); - child_hash(child, cache).hash(&mut hasher); - } - PromqlInfoEnrich { selector, child } => { - hash_own_fields(&mut hasher, &("PromqlInfoEnrich", selector)); - child_hash(child, cache).hash(&mut hasher); - } - PromqlSeriesSample { by, kind, child } => { - hash_own_fields(&mut hasher, &("PromqlSeriesSample", by, kind)); - child_hash(child, cache).hash(&mut hasher); - } - Filter { pred, child } => { - hash_own_fields(&mut hasher, &("Filter", pred)); - child_hash(child, cache).hash(&mut hasher); - } - Project { - cols, - qualifier, - child, - } => { - hash_own_fields(&mut hasher, &("Project", cols, qualifier)); - child_hash(child, cache).hash(&mut hasher); - } - Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => { - hash_own_fields( - &mut hasher, - &( - "Aggregate", - reduction, - measures, - output_names, - filters, - having, - ), - ); - child_hash(child, cache).hash(&mut hasher); - } - Dedup { cols, child } => { - hash_own_fields(&mut hasher, &("Dedup", cols)); - child_hash(child, cache).hash(&mut hasher); - } - Concat { - children, - discriminator_unique_key, - } => { - hash_own_fields(&mut hasher, &("Concat", discriminator_unique_key)); - for c in children { - // Stored by value, not `Rc` — see `rebuild_children`'s - // `intern_owned` use for this variant — so there's no - // pointer to memoize on here; recurse directly. Any - // `Rc`-typed descendant beneath `c` still gets memoized - // once this call reaches it. - structural_hash(c, cache).hash(&mut hasher); - } - } - Join { - kind, - pred, - left, - right, - } => { - hash_own_fields(&mut hasher, &("Join", kind, pred)); - child_hash(left, cache).hash(&mut hasher); - child_hash(right, cache).hash(&mut hasher); - } - SetOp { - kind, - all, - left, - right, - } => { - hash_own_fields(&mut hasher, &("SetOp", kind, all)); - child_hash(left, cache).hash(&mut hasher); - child_hash(right, cache).hash(&mut hasher); - } - Sort { - keys, - partition_by, - child, - } => { - hash_own_fields(&mut hasher, &("Sort", keys, partition_by)); - child_hash(child, cache).hash(&mut hasher); - } - Limit { n, offset, child } => { - hash_own_fields(&mut hasher, &("Limit", n, offset)); - child_hash(child, cache).hash(&mut hasher); - } - PromqlSubquery { - range, - resolution, - child, - } => { - hash_own_fields(&mut hasher, &("PromqlSubquery", range, resolution)); - child_hash(child, cache).hash(&mut hasher); - } - TimeRange { range, child } => { - hash_own_fields(&mut hasher, &("TimeRange", range)); - child_hash(child, cache).hash(&mut hasher); - } - TimeShift { shift, child } => { - hash_own_fields(&mut hasher, &("TimeShift", shift)); - child_hash(child, cache).hash(&mut hasher); - } - SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => { - hash_own_fields( - &mut hasher, - &( - "SQLWindowFunc", - func, - args, - partition_by, - order_by, - frame, - output_name, - ), - ); - child_hash(child, cache).hash(&mut hasher); - } - BinaryOp { - op, - lhs, - rhs, - vector_match, - } => { - hash_own_fields(&mut hasher, &("BinaryOp", op, vector_match)); - child_hash(lhs, cache).hash(&mut hasher); - child_hash(rhs, cache).hash(&mut hasher); - } - // `EvalTimestamp`, `PromqlScalarBridge`, and the scalar variants - // (issue #205) are all leaves for this traversal's purposes — none - // has an operator child to look up in `cache` — so hashing the - // whole node via `serde_json` in one shot is already `O(node - // size)`, not `O(sub-DAG size)`: exactly the same cost the - // per-variant `hash_own_fields` calls above pay, just without - // needing to spell out each field individually. Matches - // `rebuild_children`'s and `dag_node_count`'s identical scope - // decision for these variants ("never descended into"). - EvalTimestamp - | CurrentTimestamp - | PromqlScalarBridge(_) - | Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => hash_own_fields(&mut hasher, node), - } - hasher.finish() -} - -/// Count of *unique* nodes reachable from `root`, deduplicated by `Rc` -/// pointer identity (`Rc::as_ptr`) — the real size of the DAG rooted at -/// `root`, not a per-path walk count. -/// -/// After [`share_common_sub_dags`] runs (or even before it, for a DAG a -/// front end already built with internal `Rc` sharing — e.g. re-running -/// CSE, or a single-query repeated subexpression), `root` is generally a -/// **DAG** with internal sharing — that is this whole module's premise. Anything that -/// walks `root` as if every reference were a fresh sub-DAG (a naive -/// recursive walk with no identity tracking, or a naive full -/// `serde_json` serialization — `Rc`'s `Serialize` impl serializes the -/// pointee's *value* at every occurrence, it does not dedupe by identity) -/// re-visits/re-counts an already-shared descendant once per parent that -/// references it, over-counting relative to the actual work of holding it -/// in memory or recomputing it once. This function is the DAG-correct -/// alternative: each unique node is counted exactly once, regardless of -/// how many places within `root` reference it. -/// -/// `pub` so cost-aware callers outside this crate (e.g. -/// `asap_aware_mapping::CostModel::cse_recompute_cost`'s default) have a -/// DAG-correct structural-size proxy available, instead of reaching for -/// something per-path like a raw serialization length. -/// -/// Same operator-child traversal scope as [`share_common_sub_dags`] itself -/// (see the module doc's "Algorithm" section, and this module's private -/// `rebuild_children`) — a scalar subexpression embedded in a wrapper -/// position (`Predicate`, `ProjectItem.expr`, `Aggregate.having`, …) is not -/// separately visited, matching this module's own stated scope; it's -/// counted as part of its owning operator node, the same node -/// `rebuild_children` treats as a single opaque leaf for interning -/// purposes. -pub fn dag_node_count(root: &QueryExpr) -> usize { - let mut seen: std::collections::HashSet<*const QueryExpr> = std::collections::HashSet::new(); - count_unique(root, &mut seen) -} - -/// One node's own contribution (`1`) plus each *not-yet-seen* operator -/// child's contribution — exhaustive over every `QueryExpr` variant, -/// enumerating the same fields [`rebuild_children`] does (kept as a -/// separate, read-only traversal rather than threaded through -/// `rebuild_children` itself, since that function consumes and rebuilds -/// its input while this one only ever reads it). -fn count_unique(node: &QueryExpr, seen: &mut std::collections::HashSet<*const QueryExpr>) -> usize { - use QueryExpr::*; - - /// Visit one `Rc`-held child: counts (and recurses into) it only the - /// first time its pointer is seen, `0` on every later occurrence — - /// this is the actual dedup step. - fn visit( - child: &Rc, - seen: &mut std::collections::HashSet<*const QueryExpr>, - ) -> usize { - if seen.insert(Rc::as_ptr(child)) { - count_unique(child, seen) - } else { - 0 - } - } - - 1 + match node { - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220), never descended into — same treatment `rebuild_children` - // gives it (see that function's comment on this same variant). - Scan { .. } | PromqlScalarBridge(_) | EvalTimestamp | CurrentTimestamp => 0, - PromqlVectorFromScalar(c) | PromqlScalarFromVector(c) => visit(c, seen), - PromqlRelabel { child, .. } - | PromqlInfoEnrich { child, .. } - | PromqlSeriesSample { child, .. } - | Filter { child, .. } - | Project { child, .. } - | Aggregate { child, .. } - | Dedup { child, .. } - | Sort { child, .. } - | Limit { child, .. } - | PromqlSubquery { child, .. } - | TimeRange { child, .. } - | TimeShift { child, .. } - | SQLWindowFunc { child, .. } => visit(child, seen), - // `Concat`'s branches are stored by value (`Vec`, not - // `Rc` — see `rebuild_children`'s `intern_owned` use for - // this variant), so a branch has no `Rc` identity of its own to - // dedup on at this position; still recurse into each in case an - // `Rc`-shared descendant appears further down. - Concat { children, .. } => children.iter().map(|c| count_unique(c, seen)).sum(), - Join { left, right, .. } | SetOp { left, right, .. } => { - visit(left, seen) + visit(right, seen) - } - BinaryOp { lhs, rhs, .. } => visit(lhs, seen) + visit(rhs, seen), - // Scalar variants (issue #205) — never descended into, matching - // `rebuild_children`'s own scope exactly (see its trailing match - // arm and this module's "Algorithm" section). - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => 0, - } -} - -/// Recurse into `child`, then intern the result. `Rc::try_unwrap` recovers -/// the owned node without cloning in the overwhelmingly common case — a -/// DAG freshly built by a front end / `resolve_root`, not yet shared by any -/// prior CSE pass, where every `Rc` is uniquely owned. Falls back to cloning -/// this node's own fields (its children stay `Rc`s, not deep-copied) only -/// when `child` is already shared — e.g. re-running CSE over a DAG that -/// went through a previous `share_common_sub_dags` pass; a structural -/// duplicate collapses right back onto `child` itself via `PartialEq`, an -/// already-optimal no-op. -fn intern_child(table: &mut InternTable, child: Rc) -> Rc { - match Rc::try_unwrap(child) { - Ok(owned) => intern_bottom_up(table, owned), - Err(shared) => intern_bottom_up(table, (*shared).clone()), - } -} - -/// Like [`intern_child`], for a `Concat` branch — stored by value -/// (`Vec`, not `Rc`), so this position itself can never -/// alias another parent. Interning it anyway still lets any `Rc`-typed -/// descendant of the branch participate in sharing, and registers the -/// branch's own hash/value in the table for a *different* `Concat` elsewhere -/// with a structurally identical branch (which — being in its own `Vec` -/// slot too — still can't literally share the `Rc`, but this keeps the -/// interning behavior uniform and the table's bucket contents consistent). -fn intern_owned(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { - let rc = intern_bottom_up(table, expr); - Rc::try_unwrap(rc).unwrap_or_else(|shared| (*shared).clone()) -} - -/// Bottom-up: rebuild `expr`'s children (recursively interning each), then -/// intern the rebuilt node itself. -fn intern_bottom_up(table: &mut InternTable, expr: QueryExpr) -> Rc { - let rebuilt = rebuild_children(table, expr); - table.intern(rebuilt) -} - -/// Rebuild `expr` with each **operator** child (see the module doc on scope) -/// replaced by its interned `Rc`. Exhaustive over every `QueryExpr` variant, -/// matching `canonicalize.rs`'s `children_mut` exactly in which fields count -/// as an operator child — new variants fail to compile here until this match -/// is extended. -fn rebuild_children(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { - use QueryExpr::*; - match expr { - Scan { .. } | EvalTimestamp | CurrentTimestamp => expr, - PromqlVectorFromScalar(c) => PromqlVectorFromScalar(intern_child(table, c)), - PromqlScalarFromVector(c) => PromqlScalarFromVector(intern_child(table, c)), - PromqlRelabel { dst, value, child } => PromqlRelabel { - dst, - value, - child: intern_child(table, child), - }, - PromqlInfoEnrich { selector, child } => PromqlInfoEnrich { - selector, - child: intern_child(table, child), - }, - PromqlSeriesSample { by, kind, child } => PromqlSeriesSample { - by, - kind, - child: intern_child(table, child), - }, - Filter { pred, child } => Filter { - pred, - child: intern_child(table, child), - }, - Project { - cols, - qualifier, - child, - } => Project { - cols, - qualifier, - child: intern_child(table, child), - }, - Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => Aggregate { - reduction, - measures, - output_names, - filters, - having, - child: intern_child(table, child), - }, - Dedup { cols, child } => Dedup { - cols, - child: intern_child(table, child), - }, - Concat { - children, - discriminator_unique_key, - } => Concat { - children: children - .into_iter() - .map(|c| intern_owned(table, c)) - .collect(), - discriminator_unique_key, - }, - Join { - kind, - pred, - left, - right, - } => Join { - kind, - pred, - left: intern_child(table, left), - right: intern_child(table, right), - }, - SetOp { - kind, - all, - left, - right, - } => SetOp { - kind, - all, - left: intern_child(table, left), - right: intern_child(table, right), - }, - Sort { - keys, - partition_by, - child, - } => Sort { - keys, - partition_by, - child: intern_child(table, child), - }, - Limit { n, offset, child } => Limit { - n, - offset, - child: intern_child(table, child), - }, - PromqlSubquery { - range, - resolution, - child, - } => PromqlSubquery { - range, - resolution, - child: intern_child(table, child), - }, - TimeRange { range, child } => TimeRange { - range, - child: intern_child(table, child), - }, - TimeShift { shift, child } => TimeShift { - shift, - child: intern_child(table, child), - }, - SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child: intern_child(table, child), - }, - BinaryOp { - op, - lhs, - rhs, - vector_match, - } => BinaryOp { - op, - lhs: intern_child(table, lhs), - rhs: intern_child(table, rhs), - vector_match, - }, - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220) — same "never descended into" treatment as the scalar - // variants below; the whole bridge node is still interned as a unit - // by the `table.intern(rebuilt)` call in `intern_bottom_up`. - PromqlScalarBridge(_) => expr, - // Scalar variants (issue #205) — never descended into; see the - // module doc's "Algorithm" section on scope. Left byte-for-byte - // unchanged: predicate / project-list / sort-key / window-arg - // expressions stay embedded as opaque leaf data, compared by the - // enclosing operator node's derived `PartialEq`. - Column(_) - | Literal(_) - | Compare { .. } - | BoolAnd(_) - | BoolOr(_) - | Not(_) - | IsNull(_) - | IsNotNull(_) - | Cast { .. } - | InList { .. } - | FunctionCall { .. } - | Arithmetic { .. } - | Case { .. } => expr, - } -} - -/// Share structurally-identical, sharing-legal sub-DAGs across a workload's -/// query roots (or within one query, for `roots.len() == 1` — see the -/// module doc's "Single-query CSE" section). Every root's *value* is -/// unchanged (`PartialEq`-equal to its input) — only its internal `Rc` -/// structure may now alias another root's, or another part of its own DAG. -/// -/// `roots` must already be bound + canonicalized (post-`resolve_root`). -/// `Id` is caller-chosen — a `QueryWorkload` entry's own key, an index, a -/// query name, whatever identifies one root through the pipeline; this -/// module has no opinion on its shape. -pub fn share_common_sub_dags(roots: Vec<(Id, QueryExpr)>) -> Vec<(Id, Rc)> { - let mut table = InternTable::new(); - roots - .into_iter() - .map(|(id, expr)| (id, intern_bottom_up(&mut table, expr))) - .collect() -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::agg_intent::AggIntent; - use crate::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; - use crate::pre_asap::query_expr::{BinaryOpKind, GroupKeys, Predicate, Reduction, Source}; - use crate::pre_asap::schema::{DataType, Field, Schema}; - use crate::types::AccuracyTarget; - - /// `[ts, service, value, latency]`. - fn scan() -> QueryExpr { - QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("service", DataType::Utf8, false), - Field::plain("value", DataType::Float64, false), - Field::plain("latency", DataType::Float64, false), - ], - 0, - vec![], - ), - } - } - - fn quantile_agg(by: Vec, col: Option, q: f64) -> QueryExpr { - QueryExpr::Aggregate { - reduction: Reduction::by(by), - measures: vec![AggIntent::Quantile { - col, - q, - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - } - } - - #[test] - fn distinct_column_quantiles_do_not_merge() { - // Grouped (unique_keys present) so the legality gate isn't what's - // blocking the merge — only the differing `col` is. - let a = quantile_agg(vec![1], Some(2), 0.5); - let b = quantile_agg(vec![1], Some(3), 0.5); - let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - !Rc::ptr_eq(ra, rb), - "distinct-column Quantiles must not be shared" - ); - assert_ne!(ra, rb); - } - - // Two aggregates that differ only in one measure's `FILTER` predicate - // compute different values, so structural sharing must keep them apart. - #[test] - fn filtered_and_unfiltered_aggregates_do_not_merge() { - let a = quantile_agg(vec![1], Some(2), 0.5); - let mut b = quantile_agg(vec![1], Some(2), 0.5); - let QueryExpr::Aggregate { filters, .. } = &mut b else { - unreachable!() - }; - *filters = vec![Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(3)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), - })))]; - let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!(!Rc::ptr_eq(ra, rb), "a filtered measure must not be shared"); - assert_ne!(ra, rb); - } - - #[test] - fn no_unique_keys_means_no_merge_even_when_structurally_identical() { - // Ungrouped (global) aggregate: `by` is empty, so - // `aggregate_output_schema` reports no unique key today — not - // hoistable even though `a` and `b` are structurally identical. - let a = quantile_agg(vec![], Some(2), 0.9); - let b = quantile_agg(vec![], Some(2), 0.9); - assert_eq!(a, b, "fixture sanity: the two DAGs are structurally equal"); - assert!( - !a.output_schema().unwrap().has_unique_key(), - "fixture sanity: an ungrouped aggregate has no provable unique key" - ); - let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - !Rc::ptr_eq(ra, rb), - "no unique key ⇒ never hoisted, even for an identical structural match" - ); - } - - #[test] - fn median_and_explicit_half_percentile_merge() { - // Two front-end spellings ("median" and "approx_percentile_cont(., - // 0.5)") already lower to the identical `AggIntent::Quantile { q: - // 0.5, .. }` today (see `sql_lowering.rs`'s - // `median_is_the_same_intent_as_an_explicit_half_percentile`) — here - // built directly (grouped, so a unique key is provable) as two - // independently-constructed but structurally identical DAGs, the - // way two different call sites in a workload would produce them. - let median = quantile_agg(vec![1], Some(2), 0.5); - let approx_percentile_cont_half = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_sub_dags(vec![ - ("median", median), - ("percentile", approx_percentile_cont_half), - ]); - let [(_, m), (_, p)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - Rc::ptr_eq(m, p), - "median and an explicit 0.5 percentile must merge onto one Rc" - ); - } - - #[test] - fn single_query_shares_its_own_repeated_sub_dag() { - // One query root referencing the same grouped Aggregate on both - // BinaryOp branches — built as two separately-allocated but - // structurally identical sub-DAGs (`.clone()` into two distinct - // `Rc::new` calls), the shape a front end emitting a repeated - // sub-expression would actually produce (no sharing yet). A - // workload of size 1 still interns bottom-up within this one DAG — - // no separate single-query mechanism needed. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg), - vector_match: None, - }; - let shared = share_common_sub_dags(vec![("q", root)]); - let [(_, root)] = shared.as_slice() else { - panic!("expected 1 root"); - }; - let QueryExpr::BinaryOp { lhs, rhs, .. } = root.as_ref() else { - panic!("expected BinaryOp root, got {root:?}"); - }; - assert!( - Rc::ptr_eq(lhs, rhs), - "the two structurally identical branches must collapse onto one Rc" - ); - } - - // ── structural_hash (DAG-aware memoization) ───────────────────────── - - #[test] - fn structural_hash_is_stable_across_cache_states() { - // The hash of a given *value* must not depend on whether its cache - // started warm or cold — memoization changes how much work is - // redone, never what a node's hash actually is. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let mut cold = HashMap::new(); - let mut warm = HashMap::new(); - // Prime `warm` with an unrelated node first, so it's non-empty but - // holds nothing relevant to `agg`. - structural_hash(&scan(), &mut warm); - assert_eq!( - structural_hash(&agg, &mut cold), - structural_hash(&agg, &mut warm), - "hash must be independent of unrelated cache state" - ); - } - - #[test] - fn structural_hash_of_an_internally_shared_dag_matches_the_unshared_equivalent() { - // The same BinaryOp-with-shared-branches shape as - // `dag_node_count_deduplicates_an_internally_shared_sub_dag` below: - // hashing it (however the memoization internally short-circuits the - // second branch) must produce the exact same value as hashing a - // structurally-identical DAG built with *no* sharing at all — the - // whole point of memoization is not changing the answer, only the - // work needed to reach it. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let shared_root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg.clone()), - vector_match: None, - }; - let unshared_root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg), // a second, independently-allocated Rc with an equal value - vector_match: None, - }; - let mut cache = HashMap::new(); - assert_eq!( - structural_hash(&shared_root, &mut cache), - structural_hash(&unshared_root, &mut HashMap::new()), - ); - } - - #[test] - fn structural_hash_memoizes_a_shared_descendant_exactly_once() { - // Direct proof the cache is actually doing its job: hashing a - // BinaryOp whose two branches are the *same* Rc (2 underlying - // nodes: Scan + Aggregate) should populate the cache with exactly - // 2 entries — the shared branch's nodes, cached once each when - // first reached — not a fresh entry (or a fresh, redundant - // recursive walk) for the second occurrence. - let agg = quantile_agg(vec![1], Some(2), 0.5); - let shared = Rc::new(agg); - let root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::clone(&shared), - rhs: Rc::clone(&shared), - vector_match: None, - }; - let mut cache = HashMap::new(); - structural_hash(&root, &mut cache); - assert_eq!( - cache.len(), - 2, - "expected exactly one cache entry per unique node in the shared \ - branch (Aggregate + its Scan child), got {} entries: {:?}", - cache.len(), - cache - ); - } - - // ── dag_node_count ─────────────────────────────────────────────────── - - #[test] - fn dag_node_count_is_the_naive_count_when_nothing_is_shared() { - // scan() alone: 1 node. - assert_eq!(dag_node_count(&scan()), 1); - // quantile_agg's own child is a fresh, unshared scan(): 2 nodes. - assert_eq!(dag_node_count(&quantile_agg(vec![1], Some(2), 0.5)), 2); - } - - #[test] - fn dag_node_count_deduplicates_an_internally_shared_sub_dag() { - // Same shape as `single_query_shares_its_own_repeated_sub_dag`: a - // BinaryOp whose two branches are the *same* Rc after - // `share_common_sub_dags` (2 nodes: Scan + Aggregate) — the root - // itself makes 3 unique nodes total (BinaryOp, Aggregate, Scan), - // not 5 (which a per-path walk / naive serialization, counting the - // shared branch's 2 nodes twice, would report). - let agg = quantile_agg(vec![1], Some(2), 0.5); - let root = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(crate::pre_asap::expr_ir::CompareOpKind::Eq), - lhs: Rc::new(agg.clone()), - rhs: Rc::new(agg), - vector_match: None, - }; - let shared = share_common_sub_dags(vec![("q", root)]); - let [(_, root)] = shared.as_slice() else { - panic!("expected 1 root"); - }; - assert_eq!( - dag_node_count(root), - 3, - "the shared branch's 2 nodes must be counted once, not once per \ - occurrence — got {} for {root:?}", - dag_node_count(root) - ); - } - - #[test] - fn dag_node_count_deduplicates_across_two_workload_roots() { - // Two workload roots sharing one Aggregate after - // `share_common_sub_dags` (the `duplicate_workload_queries_...` - // shape from `crates/integration-tests/tests/cse.rs`, built - // directly here): each root's own `dag_node_count` must report the - // shared sub-DAG's real size once, not double-count anything — - // there's nothing *to* double-count from a single root's own count - // in this case (no root references the shared node twice), so this - // pins the simpler, more common case that a per-candidate cost - // proxy (`CseCandidate::sub-DAG` in `asap-aware-mapping`) actually - // exercises: counting one occurrence's own reachable DAG size. - let a = quantile_agg(vec![1], Some(2), 0.5); - let b = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!(Rc::ptr_eq(ra, rb), "fixture sanity: the two roots merged"); - assert_eq!(dag_node_count(ra), 2); - assert_eq!(dag_node_count(rb), 2); - } - - #[test] - fn dedup_gates_sharing_the_same_as_aggregate() { - // `Dedup { cols }` adds `cols` as a unique key — so two identical - // `Dedup` sub-DAGs over a keyed column *do* merge, exercising the - // legality gate on a non-`Aggregate` node. - let dedup = |cols: Vec| QueryExpr::Dedup { - cols, - child: Rc::new(scan()), - }; - let a = dedup(vec![1]); - let b = dedup(vec![1]); - let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!( - Rc::ptr_eq(ra, rb), - "Dedup on the same cols has a provable unique key and should merge" - ); - } - - #[test] - fn group_keys_gate_still_prevented_when_partition_by_without_used() { - // Sanity on the module's advertised precedent: a `without(...)` - // grouping stays open (no unique key) even though `by` is - // non-empty-shaped structurally, so two identical `without` groups - // do not merge under the same gate that blocks the ungrouped case. - let without_agg = || QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::without(vec![0])), - measures: vec![AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan()), - }; - let a = without_agg(); - let b = without_agg(); - assert!(!a.output_schema().unwrap().has_unique_key()); - let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); - let [(_, ra), (_, rb)] = shared.as_slice() else { - panic!("expected 2 roots"); - }; - assert!(!Rc::ptr_eq(ra, rb)); - } -} diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs deleted file mode 100644 index f434eb154..000000000 --- a/crates/types/src/pre_asap/mod.rs +++ /dev/null @@ -1,64 +0,0 @@ -//! The canonical pre-ASAP intent algebra IR. -//! -//! - [`query_expr`] — the canonical, language- and deployment-independent -//! intent algebra: one recursive [`QueryExpr`] DAG (relational operators -//! *and* scalar expression shapes both, since issue #205) + [`AggIntent`], -//! generic over the column-reference state (positional [`ColumnId`] once -//! bound, name-based [`ColumnRef`] before). -//! - [`agg_intent`] — the aggregation-intent vocabulary. -//! - [`expr_ir`] — the [`ColumnRef`] column-reference type and the scalar -//! operator/literal vocabulary ([`ScalarValue`], [`CompareOpKind`], [`ArithmeticOpKind`]) -//! [`QueryExpr`]'s scalar variants are built from. -//! - [`schema`] — the per-edge [`Schema`] every node carries. -//! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` -//! into a positional `ColumnId` against an in-scope [`Schema`]. -//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] DAG to -//! canonical [`ResolvedQueryExpr`] (issue #179): both front ends -//! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` -//! directly during their own `interpret` step and call -//! [`resolve_root`] on the result — there is no separate per-language -//! relational DAG or converter anymore. -//! - [`canonicalize`] — post-lowering structural normalization of [`QueryExpr`] -//! (issue #34), run by [`resolve_root`]. -//! - [`cse`] — workload-level structural common-subexpression elimination -//! over an already-`resolve_root`'d DAG (issue #212, #222, #223), run -//! *after* `resolve_root` / `canonicalize` and *before* implementation -//! (`asap_aware_mapping::replacement`). -//! -//! Formerly the separate `asap-l2` crate; folded in here since -//! `schema_resolver`/`column_resolution`/`canonicalize`/`resolve` have no -//! front-end-specific logic — they operate directly on this crate's own -//! `QueryExpr`. - -pub mod agg_intent; -pub mod canonicalize; -pub mod column_resolution; -pub mod cse; -pub mod expr_ir; -pub mod query_expr; -pub mod resolve; -pub mod scalar_type_rules; -pub mod schema; -pub mod schema_resolver; - -pub use agg_intent::{ - agg_accuracy, agg_is_exact, agg_is_mergeable, default_cardinality, default_quantile, AggIntent, - MathFunc, TimeFunc, -}; -pub use canonicalize::canonicalize; -pub use column_resolution::{ - output_schema_for_aggregate, resolve_column_ref, resolve_column_refs, resolve_expr, - ResolveError, -}; -pub use cse::share_common_sub_dags; -pub use expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; -pub use query_expr::{ - aggregate_output_schema, any_measure_filtered, AtModifier, BinaryOpKind, ColState, DataModel, - GroupKeys, GroupSide, InfoMatcher, JoinKind, Predicate, ProjectItem, PromQLVectorSetOpKind, - QueryExpr, QueryExprError, Reduction, RelationalSetOpKind, ResolvedQueryExpr, SampleKind, - SortKey, Source, TimeShift, UnresolvedQueryExpr, VectorGrouping, VectorMatch, VectorMatchKind, - WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, -}; -pub use resolve::{resolve_root, ResolveDAGError}; -pub use schema::{ColumnId, DataType, Field, FieldDataType, Schema}; -pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs deleted file mode 100644 index e389f84a2..000000000 --- a/crates/types/src/pre_asap/query_expr.rs +++ /dev/null @@ -1,2774 +0,0 @@ -//! The canonical pre-ASAP intent algebra IR. -//! -//! Language- and deployment-independent. `Rc`-owned DAG — a child field is -//! `Rc>` rather than `Box>` so a structurally -//! identical sub-expression can be shared (the same `Rc`) across more than -//! one parent, within one query or across a `QueryWorkload` batch, instead of -//! being duplicated. Nothing in this module produces that sharing on its -//! own — construction still allocates a fresh `Rc` per node, the same shape -//! as the old `Box` DAG — a separate CSE pass is what turns two -//! independently constructed, structurally-equal sub-DAGs into two -//! references to one `Rc` (issue #212, #222). Field identity is -//! **positional** (`Aggregate.reduction: Reduction`, wrapping `GroupKeys` -//! for the grouped case), resolved by the [`SchemaResolver`](super::schema_resolver) against -//! the self-contained [`Schema`] carried on each `Scan`. - -use std::rc::Rc; -use std::time::Duration; - -use serde::{Deserialize, Serialize}; -use thiserror::Error; - -use super::agg_intent::AggIntent; -use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; -use super::schema::{ColumnId, DataType, Field, FieldDataType, Schema}; - -/// The column-reference resolution state a [`QueryExpr`] DAG carries — -/// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always -/// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every -/// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] -/// before binding. The only place the two states differ in *shape* rather -/// than just in which type fills `C` is [`QueryExpr::Scan`]'s `schema` field: -/// a bound DAG's binding schema is always known (the SchemaResolver is total, so -/// [`ScanSchema`](Self::ScanSchema) `= Schema`); an unresolved front-end -/// `Scan` knows its schema only when the front end already has it without -/// binding — a SQL leaf, catalog-backed (`Some`) — `None` (PromQL) defers to -/// the SchemaResolver, so `ScanSchema = Option`. -pub trait ColState: - Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de> -{ - /// What [`QueryExpr::Scan`]'s `schema` field holds for a DAG in this state. - type ScanSchema: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de>; -} - -impl ColState for ColumnId { - type ScanSchema = Schema; -} - -impl ColState for ColumnRef { - type ScanSchema = Option; -} - -/// Errors from schema derivation over a canonical DAG. -#[derive(Debug, Error)] -pub enum QueryExprError { - #[error("invalid scalar function signature: {0}")] - InvalidScalarSignature(String), - #[error("by-column id {0} out of range (input has {1} columns)")] - InvalidGroupByColumn(ColumnId, usize), - #[error("Concat requires at least one child")] - EmptyConcat, - /// [`QueryExpr::output_schema`] called on (or reached, while recursing, a - /// child that is) one of the scalar variants (issue #205) — those have no - /// independent row schema of their own; a scalar expression's *type* only - /// makes sense against the schema it's embedded in (see `infer_expr_type`, - /// used by `Project`'s own `output_schema` arm instead). - #[error("a scalar expression has no row schema of its own")] - ScalarHasNoRowSchema, - #[error("invalid per-series sample column: {0}")] - InvalidSampleColumn(String), -} - -// ── Leaf / supporting types ─────────────────────────────────────────────────── - -/// Positional grouping keys, shared by every "operate per group" operator: -/// `Aggregate.by` (reduce per group), `Sort.partition_by` (rank per group — -/// including generic `topk`/`bottomk`), and `SQLWindowFunc.partition_by` (window -/// per group). One spelling so grouping has a single home to evolve. Empty -/// (and `by`) = no grouping (a global operation). -/// -/// Heavy-hitter `AggIntent::TopK` carries its grouping here too, via the -/// enclosing `Aggregate.by` (issue #13) — so reduce, rank, and window groupings -/// all share this one type. -/// -/// ## `by` vs `without` (issue #39) -/// -/// The stored [`keys`](Self::keys) are **kept** labels for `by(...)` and -/// **excluded** labels for `without(...)`. PromQL's `without(labels)` groups by -/// every label *except* those listed; the complement can't be enumerated at -/// lowering time under an open (usage-derived) schema, so it is deferred to the -/// runtime — the excluded positions are stored, the kept set stays open. Only -/// `Aggregate` ever produces the `without` form; `Sort` / `SQLWindowFunc` / -/// `PromqlSeriesSample` groupings are always `by`. -/// -/// Serialises as a bare array for the (overwhelmingly common) `by` case — -/// wire-compatible with the `Vec` this field held before — and as -/// `{"without": [...]}` for the exclusion case. -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub struct GroupKeys { - keys: Vec, - without: bool, -} - -// Not `#[derive(Default)]`: derive would add a `C: Default` bound, but an -// empty key set needs nothing from `C` — `ColumnRef` has no meaningful -// default anyway. -impl Default for GroupKeys { - fn default() -> Self { - Self { - keys: Vec::new(), - without: false, - } - } -} - -impl GroupKeys { - /// An empty key set — a global (ungrouped) operation. - pub fn none() -> Self { - Self::default() - } - /// `by(keys)` — group by exactly these columns. - pub fn by(keys: Vec) -> Self { - Self { - keys, - without: false, - } - } - /// `without(keys)` — group by every label *except* these (issue #39). The - /// kept set is runtime-resolved; only the excluded positions are stored. - pub fn without(keys: Vec) -> Self { - Self { - keys, - without: true, - } - } - /// Whether this is a `without(...)` exclusion grouping. - pub fn is_without(&self) -> bool { - self.without - } - /// The named keys — kept labels for `by`, excluded labels for `without`. - pub fn keys(&self) -> &[C] { - &self.keys - } -} - -impl std::ops::Deref for GroupKeys { - type Target = [C]; - fn deref(&self) -> &Self::Target { - &self.keys - } -} - -impl From> for GroupKeys { - fn from(keys: Vec) -> Self { - Self::by(keys) - } -} - -impl FromIterator for GroupKeys { - fn from_iter>(iter: I) -> Self { - Self::by(iter.into_iter().collect()) - } -} - -impl<'a, C> IntoIterator for &'a GroupKeys { - type Item = &'a C; - type IntoIter = std::slice::Iter<'a, C>; - fn into_iter(self) -> Self::IntoIter { - self.keys.iter() - } -} - -/// Compare directly against a `Vec` so call sites and tests can keep -/// writing `keys == vec![..]` / `assert_eq!(keys, &vec![..])`. A `without` -/// grouping never equals a bare `by` list. -impl PartialEq> for GroupKeys { - fn eq(&self, other: &Vec) -> bool { - !self.without && &self.keys == other - } -} - -/// (De)serialise as a bare array for `by`, or `{"without": [...]}` for the -/// exclusion form — keeping the `by` wire format identical to the old newtype. -/// Borrowed for `Serialize` (no `C: Clone` needed to write one out), owned for -/// `Deserialize` (there's nothing to borrow from). -#[derive(Serialize)] -#[serde(untagged)] -enum GroupKeysReprRef<'a, C> { - By(&'a [C]), - Without { without: &'a [C] }, -} - -#[derive(Deserialize)] -#[serde(untagged)] -enum GroupKeysRepr { - By(Vec), - Without { without: Vec }, -} - -impl Serialize for GroupKeys { - fn serialize(&self, serializer: S) -> Result { - if self.without { - GroupKeysReprRef::Without { - without: self.keys.as_slice(), - } - .serialize(serializer) - } else { - GroupKeysReprRef::By(self.keys.as_slice()).serialize(serializer) - } - } -} - -impl<'de, C: Deserialize<'de>> Deserialize<'de> for GroupKeys { - fn deserialize>(deserializer: D) -> Result { - Ok(match GroupKeysRepr::deserialize(deserializer)? { - GroupKeysRepr::By(keys) => Self::by(keys), - GroupKeysRepr::Without { without } => Self::without(without), - }) - } -} - -/// Which data model a `Source` / `AggIntent` operates over. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum DataModel { - TimeSeries, - Tabular, - Any, -} - -/// The leaf data source of a `Scan`. The schema itself rides on the -/// `Scan.schema` field (SchemaResolver-built); `Source` carries only the leaf's -/// identity. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum Source { - /// Time-series leaf — PromQL / DC lifecycle. Produces `(ts, value, *labels)`. - TimeSeries { metric: String }, - /// Tabular leaf — asap-fusion / future OLAP. Columns ride on `Scan.schema`. - Table { table_ref: String }, -} - -impl Source { - pub fn data_model(&self) -> DataModel { - match self { - Source::TimeSeries { .. } => DataModel::TimeSeries, - Source::Table { .. } => DataModel::Tabular, - } - } -} - -/// Operator on the query-level `BinaryOp` node. Reuses the scalar IR's -/// [`ArithmeticOpKind`] / [`CompareOpKind`] so every arithmetic/comparison -/// operator has exactly one representation (and one `Display`) across the IR. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum BinaryOpKind { - /// Arithmetic — `Add/Sub/Mul/Div/Mod` (shared with `QueryExpr::Arithmetic`). - Arithmetic(ArithmeticOpKind), - /// Comparison — `Eq/Ne/Lt/Le/Gt/Ge` + `Like/ILike/Regex` family (shared - /// with `QueryExpr::Compare`). PromQL keeps the matched series whose - /// comparison holds. - Compare(CompareOpKind), - /// PromQL comparison with the `bool` modifier: every matched series - /// yields 1 or 0 and loses its metric name. A separate variant, not a - /// flag, because only comparisons take `bool`. - CompareBool(CompareOpKind), - /// PromQL vector-set operation. - Set(PromQLVectorSetOpKind), -} - -impl std::fmt::Display for BinaryOpKind { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - BinaryOpKind::Arithmetic(op) => write!(f, "{op}"), - BinaryOpKind::Compare(op) => write!(f, "{op}"), - BinaryOpKind::CompareBool(op) => write!(f, "{op} bool"), - BinaryOpKind::Set(PromQLVectorSetOpKind::And) => f.write_str("AND"), - BinaryOpKind::Set(PromQLVectorSetOpKind::Or) => f.write_str("OR"), - BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) => f.write_str("unless"), - } - } -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum JoinKind { - Inner, - Left, - Right, - Full, - Cross, - /// Left semi-join — each left row that has **at least one** match, once. - /// `WHERE c IN (SELECT …)` / `WHERE EXISTS (…)` (issue #111). - /// - /// Output schema is the **left's alone**; the right side is a filter, not a - /// source of columns. The join predicate still resolves against the - /// concatenated `left ++ right` schema — its scope is deliberately wider - /// than the node's output. - Semi, - /// Left anti-join — each left row with **no** match. `WHERE NOT EXISTS (…)`. - /// Same schema rule as [`JoinKind::Semi`]. - /// - /// Note this is *not* `NOT IN (SELECT …)`: under SQL's three-valued logic a - /// NULL on the right makes `NOT IN` yield no rows at all, where an anti-join - /// yields every left row. The SQL front end rejects `NOT IN (subquery)` - /// rather than lower it here. - Anti, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum RelationalSetOpKind { - Union, - Intersect, - Except, -} - -/// PromQL vector-set operator used by [`BinaryOpKind::Set`]. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum PromQLVectorSetOpKind { - And, - Or, - Unless, -} - -/// SQL analytic window function (`fn(...) OVER (…)`). Distinct from a streaming -/// time `Window`: this is an analytic frame over already-materialised rows. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFuncKind { - RowNumber, - Rank, - DenseRank, - Lag, - Lead, - /// ClickHouse `lagInFrame`/`leadInFrame`: unlike [`Lag`](Self::Lag)/[`Lead`](Self::Lead), - /// these respect the window frame bounds (NULL/default past the frame edge) - /// rather than reaching arbitrarily far back/forward. Kept as distinct - /// variants so the frame clause is never silently discarded by conflating - /// them with `Lag`/`Lead` (#267). `WindowFuncKind` still has no frame - /// representation, so today these lower and behave exactly like - /// `Lag`/`Lead` — the tag is correct, the frame-respecting behavior isn't - /// implemented yet. See #231 for modeling window frames properly. - LagInFrame, - LeadInFrame, - FirstValue, - LastValue, - /// `NTH_VALUE(expr, n)` — `n` is resolved from the (literal) 2nd argument. - NthValue(Option), - Sum, - Avg, - Count, - Min, - Max, -} - -/// A window's frame-spec (`ROWS`/`RANGE BETWEEN … AND …`) — which rows around -/// the current one an analytic window function reads. `GROUPS` is rejected at -/// lowering time (issue #268): every SQL corpus in this repo uses only `ROWS`, -/// and nothing downstream interprets frame semantics yet, so it isn't worth -/// modelling untested. -/// -/// Meaningless (but harmless) on the rank-only and navigation functions -/// (`ROW_NUMBER`/`RANK`/`DENSE_RANK`/`LAG`/`LEAD`), which ignore the frame per -/// SQL semantics — DataFusion still attaches one, stored here verbatim. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct WindowFrame { - pub units: WindowFrameUnits, - pub start_bound: WindowFrameBound, - pub end_bound: WindowFrameBound, -} - -/// A finite window-frame displacement. Intervals are normalized to Arrow's -/// month/day/nanosecond representation so SQL `RANGE INTERVAL ...` bounds -/// survive lowering without leaking DataFusion types into the canonical IR. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFrameOffset { - Scalar(ScalarValue), - Interval { - months: i32, - days: i32, - nanoseconds: i64, - }, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFrameUnits { - /// Boundaries count physical rows: `ROWS BETWEEN 2 PRECEDING AND CURRENT ROW`. - Rows, - /// Boundaries count by value-distance on the (single) `ORDER BY` column: - /// `RANGE BETWEEN INTERVAL '1' HOUR PRECEDING AND CURRENT ROW`. - Range, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum WindowFrameBound { - /// `UNBOUNDED PRECEDING` is - /// `Preceding(WindowFrameOffset::Scalar(ScalarValue::Null))`. - Preceding(WindowFrameOffset), - CurrentRow, - /// `UNBOUNDED FOLLOWING` is - /// `Following(WindowFrameOffset::Scalar(ScalarValue::Null))`. - Following(WindowFrameOffset), -} - -/// A symbolic label matcher on the **info metric** side of an -/// [`QueryExpr::PromqlInfoEnrich`] (issue #84). Unlike a `Scan` predicate it is not -/// resolved positionally — it references the info metric's labels (`__name__` -/// picks the metric, the rest constrain data labels), which aren't in the input -/// vector's schema; the post-ASAP realization pass applies it against the info metric. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct InfoMatcher { - pub label: String, - /// One of `Eq` / `Ne` / `Regex` / `NotRegex` (PromQL `=`/`!=`/`=~`/`!~`). - pub op: CompareOpKind, - pub value: String, -} - -/// Series-sampling selection mode (PromQL `limitk` / `limit_ratio`, issue #86). -/// A [`QueryExpr::PromqlSeriesSample`] keeps a *subset of whole series*, unchanged — it does -/// not rank or reduce, so it is distinct from `TopK` and from `Sort → Limit`. -#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum SampleKind { - /// `limitk(k, v)` — up to `k` series per group. Which series survive is - /// deterministic across evaluations but otherwise unspecified (no ordering). - LimitK(usize), - /// `limit_ratio(r, v)` — a deterministic `r`-fraction of series per group. - /// `r ∈ [-1, 1]`; a negative `r` selects the complementary fraction. - LimitRatio(f64), -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct SortKey { - pub expr: QueryExpr, - pub ascending: bool, - pub nulls_first: bool, -} - -/// PromQL vector-match modifier (`on`/`ignoring` + `group_left`/`group_right`). -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct VectorMatch { - pub kind: VectorMatchKind, - pub labels: Vec, - pub grouping: Option, -} - -/// PromQL `@` modifier — pins a selector's evaluation time to an anchor instead -/// of the query evaluation time (issue #40). -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -pub enum AtModifier { - /// `@ start()` — the query range's start instant. - Start, - /// `@ end()` — the query range's end instant. - End, - /// `@ ` — an absolute instant, milliseconds since the Unix epoch (may be - /// negative). PromQL writes the timestamp in seconds; the front end scales it. - Timestamp(i64), -} - -/// PromQL per-selector **time-shift** modifiers — `offset` and `@` (issue #40). -/// Neither changes a selector's *schema*; both move *when* it is evaluated, so -/// the shift is a pass-through wrapper ([`QueryExpr::TimeShift`]) over the -/// selector rather than a new leaf shape. The runtime resolves the anchor and -/// applies the offset. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] -pub struct TimeShift { - /// `offset ` as signed milliseconds — a positive value shifts the - /// lookback *back* in time (`offset 5m`), a negative value shifts it - /// *forward* (`offset -5m`). `0` = no offset. - pub offset_ms: i64, - /// `@` anchor; `None` = evaluate at the query time. - pub at: Option, -} - -impl TimeShift { - /// Whether this shift is the identity (no `offset`, no `@`) — the state of - /// every selector that carries neither modifier. - pub fn is_identity(&self) -> bool { - self.offset_ms == 0 && self.at.is_none() - } -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum VectorMatchKind { - On, - Ignoring, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct VectorGrouping { - pub side: GroupSide, - pub labels: Vec, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub enum GroupSide { - Left, - Right, -} - -/// A row-level filter predicate (WHERE clause / PromQL label matcher). -/// Boxed: `Predicate` sits directly (not behind a `Vec`) in -/// `Filter.pred`/`Join.pred`/`Aggregate.having`, and `QueryExpr` is -/// self-recursive without further indirection once the scalar variants are -/// part of it — the box is what makes the recursive type's size finite there. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct Predicate(pub Rc>); - -/// Whether any entry of an `Aggregate.filters` vector is set — the shape -/// no binding rule accepts yet (issue #466): a filtered measure stays -/// `KeepPreAsap`, and heavy-hitter promotion skips it. -pub fn any_measure_filtered(filters: &[Option>]) -> bool { - filters.iter().any(Option::is_some) -} - -/// One item in a SELECT projection list. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ProjectItem { - pub alias: Option, - pub expr: QueryExpr, -} - -// ── Intent algebra IR ──────────────────────────────────────────────────────── - -/// What kind of computation an `Aggregate` node performs — orthogonal to -/// *which* columns it groups by (that's still [`GroupKeys`], inside -/// `Reduce`). Explicit, decided once by whichever pass constructs the node -/// (structural, at front-end lowering time), rather than inferred downstream from -/// whether a grouping-key list happens to be empty or from a neighboring -/// node's shape. See design proposal #165. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub enum Reduction { - /// Collapses input rows via `by` — `by`/`without` semantics are exactly - /// [`GroupKeys`]'s. May still collapse every row into one (an empty, - /// non-`without` `by`) — that's a genuine reduction with zero grouping - /// columns, not "no grouping concept." - Reduce(GroupKeys), - /// No grouping concept at all: preserves one output row per input - /// entity (e.g. a per-series windowed computation with no `by(...)` - /// clause to begin with, because there's no aggregation operator here - /// for such a clause to attach to). Never merges across entities, and - /// never collapses an entity's own row structure (e.g. a time axis) — - /// unlike `Reduce(GroupKeys::without(vec![]))` ("group by every - /// label"), which is still a genuine reduction and does collapse it. - PerEntity, -} - -impl Reduction { - /// Shorthand for the common case — group by these (possibly empty) - /// keys, kept rather than excluded. - pub fn by(keys: Vec) -> Self { - Self::Reduce(GroupKeys::by(keys)) - } - - /// The grouping keys, if this is a genuine reduction — `None` for - /// `PerEntity`, which has no grouping-keys concept to report. - pub fn group_keys(&self) -> Option<&GroupKeys> { - match self { - Self::Reduce(by) => Some(by), - Self::PerEntity => None, - } - } - - /// The grouping keys, panicking if this is `PerEntity` — for call sites - /// (tests, mostly) that already know, from the shape they built or are - /// asserting on, that this must be a genuine reduction. Prefer - /// [`group_keys`](Self::group_keys) wherever the caller can't assume that. - pub fn expect_reduce(&self) -> &GroupKeys { - match self { - Self::Reduce(by) => by, - Self::PerEntity => panic!("expected Reduction::Reduce, got PerEntity"), - } - } -} - -/// A caller-proven compound unique key for a [`QueryExpr::Concat`] (issue -/// #228) — built only via [`QueryExpr::concat_with_discriminator`] / -/// [`ConcatDiscriminatorKey::new`], never by naming `discriminator` directly -/// in a struct literal (both fields are private): from *other Rust code*, -/// the only way to end up with one of these is to hand over a specific -/// column as the discriminator, by name, at the call site. -/// -/// Caveat: this is a Rust-API-level guarantee, not a data-level one. The -/// derived `Deserialize` impl below builds a `ConcatDiscriminatorKey` -/// directly from field values, bypassing `new()`. Deserialization is therefore -/// equivalent to a caller supplying the assertion directly; it does not prove -/// either fact below. An external boundary accepting `QueryExpr` data must -/// reject this field or validate both obligations before treating it as -/// uniqueness evidence. -/// -/// # Soundness -/// -/// `Concat`'s default (see its own doc) is to drop `unique_keys` -/// unconditionally, because a key unique **within** one branch is not unique -/// **across** the concatenation unless the branches' value sets for that key -/// are provably disjoint — nothing about matching schemas or matching -/// per-branch keys establishes that on its own. Two different branches can -/// trivially emit the same `inner_key` value (e.g. two PromQL -/// `histogram_quantiles` branches keyed on `(host, le)` can both produce a -/// `(host, le)` pair for different φ). -/// -/// Prepending `discriminator` restores a compound key only when two facts -/// hold: `inner_key` uniquely identifies rows **within every branch**, and -/// `discriminator`'s value is **guaranteed to differ between branches** — a -/// literal the producer just tagged the branch with (PromQL φ riding along via -/// [`QueryExpr::PromqlRelabel`], a Postgres-style synthetic `GROUPING()` id -/// for `ROLLUP`/`CUBE`, …), never something inferred structurally from the -/// branches' own data — then `discriminator` alone partitions rows into -/// disjoint sets independent of what the branches actually contain, so -/// `(discriminator, inner_key)` is sound even when otherwise-identical -/// `inner_key` values occur in different branches. Neither fact is verified -/// here; both are part of the caller-proven claim. -/// -/// This is a **caller-proven claim, not something `Concat` can verify**: -/// nothing stops a caller from asserting a discriminator that in fact -/// repeats across branches, in which case the resulting `unique_keys` claim -/// is simply wrong — `output_schema` trusts it without checking. The -/// obligation is on the constructor call site, exactly as it is on -/// [`QueryExpr::Dedup`]'s `cols` or any other unverified `unique_keys` -/// producer in this module. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub struct ConcatDiscriminatorKey { - discriminator: C, - inner_key: Vec, -} - -impl ConcatDiscriminatorKey { - /// The only constructor — `discriminator` must be named explicitly by - /// the caller. See the type's doc for the soundness obligation this - /// puts on that caller. - pub fn new(discriminator: C, inner_key: Vec) -> Self { - Self { - discriminator, - inner_key, - } - } - - pub fn discriminator(&self) -> &C { - &self.discriminator - } - - pub fn inner_key(&self) -> &[C] { - &self.inner_key - } -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] -pub enum QueryExpr { - /// Outermost leaf. `schema` is the **binding schema** — the resolved column - /// set every positional `ColumnId` in the DAG indexes into, *not* a full - /// description of the runtime row — once bound (`schema: Schema`, always - /// present: the [`SchemaResolver`](super::schema_resolver) is total). Before binding, a - /// front-end-emitted `Scan` (`C = ColumnRef`) knows it only when the front - /// end already has it without binding — a catalog-backed SQL leaf — `None` - /// (PromQL) defers to the SchemaResolver; see [`ColState::ScanSchema`]. Complete - /// when catalog-backed (SQL); for schemaless PromQL the bound schema is - /// usage-derived (the `(ts, value)` floor + the labels the query - /// references), since a metric's label set is open and known only at - /// runtime. That distinction is carried explicitly by - /// [`Schema::closed`](super::schema::Schema::closed) (SQL leaf → `true`, - /// PromQL leaf → `false`). `predicates` are leaf-level row filters (PromQL - /// label matchers, pushed-down `WHERE` conjuncts). - Scan { - source: Source, - #[serde(default)] - predicates: Vec>, - schema: C::ScanSchema, - }, - /// A scalar sub-expression sitting in an **operator-DAG position** — a - /// [`BinaryOp`](Self::BinaryOp) operand for ` op ` - /// thresholds / unit conversions (#35), a - /// [`PromqlVectorFromScalar`](Self::PromqlVectorFromScalar) child, or a - /// whole query's root (a bare PromQL scalar query, e.g. `5`). - /// - /// Formerly its own leaf variant, `PromqlScalar(f64)`. Issue #220: that - /// variant held exactly the same value [`Literal`](Self::Literal) does - /// (every PromQL scalar is `f64`), duplicating it for no reason but - /// *which DAG position* it was allowed to appear in. This wrapper - /// carries that position instead of the value — the inner node is an - /// ordinary scalar sub-language expression (in practice always - /// `Literal(ScalarValue::Float64(_))`, since a front end only ever - /// constructs this fully constant-folded — see - /// [`promql_scalar`](Self::promql_scalar)) — and is what `output_schema`, - /// `canonicalize`, and `resolve` now key off to tell "this operand has - /// its own row schema" from "this is a nested scalar leaf with none," - /// in place of the old `PromqlScalar` vs. `Literal` variant tag. - PromqlScalarBridge(Rc>), - - /// The query **evaluation timestamp** as Unix seconds, exposed by PromQL - /// `time()`. This is not inherently the current wall-clock time: its value - /// is the instant or range-step at which the expression is evaluated. It - /// is also the implicit input of no-argument calendar functions. Issue #46. - EvalTimestamp, - - /// The SQL statement evaluation time (`NOW()` / `CURRENT_TIMESTAMP`) as - /// a SQL [`DataType::Timestamp`]. Kept distinct from [`EvalTimestamp`], - /// whose PromQL `time()` contract is Unix seconds as `Float64`. - CurrentTimestamp, - - /// PromQL `vector(s)` — the scalar→instant-vector bridge. Promotes a - /// scalar-typed child to a single label-less series carrying the scalar's - /// value at every step. Lets a scalar participate where a vector is required - /// (`up or vector(0)` dead-man's-switch). Issue #48. - PromqlVectorFromScalar(Rc>), - - /// PromQL `scalar(v)` — the instant-vector→scalar bridge. Collapses a - /// single-element vector to its value (NaN at runtime if the input is not - /// exactly one series). Lets a vector feed a scalar position (`vector` / - /// aggregation `k` args, thresholds). Issue #48. - PromqlScalarFromVector(Rc>), - - /// ρ — a per-series **label rewrite** (PromQL `label_replace` / - /// `label_join`). Every input row passes through unchanged except for the - /// destination label `dst`, whose new value is computed by `value` — a - /// scalar expression over the child's (source) label columns: - /// `label_replace` → a `label_replace(src, regex, replacement)` function - /// call (regex capture-expansion), `label_join` → a `label_join(sep, srcs…)` - /// concatenation. Sample values and the time axis are untouched. Issue #50. - PromqlRelabel { - /// The label written by this rewrite (PromQL `dst_label`). - dst: String, - value: Rc>, - child: Rc>, - }, - - /// PromQL `info(v, [selector])` — left-join **label enrichment** (#84). Each - /// series in `child` is enriched with labels from the matching info metric(s) - /// (`target_info` by default; `selector`'s `__name__` matchers pick the - /// metric(s), the rest constrain the data labels), joined on their shared - /// identifying labels. Those join keys are the info metric's identifying - /// labels — runtime/metadata-resolved, since an open PromQL schema can't - /// enumerate them — so they are NOT carried here; the post-ASAP realization pass - /// resolves them from the info metric's schema. The output keeps - /// `child`'s (open) schema: the - /// grafted labels appear at runtime. - PromqlInfoEnrich { - #[serde(default)] - selector: Vec, - child: Rc>, - }, - - /// Series-sampling **selection** — PromQL `limitk` / `limit_ratio` (#86). - /// Keeps a subset of whole series per `by` group (empty = global), passing - /// each surviving series through unchanged. Not a ranking (`TopK`) and not a - /// reduction: the output schema equals the child's. - PromqlSeriesSample { - #[serde(default)] - by: GroupKeys, - kind: SampleKind, - child: Rc>, - }, - - /// σ — row-level filter. Output schema = child schema. - Filter { - pred: Predicate, - child: Rc>, - }, - /// π — column projection. - Project { - cols: Vec>, - /// Re-qualifies every output column with this table alias (a derived - /// table / inline view). `None` for an ordinary SELECT list. - #[serde(default)] - qualifier: Option, - child: Rc>, - }, - - /// γ + α — GROUP BY (positional) + aggregate intents. - Aggregate { - reduction: Reduction, - measures: Vec>, - /// Output column names parallel to `measures`. A non-empty entry overrides - /// the synthetic intent-keyed name — SQL threads DataFusion's generated - /// name (e.g. `"sum(metrics.bytes)"`) here so an enclosing `Project` - /// resolves the aggregate output by the name it references. An empty - /// entry (or empty vec) falls back to `AggIntent::output_column`'s name - /// (PromQL's convention). - #[serde(default)] - output_names: Vec, - /// Per-measure row predicates, parallel to `measures` — SQL - /// `FILTER (WHERE …)` semantics (issue #466): only rows where - /// `filters[i]` is `TRUE` update `measures[i]`; groups are still - /// formed from every row. Positional against `child`'s output - /// schema, like `Filter.pred` — not against this node's output like - /// `having`. `None` (or an entry past the end of a shorter vec) is - /// an unfiltered measure, so an empty vec is the pre-#466 shape. - #[serde(default)] - filters: Vec>>, - #[serde(default)] - having: Option>, - child: Rc>, - }, - - /// δ — SQL `DISTINCT` / row deduplication. Positional like every other - /// column reference here; empty = dedup on all columns (`SELECT DISTINCT *`). - Dedup { - cols: Vec, - child: Rc>, - }, - /// ⊕ — exact, n-ary `UNION ALL` of independent branches. Rows are - /// concatenated, never deduplicated; SQL's `UNION`/`INTERSECT`/`EXCEPT` are - /// [`QueryExpr::SetOp`], not this. - /// - /// Used for the branches of one query that a single `Aggregate` cannot - /// express — PromQL `histogram_quantiles` (one branch per φ, issue #109) and - /// SQL `ROLLUP`/`CUBE`/`GROUPING SETS` (one branch per grouping level, issue - /// #118) — as well as for sharded / fan-in plans. - /// - /// **The branches must be union-compatible; nothing here enforces it.** The - /// output schema is the *first* child's, so branches that disagree on a - /// column name or type leave the merged schema silently misdescribing every - /// branch but one. A producer that cannot guarantee compatibility must - /// project the branches into a common shape first. - /// - /// A row may appear in several branches, so no branch's unique key survives - /// the union — `unique_keys` is dropped, as in `SetOp`. **Unless** the - /// constructor asserted `discriminator_unique_key` (issue #228, - /// [`QueryExpr::concat_with_discriminator`]): a caller-proven claim that - /// one column's value is guaranteed distinct per branch, which makes - /// `(discriminator, inner_key)` a sound compound unique key regardless of - /// whether `inner_key` alone repeats across branches. `None` — every - /// ordinary construction path, including the plain struct literal and - /// [`QueryExpr::concat`] — reproduces the old, unconditional-drop - /// behavior exactly; see [`ConcatDiscriminatorKey`]'s doc for the - /// soundness argument and the obligation this puts on whoever asserts it. - /// - /// Empty children is an error ([`QueryExprError::EmptyConcat`]), not an - /// empty relation: there would be no schema to derive. - Concat { - children: Vec>, - /// See the field-level doc above and [`ConcatDiscriminatorKey`]. - #[serde(default)] - discriminator_unique_key: Option>, - }, - - /// Logical join. Post-ASAP binding picks the physical alternative. - Join { - kind: JoinKind, - pred: Predicate, - left: Rc>, - right: Rc>, - }, - SetOp { - kind: RelationalSetOpKind, - all: bool, - left: Rc>, - right: Rc>, - }, - - /// Generic order-by for non-heavy-hitter cases. - /// - /// `partition_by` makes the ordering **per-group**: a non-empty set means - /// "rank within each `partition_by` group" — the semantics behind PromQL - /// `topk by (host) (…)` / SQL `… OVER (PARTITION BY host ORDER BY …)`. It is - /// row-preserving (schema pass-through) and is where the grouping of a - /// generic (non-heavy-hitter) ranking lives, so there is no separate - /// `Partition` node (issue #12: reducing GROUP BY → `Aggregate.by`, per-group - /// ranking → here, parallel sharding → a deployment's own physical - /// stage). Empty = a global order-by. - Sort { - keys: Vec>, - #[serde(default)] - partition_by: GroupKeys, - child: Rc>, - }, - Limit { - n: usize, - offset: usize, - child: Rc>, - }, - - /// PromQL sub-query (`[range:resolution]`). Logical pass-through. - PromqlSubquery { - range: Duration, - #[serde(default)] - resolution: Option, - child: Rc>, - }, - - /// Temporal range selection — "look back `range` of history for this - /// computation." Used for all range-vector functions: `rate`, `increase`, - /// `*_over_time`. The range is distinct from a row-level `Filter`. - /// - /// Structural marker: an `Aggregate` whose direct child is a `TimeRange` - /// is a *per-series* reduction (label-preserving); one whose child is a - /// plain `Scan` or another `Aggregate` is a *cross-series* reduction. - TimeRange { - range: Duration, - child: Rc>, - }, - - /// PromQL `offset` / `@` **time shift** on a selector (issue #40). A - /// pass-through wrapper: it moves *when* `child` is evaluated (the runtime - /// resolves the `@` anchor and applies the offset) but leaves its schema - /// unchanged. Wraps the shifted selector directly — `m offset 1h` → - /// `TimeShift { Scan }`; a ranged selector `m[5m] offset 1h` → - /// `TimeRange { 5m, TimeShift { Scan } }` (the range is taken at the shifted - /// time). A shifted subquery wraps the `PromqlSubquery`, moving its step - /// grid. Never carries the identity shift (the converter emits a bare - /// selector when neither modifier is present). - TimeShift { - shift: TimeShift, - child: Rc>, - }, - - /// SQL analytic window function: `func(args) OVER (PARTITION BY … ORDER BY … - /// ROWS/RANGE BETWEEN …)`. Output schema = child schema + one column named - /// `output_name` (the name the enclosing `Project` references). - SQLWindowFunc { - func: WindowFuncKind, - /// Operand expressions (`LAG(value)` → `[Column(value_id)]`); empty for - /// the rank-only functions (`ROW_NUMBER`/`RANK`/`DENSE_RANK`). - args: Vec>, - partition_by: GroupKeys, - order_by: Vec>, - /// `None` is accepted only for backward compatibility with serialized - /// pre-#268 IR, where the engine's implicit frame was not retained. - /// Newly lowered SQL always carries `Some` with DataFusion's resolved - /// concrete default or explicit frame. - #[serde(default)] - frame: Option, - /// The output column's name — DataFusion's window-expr field name, so a - /// `Project` above resolves it (cf. `Aggregate.output_names`). - output_name: String, - child: Rc>, - }, - - /// Arithmetic / comparison / boolean composition (PromQL binary ops). - BinaryOp { - op: BinaryOpKind, - lhs: Rc>, - rhs: Rc>, - #[serde(default)] - vector_match: Option, - }, - - // ── Scalar expression shapes (issue #205) ─────────────────────────── - // - // Formerly a separate, self-recursive `Expr` DAG, reachable from the - // operator variants above only through wrapper fields (`Predicate`, - // `ProjectItem`, `SortKey`). They're variants of this same DAG now — a - // scalar sub-expression is only ever reachable through one of those same - // wrapper positions (`Filter.pred`, `ProjectItem.expr`, `Aggregate.having`, - // `PromqlRelabel.value`, `SQLWindowFunc.args`, …), which is a *convention* this - // type no longer enforces at compile time the way the old, closed - // `Expr` variant set did — nothing stops constructing, say, a `Scan` - // where a `Compare`'s `left` operand belongs. `output_schema` and every - // scalar-position consumer (`resolve`, `canonicalize`, `infer_expr_type`) - // reject a non-scalar variant found there instead (a `QueryExprError` or - // an `unreachable!`, depending on the call site) — the accepted - // replacement, since the alternative (a marker-trait/sub-enum bound - // restricting which variants are constructible in a scalar position) adds - // real type-level machinery for a distinction every constructor already - // has to get right structurally anyway (a `Filter` is never built with an - // operator sub-DAG as its `pred`). - /// A column reference — unresolved [`ColumnRef`] (front-end-emitted, `C = - /// ColumnRef`) or positional [`ColumnId`] (once bound, `C = ColumnId`). - Column(C), - /// A constant literal value. - Literal(ScalarValue), - /// `left op right` — binary comparison. - Compare { - left: Rc>, - op: CompareOpKind, - right: Rc>, - }, - /// Flat conjunction (logical AND). An empty list is vacuously true. - BoolAnd(Vec>), - /// Flat disjunction (logical OR). An empty list is vacuously false. - BoolOr(Vec>), - /// Logical NOT. - Not(Rc>), - /// `expr IS NULL`. - IsNull(Rc>), - /// `expr IS NOT NULL`. - IsNotNull(Rc>), - /// `CAST(expr AS to)`; `try_cast` for SQL `TRY_CAST` (NULL on failure). - Cast { - expr: Rc>, - to: DataType, - try_cast: bool, - }, - /// `expr [NOT] IN (v1, v2, …)`. - InList { - expr: Rc>, - list: Vec>, - negated: bool, - }, - /// Scalar function call, e.g. `LOWER(col)`, `ABS(x)`. - FunctionCall { - name: String, - args: Vec>, - }, - /// Binary arithmetic: `left op right`. - Arithmetic { - op: ArithmeticOpKind, - left: Rc>, - right: Rc>, - }, - /// SQL `CASE` (both searched and simple forms). `operand` present for the - /// simple form (`CASE expr WHEN …`), absent for searched. - Case { - operand: Option>>, - branches: Vec<(QueryExpr, QueryExpr)>, - else_expr: Option>>, - }, -} - -impl QueryExpr { - /// Construct the [`PromqlScalarBridge`](Self::PromqlScalarBridge) leaf - /// for a bare PromQL numeric literal / folded constant scalar (issue - /// #220) — `Literal(ScalarValue::Float64(v))` at an operator-DAG - /// position. The one constructor every front end / test that used to - /// write `QueryExpr::PromqlScalar(v)` should use instead. - pub fn promql_scalar(v: f64) -> Self { - QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(v)))) - } - - /// Build an ordinary [`Concat`](Self::Concat) — the ordinary/default - /// construction path every call site should prefer over the bare struct - /// literal: `output_schema` drops `unique_keys` unconditionally, exactly - /// as before issue #228. Use - /// [`concat_with_discriminator`](Self::concat_with_discriminator) instead - /// when the caller can prove branch disjointness via a discriminator - /// column. - pub fn concat(children: Vec>) -> Self { - QueryExpr::Concat { - children, - discriminator_unique_key: None, - } - } - - /// Build a [`Concat`](Self::Concat) whose output schema carries the - /// caller-proven compound unique key `(discriminator, inner_key)` (issue - /// #228). See [`ConcatDiscriminatorKey`]'s doc for the soundness - /// argument and the obligation this puts on the caller — - /// `output_schema` trusts this claim without verifying it: nothing here - /// checks that `inner_key` is unique within every branch or that - /// `discriminator`'s value is distinct between branches. - pub fn concat_with_discriminator( - children: Vec>, - discriminator: C, - inner_key: Vec, - ) -> Self { - QueryExpr::Concat { - children, - discriminator_unique_key: Some(ConcatDiscriminatorKey::new(discriminator, inner_key)), - } - } - - /// The value of a [`PromqlScalarBridge`](Self::PromqlScalarBridge) leaf - /// wrapping a plain `Literal(ScalarValue::Float64(_))` — every one a - /// front end constructs today (see [`promql_scalar`](Self::promql_scalar)). - /// `None` for any other shape, including a `PromqlScalarBridge` wrapping - /// something else (not constructed today, but not precluded by the type). - pub fn as_promql_scalar(&self) -> Option { - match self { - QueryExpr::PromqlScalarBridge(inner) => match inner.as_ref() { - QueryExpr::Literal(ScalarValue::Float64(v)) => Some(*v), - _ => None, - }, - _ => None, - } - } - - /// If this expression is a `BoolAnd`, return its elements; otherwise a - /// single-element slice containing `self`. - pub fn conjuncts(&self) -> &[QueryExpr] { - match self { - QueryExpr::BoolAnd(v) => v.as_slice(), - _ => std::slice::from_ref(self), - } - } - - /// If this expression is a `BoolOr`, return its elements; otherwise a - /// single-element slice containing `self`. - pub fn disjuncts(&self) -> &[QueryExpr] { - match self { - QueryExpr::BoolOr(v) => v.as_slice(), - _ => std::slice::from_ref(self), - } - } - - /// Recursively collect every column reference in a **scalar** sub-DAG — - /// used by the [`SchemaResolver`](super::schema_resolver::SchemaResolver) to seed usage-derived - /// leaf schemas, and available to post-ASAP binding for column-lineage / - /// selectivity. - /// `self` must be one of the scalar variants (see the module doc on - /// [`QueryExpr`]'s scalar shapes) — every caller already only reaches - /// this through a scalar-typed position (`Predicate`, `ProjectItem.expr`, - /// …), so an operator variant here indicates a construction bug, not a - /// shape this needs to handle silently. - pub fn columns_referenced(&self) -> Vec<&C> { - match self { - QueryExpr::Column(c) => vec![c], - QueryExpr::Literal(_) => vec![], - QueryExpr::EvalTimestamp => vec![], - QueryExpr::CurrentTimestamp => vec![], - QueryExpr::Compare { left, right, .. } | QueryExpr::Arithmetic { left, right, .. } => { - let mut v = left.columns_referenced(); - v.extend(right.columns_referenced()); - v - } - QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { - parts.iter().flat_map(|e| e.columns_referenced()).collect() - } - QueryExpr::Not(e) | QueryExpr::IsNull(e) | QueryExpr::IsNotNull(e) => { - e.columns_referenced() - } - QueryExpr::Cast { expr, .. } => expr.columns_referenced(), - QueryExpr::InList { expr, list, .. } => { - let mut v = expr.columns_referenced(); - v.extend(list.iter().flat_map(|e| e.columns_referenced())); - v - } - QueryExpr::FunctionCall { args, .. } => { - args.iter().flat_map(|e| e.columns_referenced()).collect() - } - QueryExpr::Case { - operand, - branches, - else_expr, - } => { - let mut v = vec![]; - if let Some(op) = operand { - v.extend(op.columns_referenced()); - } - for (when, then) in branches { - v.extend(when.columns_referenced()); - v.extend(then.columns_referenced()); - } - if let Some(e) = else_expr { - v.extend(e.columns_referenced()); - } - v - } - other => unreachable!( - "columns_referenced called on a non-scalar QueryExpr variant: {other:?}" - ), - } - } -} - -/// The canonical, positional, resolved DAG — what the bare `QueryExpr` name -/// has always meant (the default `C = ColumnId`). Every existing consumer -/// keeps using `QueryExpr` unparameterized; this alias exists only to name -/// the resolved state explicitly at a use site that also wants to name -/// [`UnresolvedQueryExpr`] nearby. -pub type ResolvedQueryExpr = QueryExpr; - -/// The front-end-emitted, name-based, unresolved DAG — -/// `QueryExpr`: front ends construct this directly during their -/// own `interpret` step (issue #179), and the [`SchemaResolver`](super::schema_resolver) -/// resolves it into [`ResolvedQueryExpr`]. -pub type UnresolvedQueryExpr = QueryExpr; - -// `output_schema` needs a fully bound DAG — it reads `Scan.schema` as a plain -// `Schema` and resolves every scalar `Expr::Column` positionally — so it lives -// only on the resolved instantiation, not `impl QueryExpr`. -// Same reasoning as `AggIntent`'s `output_column`/`requires`/`is_per_series` -// (#205): a schema-shaped property that is only meaningful post-binding. -impl QueryExpr { - /// Infer a scalar expression against its input relation using the same - /// canonical rules as projection schema derivation. - pub fn scalar_type(&self, input: &Schema) -> Result<(DataType, bool), QueryExprError> { - infer_expr_type(self, input) - } - - /// Output schema of the root of a canonical DAG. - pub fn output_schema(&self) -> Result { - match self { - QueryExpr::Scan { schema, .. } => Ok(schema.clone()), - - QueryExpr::Aggregate { - reduction, - measures, - output_names, - child, - .. - } => { - let in_schema = child.output_schema()?; - aggregate_output_schema(&in_schema, reduction, measures, output_names) - } - - QueryExpr::Filter { child, .. } - | QueryExpr::Sort { child, .. } - | QueryExpr::Limit { child, .. } - | QueryExpr::PromqlSubquery { child, .. } - // Series sampling keeps a subset of whole series unchanged, so the - // output schema (and row-uniqueness) is exactly the child's (#86). - | QueryExpr::PromqlSeriesSample { child, .. } - // Info enrichment adds runtime info labels — the statically-known - // schema is the child's (open), so it passes through (#84). - | QueryExpr::PromqlInfoEnrich { child, .. } - | QueryExpr::TimeRange { child, .. } - // A time shift (`offset`/`@`) moves *when* the child is evaluated, - // never its columns — schema passes through (#40). - | QueryExpr::TimeShift { child, .. } => child.output_schema(), - - // ρ — relabel preserves every input column and writes one label - // `dst` (Utf8): overwritten in place if it already exists, else - // appended (nullable — a `label_replace` regex non-match leaves it - // unset). The schema stays open (other labels remain runtime-only). - // A rewrite can collapse two label sets into one, so row-uniqueness - // is no longer provable — drop unique_keys. - QueryExpr::PromqlRelabel { dst, child, .. } => { - let mut out = child.output_schema()?; - if let Some(existing) = out.fields.iter_mut().find(|c| c.name == *dst) { - existing.dtype = FieldDataType::Plain(DataType::Utf8); - existing.nullable = true; - } else { - out.fields.push(Field::plain(dst.clone(), DataType::Utf8, true)); - } - out.unique_keys.clear(); - Ok(out) - } - - // π — one output column per projection item. Each item's type is - // inferred from its expression against the child schema; the name - // is the explicit alias or a derived default. A child unique key - // survives exactly when every one of its columns is passed through - // as a bare `Field` item (possibly reordered or aliased). Derived - // expressions cannot carry key identity. `time_index` is re-found - // by name. - QueryExpr::Project { cols, qualifier, child } => { - let in_schema = child.output_schema()?; - let columns: Vec = cols - .iter() - .enumerate() - .map(|(i, item)| { - let (dtype, nullable) = infer_expr_type(&item.expr, &in_schema)?; - let name = item - .alias - .clone() - .unwrap_or_else(|| default_proj_name(&item.expr, i, &in_schema)); - let c = Field::plain(name, dtype, nullable); - // A derived table re-qualifies its output columns with - // its alias, so `t.col` (and a join over two derived - // tables) resolves to the right relation. - Ok(match qualifier { - Some(q) => c.with_table(q), - None => c, - }) - }) - .collect::, QueryExprError>>()?; - let time_index = columns.iter().position(|c| c.name == "ts"); - let unique_keys = in_schema - .unique_keys - .iter() - .filter_map(|key| { - key.iter() - .map(|input_col| { - cols.iter().position(|item| { - matches!(&item.expr, QueryExpr::Column(col) if col == input_col) - }) - }) - .collect::>>() - }) - .collect(); - Ok(Schema { - fields: columns, - time_index, - unique_keys, - // Projection enumerates exactly its items → closed. - closed: true, - }) - } - - QueryExpr::Dedup { cols, child } => { - let mut out = child.output_schema()?; - // Deduplicating on `cols` makes them a unique key of the result. - if !cols.is_empty() { - out.add_unique_key(cols.clone()); - } - Ok(out) - } - - // ⊕ — the branches are union-compatible by construction, so the - // output shape is the first child's. A row can appear in more than - // one branch, so no key of one branch is a key of the union: drop - // unique_keys, exactly as `SetOp` does — unless the constructor - // asserted `discriminator_unique_key` (issue #228), in which case - // `(discriminator, inner_key)` becomes the sole unique key. That - // assertion is trusted verbatim here, never checked: see - // `ConcatDiscriminatorKey`'s doc for the soundness argument and - // whose obligation it is. - QueryExpr::Concat { - children, - discriminator_unique_key, - } => { - let mut s = children - .first() - .ok_or(QueryExprError::EmptyConcat) - .and_then(|c| c.output_schema())?; - s.unique_keys.clear(); - if let Some(key) = discriminator_unique_key { - let mut compound = vec![*key.discriminator()]; - compound.extend(key.inner_key().iter().copied()); - s.add_unique_key(compound); - } - Ok(s) - } - // Set operations are union-compatible: both sides share the left's - // column shape, so the output schema is the left's. (Row identity - // is not preserved across a UNION, so unique_keys are dropped.) - QueryExpr::SetOp { left, .. } => { - let mut s = left.output_schema()?; - s.unique_keys.clear(); - Ok(s) - } - // ⋈ — output is the concatenation of both inputs' columns. Outer - // joins make the non-preserved side nullable. Post-join row - // identity isn't provable in general, so unique_keys reset. - QueryExpr::Join { - kind, left, right, .. - } => { - let l = left.output_schema()?; - let r = right.output_schema()?; - // Semi / anti joins filter the left side; the right contributes - // no columns, so the output is the left's schema unchanged. Row - // identity *is* preserved (each left row appears at most once), - // but a left row can be dropped, so unique_keys still reset. - if matches!(kind, JoinKind::Semi | JoinKind::Anti) { - return Ok(Schema { - unique_keys: Vec::new(), - ..l - }); - } - let (left_null, right_null) = match kind { - JoinKind::Left => (false, true), - JoinKind::Right => (true, false), - JoinKind::Full => (true, true), - JoinKind::Inner | JoinKind::Cross => (false, false), - JoinKind::Semi | JoinKind::Anti => unreachable!("handled above"), - }; - let l_len = l.fields.len(); - let mut columns = Vec::with_capacity(l_len + r.fields.len()); - columns.extend(l.fields.iter().cloned().map(|mut c| { - c.nullable |= left_null; - c - })); - columns.extend(r.fields.iter().cloned().map(|mut c| { - c.nullable |= right_null; - c - })); - let time_index = l.time_index.or(r.time_index.map(|i| i + l_len)); - Ok(Schema { - fields: columns, - time_index, - unique_keys: Vec::new(), - // The concatenation is complete only if both sides are. - closed: l.closed && r.closed, - }) - } - // ψ-analytic — child schema + one appended window-output column. - QueryExpr::SQLWindowFunc { - func, - args, - output_name, - child, - .. - } => { - let mut out = child.output_schema()?; - // First operand's (dtype, nullable) from the child schema, owned - // so the borrow ends before we append. - let arg = args.first().and_then(|a| match a { - QueryExpr::Column(id) => out.fields.get(*id), - _ => None, - }); - let arg_dtype = || { - arg.and_then(|c| c.plain_dtype().cloned()) - .unwrap_or(DataType::Float64) - }; - let (dtype, nullable) = match func { - WindowFuncKind::RowNumber - | WindowFuncKind::Rank - | WindowFuncKind::DenseRank - | WindowFuncKind::Count => (DataType::Int64, false), - WindowFuncKind::Sum | WindowFuncKind::Avg => (DataType::Float64, true), - // Navigation funcs: arg type, nullable (boundary rows are NULL). - WindowFuncKind::Lag - | WindowFuncKind::Lead - | WindowFuncKind::LagInFrame - | WindowFuncKind::LeadInFrame - | WindowFuncKind::FirstValue - | WindowFuncKind::LastValue - | WindowFuncKind::NthValue(_) => (arg_dtype(), true), - WindowFuncKind::Min | WindowFuncKind::Max => { - (arg_dtype(), arg.is_none_or(|c| c.nullable)) - } - }; - out.fields - .push(Field::plain(output_name.clone(), dtype, nullable)); - Ok(out) - } - - // A scalar bridge has no series — model it as a single `value` - // column so it can sit as a `BinaryOp` operand. Both scalar - // leaves — a bridged scalar sub-expression and the eval time — - // are a single `value` column with no labels. Every - // `PromqlScalarBridge` constructed today wraps a plain - // `Literal(Float64)` (issue #220), so the schema doesn't need to - // inspect the inner node. - QueryExpr::PromqlScalarBridge(_) | QueryExpr::EvalTimestamp => Ok(Schema { - fields: vec![Field::plain("value", DataType::Float64, false)], - time_index: None, - unique_keys: Vec::new(), - closed: true, - }), - - QueryExpr::CurrentTimestamp => Ok(Schema { - fields: vec![Field::plain("value", DataType::Timestamp, false)], - time_index: None, - unique_keys: Vec::new(), - closed: true, - }), - - // `vector(s)` yields a label-less instant vector: the (ts, value) - // floor and nothing else. `closed` — its full label set (empty) is - // known statically (#48). - QueryExpr::PromqlVectorFromScalar(_) => Ok(Schema { - fields: vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ], - time_index: Some(0), - unique_keys: Vec::new(), - closed: true, - }), - - // `scalar(v)` collapses to a single `value`, no time index — the same - // scalar shape as a constant or `time()` (#48). - QueryExpr::PromqlScalarFromVector(_) => Ok(Schema { - fields: vec![Field::plain("value", DataType::Float64, false)], - time_index: None, - unique_keys: Vec::new(), - closed: true, - }), - - // The output shape of ` op ` (or ` op - // `) is the vector side's — a scalar operand (a constant or - // `time()`) contributes only its value, no labels. Prefer the - // non-scalar side. - QueryExpr::BinaryOp { lhs, rhs, op, vector_match } => { - fn scalar(expression: &QueryExpr) -> bool { - match expression { - QueryExpr::PromqlScalarBridge(_) | QueryExpr::EvalTimestamp | QueryExpr::PromqlScalarFromVector(_) => true, - QueryExpr::BinaryOp { lhs, rhs, .. } => scalar(lhs) && scalar(rhs), - _ => false, - } - } - let left = lhs.output_schema()?; - let right = rhs.output_schema()?; - if scalar(lhs) { return Ok(right); } - let mut output = left; - let grouping = vector_match.as_ref().and_then(|m| m.grouping.as_ref()); - let right_rows = matches!(op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)) - || matches!(grouping, Some(g) if g.side == GroupSide::Right); - let mut additions = Vec::new(); - if right_rows { - additions.extend(right.fields.iter().filter(|c| c.dtype == DataType::Utf8).cloned()); - } - if let Some(grouping) = grouping { - additions.extend(grouping.labels.iter().map(|name| Field::plain(name.clone(), DataType::Utf8, true))); - } - for column in additions { - if !output.fields.iter().any(|c| c.name == column.name) { - output.fields.push(column); - } - } - Ok(output) - }, - - // The scalar variants (issue #205) — see `QueryExprError::ScalarHasNoRowSchema`. - QueryExpr::Column(_) - | QueryExpr::Literal(_) - | QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::Cast { .. } - | QueryExpr::InList { .. } - | QueryExpr::FunctionCall { .. } - | QueryExpr::Arithmetic { .. } - | QueryExpr::Case { .. } => Err(QueryExprError::ScalarHasNoRowSchema), - } - } -} - -/// Output schema of a *per-series* window/range reduction (`rate`/`increase`, -/// or an `*_over_time` reducer under a time `Window`). Such a reduction emits -/// one value per series, so every label column of `input` is preserved and only -/// the sample value is replaced — kept named `value` so the PromQL sample-value -/// convention (and any outer `SampleValue` reference) still resolves it by name. -fn per_series_reduction_schema(input: &Schema, agg: &AggIntent) -> Result { - let vi = if let Some(index) = agg.input_cols().first() { - *index - } else { - super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, input) - .map_err(|error| QueryExprError::InvalidSampleColumn(error.to_string()))? - }; - if !matches!( - input.fields.get(vi).map(|column| &column.dtype), - Some(FieldDataType::Plain(DataType::Float64 | DataType::Int64)) - ) { - return Err(QueryExprError::InvalidSampleColumn(format!( - "column {vi} is not numeric" - ))); - } - let mut columns = input.fields.clone(); - { - let mut out = agg.output_column(&columns[vi]); - out.name = "value".into(); - // A per-series range reduction produces a PromQL sample value, which is - // always `float64` — override the reducer's own output dtype so - // `count_over_time` (whose `Count` intent types `Int64`) matches every - // other range reducer instead of leaking an `Int64` value column (#69). - out.dtype = FieldDataType::Plain(DataType::Float64); - columns[vi] = out; - } - Ok(Schema { - fields: columns, - time_index: input.time_index, - unique_keys: input.unique_keys.clone(), - // Per-series reduction is label-preserving: it inherits its input's - // completeness (an open scan stays open; a closed one stays closed). - closed: input.closed, - }) -} - -/// The output schema of an `Aggregate { reduction, measures }` over `in_schema` — -/// the **single** canonical derivation shared by -/// [`QueryExpr::output_schema`]'s `Aggregate` arm and the converter's -/// HAVING-resolution path (`column_resolution::output_schema_for_aggregate`), -/// so the two can never drift (issue #41). -/// -/// `Reduction::PerEntity` selects the label-preserving -/// [`per_series_reduction_schema`] (`rate`/`increase`/`*_over_time`) instead -/// of the cross-series `by ++ measures` shape. Which one applies is read directly -/// off `reduction` — decided once, at construction, by whoever built the -/// `Aggregate` node (issue #165) — not re-derived here from `by`/child shape. -pub fn aggregate_output_schema( - in_schema: &Schema, - reduction: &Reduction, - measures: &[AggIntent], - output_names: &[String], -) -> Result { - let by = match reduction { - Reduction::PerEntity => { - debug_assert_eq!( - measures.len(), - 1, - "a per-entity reduction is single-aggregate" - ); - return per_series_reduction_schema(in_schema, &measures[0]); - } - Reduction::Reduce(by) => by, - }; - - // `without(excluded)` groups by every label *except* those listed: the kept - // labels are the input's label columns minus the excluded positions (and the - // ts / sample-value columns), and the schema stays **open** because the full - // runtime label set isn't known. The `by(...)` path instead enumerates its - // kept columns and freezes to closed (issue #39). - if by.is_without() { - return without_output_schema(in_schema, by.keys(), measures, output_names); - } - - let mut out_cols: Vec = Vec::with_capacity(by.len() + measures.len()); - for &id in by.keys() { - let c = in_schema - .fields - .get(id) - .ok_or(QueryExprError::InvalidGroupByColumn( - id, - in_schema.fields.len(), - ))?; - out_cols.push(c.clone()); - } - let value_col_idx = - super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema) - .ok() - .or_else(|| (0..in_schema.fields.len()).find(|i| !by.contains(i))); - let probe = value_col_idx - .and_then(|i| in_schema.fields.get(i)) - .cloned() - .unwrap_or_else(|| Field::plain("value", DataType::Float64, false)); - // Each reducer types off its own input column (`SUM(bytes)` vs `AVG(latency)` - // in one node); `None` falls back to the sample-value probe (PromQL's - // single-column convention). A non-empty `output_names[i]` overrides the - // synthetic output column name. - for (i, intent) in measures.iter().enumerate() { - // `count_values("l", v)` emits TWO columns: the synthesized `Utf8` label - // `l` (the stringified sample value it groups by) and the per-value - // count. If `l` collides with a group-by key of the same name, PromQL's - // synthesized label takes precedence — emit a single column, never a - // duplicate. - if let AggIntent::CountValues { label } = intent { - if !out_cols.iter().any(|c| c.name == *label) { - out_cols.push(Field::plain(label.clone(), DataType::Utf8, false)); - } - let mut cnt = intent.output_column(&probe); - if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { - cnt.name = name.clone(); - } - out_cols.push(cnt); - continue; - } - // Only the output *type* is read from here, so the leading column is - // enough for the multi-column intents: `Cardinality` and `PearsonCorr` - // both have a fixed output type that ignores it. - let in_col = intent - .input_cols() - .first() - .and_then(|id| in_schema.fields.get(*id)) - .unwrap_or(&probe); - let mut out = intent.output_column(in_col); - // A global extremum emits NULL for an empty input, even if its input - // column is non-nullable. Grouped extrema only emit existing groups. - if by.is_empty() && matches!(intent, AggIntent::Min { .. } | AggIntent::Max { .. }) { - out.nullable = true; - } - if let Some((arg, _)) = intent - .arg_selector_columns(in_schema) - .map_err(QueryExprError::InvalidScalarSignature)? - { - out.dtype = in_schema.fields[arg].dtype.clone(); - out.nullable = in_schema.fields[arg].nullable; - } - if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { - out.name = name.clone(); - } - out_cols.push(out); - } - // `count_values` groups by (by-keys ∪ the synthesized value label), so the - // by-keys alone are not a unique key — be conservative and claim none. - let has_count_values = measures - .iter() - .any(|a| matches!(a, AggIntent::CountValues { .. })); - let unique_keys = if by.is_empty() || has_count_values { - Vec::new() - } else { - vec![(0..by.len()).collect()] - }; - Ok(Schema { - fields: out_cols, - time_index: None, - unique_keys, - // A cross-series aggregate enumerates exactly `by ++ measures`, so its output - // is closed even over an open input — this is where an open schema - // freezes to closed. - closed: true, - }) -} - -/// Output schema of a `without(excluded)` aggregate: the kept labels (every -/// input label column except the `excluded` positions, the time axis, and the -/// sample-value column) followed by the aggregate output column(s). Unlike the -/// `by` path this stays **open** — the excluded set is enumerable but the kept -/// set is not (the runtime carries labels the usage-derived schema never saw), -/// so the schema can't freeze to closed and claims no unique key (issue #39). -fn without_output_schema( - in_schema: &Schema, - excluded: &[ColumnId], - measures: &[AggIntent], - output_names: &[String], -) -> Result { - for &id in excluded { - if id >= in_schema.fields.len() { - return Err(QueryExprError::InvalidGroupByColumn( - id, - in_schema.fields.len(), - )); - } - } - // A nested aggregate renames the sample value (`sum by (le) (…)` → `sum`); - // it is still the value, not a kept label. - let value = - super::column_resolution::resolve_column_ref(&ColumnRef::SampleValue, in_schema).ok(); - let mut out_cols: Vec = Vec::new(); - for (i, col) in in_schema.fields.iter().enumerate() { - let is_time = in_schema.time_index == Some(i); - if !is_time && value != Some(i) && !excluded.contains(&i) { - out_cols.push(col.clone()); - } - } - let probe = value - .and_then(|i| in_schema.fields.get(i)) - .cloned() - .unwrap_or_else(|| Field::plain("value", DataType::Float64, false)); - for (i, intent) in measures.iter().enumerate() { - // Only the output *type* is read from here, so the leading column is - // enough for the multi-column intents: `Cardinality` and `PearsonCorr` - // both have a fixed output type that ignores it. - let in_col = intent - .input_cols() - .first() - .and_then(|id| in_schema.fields.get(*id)) - .unwrap_or(&probe); - let mut out = intent.output_column(in_col); - if let Some((arg, _)) = intent - .arg_selector_columns(in_schema) - .map_err(QueryExprError::InvalidScalarSignature)? - { - out.dtype = in_schema.fields[arg].dtype.clone(); - out.nullable = in_schema.fields[arg].nullable; - } - if let Some(name) = output_names.get(i).filter(|s| !s.is_empty()) { - out.name = name.clone(); - } - out_cols.push(out); - } - Ok(Schema { - fields: out_cols, - time_index: None, - unique_keys: Vec::new(), - // The kept label set is runtime-only, so — unlike `by` — this does not - // freeze the open schema to closed. - closed: false, - }) -} - -/// Infer the `(DataType, nullable)` a scalar [`QueryExpr`] produces against an -/// input [`Schema`]. Used by `Project` schema derivation. Approximate here: -/// unknown columns and bare `FunctionCall`s fall back to a permissive default -/// (post-ASAP binding refines with a real function/type registry). `expr` -/// must be one of the scalar variants (issue #205) — an operator variant here -/// is a construction bug, not a shape this needs to handle silently. -fn infer_expr_type( - expr: &QueryExpr, - schema: &Schema, -) -> Result<(DataType, bool), QueryExprError> { - Ok(match expr { - QueryExpr::CurrentTimestamp => (DataType::Timestamp, false), - QueryExpr::Column(id) => match schema.fields.get(*id) { - Some(c) => match c.plain_dtype() { - Some(dtype) => (dtype.clone(), c.nullable), - // Summary state is not a scalar value: it has to be read - // out (estimated / finalized) before an expression can use it. - None => { - return Err(QueryExprError::InvalidScalarSignature(format!( - "column `{}` carries summary state and cannot be read as a value", - c.name - ))) - } - }, - None => (DataType::Float64, true), - }, - QueryExpr::Literal(s) => match s { - ScalarValue::Int64(_) => (DataType::Int64, false), - ScalarValue::Float64(_) => (DataType::Float64, false), - ScalarValue::Utf8(_) => (DataType::Utf8, false), - ScalarValue::Boolean(_) => (DataType::Bool, false), - ScalarValue::Null => (DataType::Null, true), - ScalarValue::Interval { .. } => (DataType::Interval, false), - }, - // Boolean-valued expressions (SQL three-valued logic → nullable). - QueryExpr::Compare { .. } - | QueryExpr::BoolAnd(_) - | QueryExpr::BoolOr(_) - | QueryExpr::Not(_) - | QueryExpr::IsNull(_) - | QueryExpr::IsNotNull(_) - | QueryExpr::InList { .. } => (DataType::Bool, true), - QueryExpr::Arithmetic { op, left, right } => { - let (lt, ln) = infer_expr_type(left, schema)?; - let (rt, rn) = infer_expr_type(right, schema)?; - // Temporal subtraction yields a fixed duration with a unit, not a - // calendar interval or a floating-point number. Until the IR can - // preserve that unit, fail instead of publishing a numeric schema. - if matches!(op, ArithmeticOpKind::Sub) - && matches!(lt, DataType::Date | DataType::Timestamp) - && matches!(rt, DataType::Date | DataType::Timestamp) - { - return Err(QueryExprError::InvalidScalarSignature( - "temporal subtraction produces an unsupported duration type".into(), - )); - } - - // Operand order is not checked: the orders that are not valid SQL - // (`Interval - Timestamp`) are rejected by the planner upstream, so - // a pair rule stays as small as the numeric one it sits beside. - let dtype = match (<, &rt) { - // SQL unary minus lowers to -1 * expression, including intervals. - (DataType::Int64, DataType::Interval) | (DataType::Interval, DataType::Int64) - if matches!(op, ArithmeticOpKind::Mul) => - { - DataType::Interval - } - (DataType::Timestamp, DataType::Interval) - | (DataType::Interval, DataType::Timestamp) => DataType::Timestamp, - (DataType::Date, DataType::Interval) | (DataType::Interval, DataType::Date) => { - DataType::Date - } - (DataType::Interval, DataType::Interval) => DataType::Interval, - (DataType::Int64, DataType::Int64) => DataType::Int64, - _ => DataType::Float64, - }; - (dtype, ln || rn) - } - QueryExpr::Cast { to, try_cast, expr } => { - let (_, nullable) = infer_expr_type(expr, schema)?; - (to.clone(), *try_cast || nullable) - } - QueryExpr::FunctionCall { name, args } => { - if name == "asap_element_access" { - super::scalar_type_rules::element_access_type(args, schema) - .map_err(QueryExprError::InvalidScalarSignature)? - } else if name == "asap_struct_field" { - super::scalar_type_rules::struct_field_type(args, schema) - .map_err(QueryExprError::InvalidScalarSignature)? - } else if let Some(function) = - super::scalar_type_rules::MapScalarFunction::from_name(name) - { - let arguments = args - .iter() - .map(|arg| infer_expr_type(arg, schema)) - .collect::, _>>()?; - function - .output_type(&arguments) - .map_err(QueryExprError::InvalidScalarSignature)? - } else { - // Legacy unknown functions retain their existing policy. - (DataType::Float64, true) - } - } - QueryExpr::Case { - branches, - else_expr, - .. - } => { - if let Some((_, then)) = branches.first() { - (infer_expr_type(then, schema)?.0, true) - } else if let Some(other) = else_expr { - infer_expr_type(other, schema)? - } else { - (DataType::Null, true) - } - } - other => { - unreachable!("infer_expr_type called on a non-scalar QueryExpr variant: {other:?}") - } - }) -} - -/// Default output-column name for a projection item with no explicit alias: -/// a bare column keeps its (schema) name; anything else gets `col_{i}`. -fn default_proj_name(expr: &QueryExpr, idx: usize, schema: &Schema) -> String { - match expr { - QueryExpr::Column(id) => schema - .fields - .get(*id) - .map(|c| c.name.clone()) - .unwrap_or_else(|| format!("col_{idx}")), - _ => format!("col_{idx}"), - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::expr_ir::{ArithmeticOpKind, CompareOpKind}; - use crate::types::AccuracyTarget; - - fn col(name: &str, dtype: DataType, nullable: bool) -> Field { - Field::plain(name, dtype, nullable) - } - - /// Shifting an instant by a duration stays an instant, and shifting a date - /// stays a date — neither falls through to the numeric default, which is - /// what `l_shipdate + INTERVAL '30' DAY` would otherwise be typed as. - #[test] - fn interval_arithmetic_keeps_the_temporal_type() { - let schema = Schema::new(vec![ - col("ts", DataType::Timestamp, false), - col("d", DataType::Date, false), - ]); - let thirty_days = || { - Rc::new(QueryExpr::Literal(ScalarValue::Interval { - months: 0, - days: 30, - nanos: 0, - })) - }; - let shift = |column, op| QueryExpr::Arithmetic { - op, - left: Rc::new(QueryExpr::Column(column)), - right: thirty_days(), - }; - - assert_eq!( - shift(0, ArithmeticOpKind::Add) - .scalar_type(&schema) - .unwrap() - .0, - DataType::Timestamp - ); - assert_eq!( - shift(1, ArithmeticOpKind::Sub) - .scalar_type(&schema) - .unwrap() - .0, - DataType::Date - ); - assert_eq!( - QueryExpr::Arithmetic { - op: ArithmeticOpKind::Add, - left: thirty_days(), - right: thirty_days(), - } - .scalar_type(&schema) - .unwrap() - .0, - DataType::Interval - ); - } - - fn scan( - columns: Vec, - time_index: Option, - uk: Vec>, - ) -> QueryExpr { - QueryExpr::Scan { - source: Source::Table { - table_ref: "t".into(), - }, - predicates: vec![], - schema: Schema { - fields: columns, - time_index, - unique_keys: uk, - closed: true, - }, - } - } - - #[test] - fn project_preserves_unique_keys_that_are_passed_through() { - let input = Rc::new(scan( - vec![ - col("tenant", DataType::Utf8, false), - col("region", DataType::Utf8, false), - col("value", DataType::Int64, false), - ], - None, - vec![vec![0, 1]], - )); - let projected = QueryExpr::Project { - cols: vec![ - ProjectItem { - alias: Some("r".into()), - expr: QueryExpr::Column(1), - }, - ProjectItem { - alias: Some("t".into()), - expr: QueryExpr::Column(0), - }, - ProjectItem { - alias: None, - expr: QueryExpr::Arithmetic { - op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(2)), - right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), - }, - }, - ], - qualifier: None, - child: input, - }; - - assert_eq!( - projected.output_schema().unwrap().unique_keys, - vec![vec![1, 0]] - ); - } - - #[test] - fn project_drops_a_unique_key_when_a_key_column_is_omitted() { - let input = Rc::new(scan( - vec![ - col("tenant", DataType::Utf8, false), - col("region", DataType::Utf8, false), - ], - None, - vec![vec![0, 1]], - )); - let projected = QueryExpr::Project { - cols: vec![ProjectItem { - alias: None, - expr: QueryExpr::Column(0), - }], - qualifier: None, - child: input, - }; - - assert!(projected.output_schema().unwrap().unique_keys.is_empty()); - } - - #[test] - fn legacy_window_json_without_frame_deserializes_as_unspecified() { - let window = QueryExpr::SQLWindowFunc { - func: WindowFuncKind::RowNumber, - args: vec![], - partition_by: GroupKeys::by(vec![]), - order_by: vec![], - frame: Some(WindowFrame { - units: WindowFrameUnits::Range, - start_bound: WindowFrameBound::Preceding(WindowFrameOffset::Scalar( - ScalarValue::Null, - )), - end_bound: WindowFrameBound::CurrentRow, - }), - output_name: "row_number".into(), - child: Rc::new(scan(vec![col("v", DataType::Int64, false)], None, vec![])), - }; - let mut json = serde_json::to_value(window).unwrap(); - json.get_mut("SQLWindowFunc") - .and_then(serde_json::Value::as_object_mut) - .unwrap() - .remove("frame"); - - let decoded: QueryExpr = serde_json::from_value(json).unwrap(); - assert!(matches!( - decoded, - QueryExpr::SQLWindowFunc { frame: None, .. } - )); - } - - /// A row can appear in more than one branch, so no branch's unique key is a - /// key of the union. `Concat` took the first child's schema verbatim, which - /// let a `Dedup`'s key leak out and claim a uniqueness the merged rows do - /// not have — `unique_keys` feeds CSE's producer-sharing legality check. - #[test] - fn merge_drops_the_branches_unique_keys() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan( - vec![ - col("k", DataType::Utf8, false), - col("v", DataType::Int64, false), - ], - None, - vec![], - )), - }; - assert_eq!( - branch().output_schema().unwrap().unique_keys, - vec![vec![0]], - "a Dedup branch does have a unique key on its own" - ); - - let merged = QueryExpr::concat(vec![branch(), branch()]); - let schema = merged.output_schema().unwrap(); - assert!( - schema.unique_keys.is_empty(), - "the union of two deduplicated branches is not deduplicated" - ); - // The column shape is still the first branch's. - assert_eq!(schema.fields.len(), 2); - } - - /// Same rule as `SetOp`, which already dropped them. - #[test] - fn merge_and_setop_agree_on_unique_keys() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), - }; - let merged = QueryExpr::concat(vec![branch(), branch()]); - let setop = QueryExpr::SetOp { - kind: RelationalSetOpKind::Union, - all: true, - left: Rc::new(branch()), - right: Rc::new(branch()), - }; - assert_eq!( - merged.output_schema().unwrap().unique_keys, - setop.output_schema().unwrap().unique_keys, - ); - } - - #[test] - fn an_empty_merge_has_no_schema() { - assert!(matches!( - QueryExpr::concat(vec![]).output_schema(), - Err(QueryExprError::EmptyConcat) - )); - } - - /// Issue #228: a `Concat` built via `concat_with_discriminator` gets a - /// sound compound `(discriminator, inner_key)` unique key, even though - /// each branch's own `inner_key` alone repeats across branches (exactly - /// the shape `merge_drops_the_branches_unique_keys` shows is unsafe - /// *without* a discriminator). - #[test] - fn discriminator_override_produces_a_compound_unique_key() { - // Two branches, each individually deduplicated on column 0 (`k`) — - // but, per `merge_drops_the_branches_unique_keys`, that alone proves - // nothing about the union. Field 1 (`branch_id`) stands in for a - // discriminator the constructor has separately proven distinct per - // branch (PromQL φ, a synthetic `GROUPING()` id, ...) — this - // schema-level test only checks the shape `output_schema` derives - // from asserting one, not how a real caller proves distinctness. - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan( - vec![ - col("k", DataType::Utf8, false), - col("branch_id", DataType::Int64, false), - ], - None, - vec![], - )), - }; - let merged = QueryExpr::concat_with_discriminator( - vec![branch(), branch()], - /* discriminator */ 1, - /* inner_key */ vec![0], - ); - let schema = merged.output_schema().unwrap(); - assert_eq!( - schema.unique_keys, - vec![vec![1, 0]], - "(discriminator, inner_key) is the sole asserted unique key" - ); - assert_eq!( - schema.fields.len(), - 2, - "column shape is still the first branch's" - ); - } - - #[test] - fn discriminator_assertion_rejects_unknown_wire_fields() { - let json = r#"{"discriminator":1,"inner_key":[0],"unverified":true}"#; - assert!(serde_json::from_str::(json).is_err()); - } - - /// The override is opt-in: building a `Concat` without asserting a - /// discriminator — via the plain struct literal, exactly like every call - /// site before issue #228 — still drops `unique_keys` by default, - /// unchanged. - #[test] - fn ordinary_concat_struct_literal_still_drops_unique_keys_by_default() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), - }; - let merged = QueryExpr::Concat { - children: vec![branch(), branch()], - discriminator_unique_key: None, - }; - assert!(merged.output_schema().unwrap().unique_keys.is_empty()); - } - - /// Misuse check (issue #228): there is no way to end up with a - /// discriminator-backed unique key without a call site literally naming - /// a column as the discriminator. Neither the ordinary `concat` - /// constructor nor a bare struct literal with `discriminator_unique_key: - /// None` can be coaxed into fabricating one — the only path that - /// produces `Some` is `concat_with_discriminator` / - /// `ConcatDiscriminatorKey::new`, both of which require `discriminator` - /// as an explicit, named argument. - #[test] - fn no_way_to_fabricate_a_unique_key_without_naming_a_discriminator() { - let branch = || QueryExpr::Dedup { - cols: vec![0], - child: Rc::new(scan(vec![col("k", DataType::Utf8, false)], None, vec![])), - }; - // The ordinary builder. - assert_eq!( - QueryExpr::concat(vec![branch(), branch()]) - .output_schema() - .unwrap() - .unique_keys, - Vec::>::new() - ); - // The bare struct literal, explicitly opting out. - assert_eq!( - QueryExpr::Concat { - children: vec![branch(), branch()], - discriminator_unique_key: None, - } - .output_schema() - .unwrap() - .unique_keys, - Vec::>::new() - ); - } - - #[test] - fn project_retypes_and_renames_per_item() { - let child = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("host", DataType::Utf8, false), - col("value", DataType::Float64, false), - ], - Some(0), - vec![vec![0, 1]], - ); - let q = QueryExpr::Project { - qualifier: None, - cols: vec![ - // bare column passthrough keeps its (schema) name + type: host=col 1 - ProjectItem { - alias: None, - expr: QueryExpr::Column(1), - }, - // arithmetic over value (col 2) → Float64 - ProjectItem { - alias: Some("dbl".into()), - expr: QueryExpr::Arithmetic { - op: ArithmeticOpKind::Add, - left: Rc::new(QueryExpr::Column(2)), - right: Rc::new(QueryExpr::Column(2)), - }, - }, - // comparison → Bool (nullable under 3-valued logic) - ProjectItem { - alias: Some("flag".into()), - expr: QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(2)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(0.0))), - }, - }, - ], - child: Rc::new(child), - }; - let s = q.output_schema().unwrap(); - assert_eq!(s.fields.len(), 3); - assert_eq!(s.fields[0], col("host", DataType::Utf8, false)); - assert_eq!(s.fields[1], col("dbl", DataType::Float64, false)); - assert_eq!(s.fields[2], col("flag", DataType::Bool, true)); - // projection drops the time axis + unique keys (ts not retained) - assert!(s.time_index.is_none()); - assert!(s.unique_keys.is_empty()); - } - - #[test] - fn group_keys_by_vs_without_semantics() { - let by = GroupKeys::by(vec![1, 2]); - let without = GroupKeys::without(vec![1, 2]); - assert!(!by.is_without()); - assert!(without.is_without()); - // Deref / iteration expose the stored keys regardless of mode. - assert_eq!(by.len(), 2); - assert_eq!(without.keys(), &[1, 2]); - // A `by` compares equal to its bare vec; a `without` never does. - assert_eq!(by, vec![1, 2]); - assert_ne!(without, vec![1, 2]); - assert_ne!(by, without); - } - - #[test] - fn group_keys_serde_by_is_bare_array_without_is_tagged() { - // `by` keeps the pre-#39 bare-array wire format; `without` uses an object. - let by = serde_json::to_string(&GroupKeys::by(vec![2, 3])).unwrap(); - assert_eq!(by, "[2,3]"); - let without = serde_json::to_string(&GroupKeys::without(vec![2])).unwrap(); - assert_eq!(without, r#"{"without":[2]}"#); - // Round-trip both. - for g in [GroupKeys::by(vec![2, 3]), GroupKeys::without(vec![2])] { - let json = serde_json::to_string(&g).unwrap(); - let back: GroupKeys = serde_json::from_str(&json).unwrap(); - assert_eq!(back, g); - } - } - - #[test] - fn without_aggregate_keeps_open_schema_minus_excluded() { - // `sum without (instance) (m)` over `[ts, value, instance, job]`: the - // kept labels are the input labels minus the excluded `instance` (and ts - // / value), followed by the `sum` column, and the schema stays OPEN - // (issue #39). `job` survives; `instance` is dropped. - let scan_node = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::with_time_index( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("instance", DataType::Utf8, true), - col("job", DataType::Utf8, true), - ], - 0, - vec![], - ), - }; - let agg = QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::without(vec![2])), // exclude `instance` - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(scan_node), - }; - let s = agg.output_schema().unwrap(); - let names: Vec<_> = s.fields.iter().map(|c| c.name.as_str()).collect(); - assert_eq!(names, vec!["job", "sum"], "kept `job`, dropped `instance`"); - assert!(!s.closed, "a `without` result stays open"); - assert!(s.time_index.is_none()); - assert!(s.unique_keys.is_empty(), "kept set unknown → no unique key"); - } - - // A nested aggregate's renamed sample value is not a kept label. - #[test] - fn without_aggregate_drops_a_renamed_sample_value() { - // `sum without (inst) (sum by (inst, job) (m))` over `[inst, job, sum]`. - let inner = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: Schema::new(vec![ - col("inst", DataType::Utf8, true), - col("job", DataType::Utf8, true), - col("sum", DataType::Float64, false), - ]), - }; - let agg = QueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::without(vec![0])), - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(inner), - }; - let s = agg.output_schema().unwrap(); - let names: Vec<_> = s.fields.iter().map(|c| c.name.as_str()).collect(); - assert_eq!(names, vec!["job", "sum"]); - } - - #[test] - fn time_shift_is_schema_pass_through() { - // `offset`/`@` move *when* a selector is evaluated, never its columns — - // a `TimeShift` output schema equals its child's (issue #40). - let scan_node = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - Some(0), - vec![], - ); - let shifted = QueryExpr::TimeShift { - shift: TimeShift { - offset_ms: 3_600_000, - at: Some(AtModifier::Timestamp(1_609_746_000_000)), - }, - child: Rc::new(scan_node.clone()), - }; - assert_eq!( - shifted.output_schema().unwrap(), - scan_node.output_schema().unwrap(), - ); - } - - #[test] - fn time_shift_identity_and_serde() { - let offset_only = TimeShift { - offset_ms: 1, - at: None, - }; - let at_only = TimeShift { - offset_ms: 0, - at: Some(AtModifier::End), - }; - assert!(TimeShift::default().is_identity()); - assert!(!offset_only.is_identity()); - assert!(!at_only.is_identity()); - // Round-trip the shift + anchor. - let s = TimeShift { - offset_ms: -300_000, - at: Some(AtModifier::Timestamp(60_000)), - }; - let back: TimeShift = serde_json::from_str(&serde_json::to_string(&s).unwrap()).unwrap(); - assert_eq!(back, s); - } - - // Nested temporal aggregation must replace the sample, never the grouping label. - #[test] - fn temporal_reduction_of_grouped_sum_preserves_job() { - let input = Schema::new(vec![ - col("job", DataType::Utf8, true), - col("sum", DataType::Float64, false), - ]); - for aggregate in [ - AggIntent::Avg { col: None }, - AggIntent::Avg { col: Some(1) }, - AggIntent::Rate, - ] { - let output = - aggregate_output_schema(&input, &Reduction::PerEntity, &[aggregate], &[]).unwrap(); - assert_eq!(output.fields[0], input.fields[0]); - assert_eq!(output.fields[1].name, "value"); - assert_eq!(output.fields[1].dtype, DataType::Float64); - } - } - - #[test] - fn per_series_rate_preserves_labels() { - // A per-series range reduction (`rate`) is label-preserving: it produces - // one value per series, so every label survives and only the sample - // value is replaced (kept named `value`). The TimeRange child is the - // structural marker; the outer Aggregate carries the Rate intent. - let scan_node = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - Some(0), - vec![], - ); - let rate = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Rate], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan_node), - }), - }; - let s = rate.output_schema().unwrap(); - assert_eq!( - s.fields.iter().map(|c| c.name.as_str()).collect::>(), - vec!["ts", "value", "job"], - "rate preserves all labels; only the sample value is replaced" - ); - assert_eq!(s.time_index, Some(0)); - assert!(s.column_id("job").is_some(), "label survives the reduction"); - } - - #[test] - fn over_time_reduction_preserves_labels() { - // `*_over_time` lowers to `Aggregate { by:[], [reducer], TimeRange { Scan } }`: - // a per-series time-range reduction. The TimeRange child confers per-series - // semantics on otherwise cross-series intents like `Avg`, so an outer - // `sum by(job)(avg_over_time(...))` resolves its key positionally. - let scan_node = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - Some(0), - vec![], - ); - let avg_over_time = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Avg { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(QueryExpr::TimeRange { - range: Duration::from_secs(300), - child: Rc::new(scan_node), - }), - }; - let s = avg_over_time.output_schema().unwrap(); - assert_eq!( - s.fields.iter().map(|c| c.name.as_str()).collect::>(), - vec!["ts", "value", "job"], - "TimeRange-child marks per-series: labels preserved, value renamed" - ); - assert!( - s.column_id("job").is_some(), - "outer Aggregate.by can resolve it" - ); - } - - #[test] - fn completeness_open_leaf_freezes_to_closed_at_cross_series_aggregate() { - // A schemaless (PromQL-style) leaf is *open*; it stays open through a - // per-series reduction (`rate`), then is **frozen to closed** by a - // cross-series aggregate (which enumerates exactly its output columns). - let open_leaf = QueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - // `with_time_index` defaults to `closed: false` (open). - schema: Schema::with_time_index( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - col("job", DataType::Utf8, true), - ], - 0, - vec![], - ), - }; - assert!( - !open_leaf.output_schema().unwrap().closed, - "schemaless leaf is open" - ); - - let rate = QueryExpr::Aggregate { - reduction: Reduction::PerEntity, - measures: vec![AggIntent::Rate], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(open_leaf), - }; - assert!( - !rate.output_schema().unwrap().closed, - "per-series rate is label-preserving → stays open" - ); - - let sum_by_job = QueryExpr::Aggregate { - reduction: Reduction::by(vec![2]), // `job` - measures: vec![AggIntent::Sum { col: None }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(rate), - }; - assert!( - sum_by_job.output_schema().unwrap().closed, - "cross-series aggregate enumerates `by ++ measures` → frozen to closed" - ); - } - - #[test] - fn project_keeps_time_index_when_ts_passed_through() { - let child = scan( - vec![ - col("ts", DataType::Timestamp, false), - col("value", DataType::Float64, false), - ], - Some(0), - vec![], - ); - let q = QueryExpr::Project { - qualifier: None, - cols: vec![ - // value=col 1, ts=col 0 - ProjectItem { - alias: None, - expr: QueryExpr::Column(1), - }, - ProjectItem { - alias: None, - expr: QueryExpr::Column(0), - }, - ], - child: Rc::new(child), - }; - let s = q.output_schema().unwrap(); - assert_eq!(s.fields[0].name, "value"); - assert_eq!(s.fields[1].name, "ts"); - assert_eq!(s.time_index, Some(1)); - } - - fn join(kind: JoinKind) -> QueryExpr { - let left = scan(vec![col("a", DataType::Int64, false)], None, vec![vec![0]]); - let right = scan(vec![col("b", DataType::Utf8, false)], None, vec![]); - QueryExpr::Join { - kind, - pred: Predicate(Rc::new(QueryExpr::Literal(ScalarValue::Boolean(true)))), - left: Rc::new(left), - right: Rc::new(right), - } - } - - #[test] - fn inner_join_concatenates_both_sides() { - let s = join(JoinKind::Inner).output_schema().unwrap(); - assert_eq!(s.fields.len(), 2); - assert_eq!(s.fields[0], col("a", DataType::Int64, false)); - assert_eq!(s.fields[1], col("b", DataType::Utf8, false)); - // post-join row identity not provable → no unique keys - assert!(s.unique_keys.is_empty()); - } - - #[test] - fn left_join_makes_right_side_nullable() { - let s = join(JoinKind::Left).output_schema().unwrap(); - assert!(!s.fields[0].nullable, "preserved left side stays non-null"); - assert!(s.fields[1].nullable, "right side nullable under LEFT JOIN"); - } - - #[test] - fn full_join_makes_both_sides_nullable() { - let s = join(JoinKind::Full).output_schema().unwrap(); - assert!(s.fields[0].nullable); - assert!(s.fields[1].nullable); - } - - #[test] - fn setop_takes_left_shape_and_drops_unique_keys() { - let left = scan( - vec![ - col("k", DataType::Utf8, false), - col("v", DataType::Int64, false), - ], - None, - vec![vec![0]], - ); - let right = scan( - vec![ - col("k", DataType::Utf8, false), - col("v", DataType::Int64, false), - ], - None, - vec![vec![0]], - ); - let q = QueryExpr::SetOp { - kind: RelationalSetOpKind::Union, - all: false, - left: Rc::new(left), - right: Rc::new(right), - }; - let s = q.output_schema().unwrap(); - assert_eq!(s.fields.len(), 2); - assert_eq!(s.fields[0].name, "k"); - assert!( - s.unique_keys.is_empty(), - "UNION does not preserve row identity" - ); - } - - // ── PromqlScalarBridge / Literal dedup (issue #220) ───────────────────── - - /// `QueryExpr::promql_scalar(v)` — what every front end now constructs in - /// place of the old `PromqlScalar(v)` leaf — wraps exactly - /// `Literal(ScalarValue::Float64(v))`: the same value a SQL-emitted typed - /// float literal in a scalar-sub-language position would carry, just at a - /// different DAG position. `as_promql_scalar` is the round-trip inverse. - #[test] - fn promql_scalar_bridges_a_literal_float_at_an_operator_position() { - let bridge = QueryExpr::::promql_scalar(2.5); - assert_eq!( - bridge, - QueryExpr::PromqlScalarBridge(Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.5)))) - ); - assert_eq!(bridge.as_promql_scalar(), Some(2.5)); - - // The same value a SQL `Compare`/`Arithmetic` operand would carry, in - // its native (unwrapped, no row schema) scalar-sub-language position — - // no longer a different variant, just not bridged to this DAG - // position. - let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); - assert_eq!(bridge.as_promql_scalar(), Some(2.5)); - assert_ne!( - bridge, sql_literal, - "bridge and bare literal are distinct nodes" - ); - // Not every shape is a scalar bridge: neither a bare `Literal` nor an - // operator node reports a value. - assert_eq!(sql_literal.as_promql_scalar(), None); - assert_eq!(scan(vec![], None, vec![]).as_promql_scalar(), None); - } - - /// Pins the DAG-position distinction issue #220 asks for: the very same - /// `Literal(ScalarValue::Float64(_))` value has a row schema when it sits - /// at the operator-DAG position (wrapped in `PromqlScalarBridge` — a - /// `BinaryOp` operand, `PromqlVectorFromScalar` child, or a query root), - /// and has none when it sits bare, in a scalar-sub-language position - /// (`Compare`/`Arithmetic`/… operand) — no longer decided by which of two - /// duplicate variants was used, only by whether the wrapper is present. - #[test] - fn row_schema_rides_on_the_bridge_wrapper_not_the_literal_variant() { - let bridged = QueryExpr::::promql_scalar(42.0); - let schema = bridged.output_schema().expect("bridge has a row schema"); - assert_eq!(schema.fields.len(), 1); - assert_eq!(schema.fields[0].name, "value"); - assert_eq!(schema.fields[0].dtype, DataType::Float64); - assert!(schema.time_index.is_none()); - - // The identical value, unwrapped (the scalar-sub-language position a - // `Compare`/`Arithmetic` operand would occupy) has no row schema of - // its own — it's a construction bug to call `output_schema` on it - // directly, caught as `ScalarHasNoRowSchema` rather than panicking. - let bare = QueryExpr::::Literal(ScalarValue::Float64(42.0)); - assert!(matches!( - bare.output_schema(), - Err(QueryExprError::ScalarHasNoRowSchema) - )); - } - - /// `BinaryOp`'s schema derivation follows the non-scalar (vector) side - /// when the other operand is a `PromqlScalarBridge`, and a `VectorMatch` - /// modifier survives unchanged alongside it — the relational binary-op - /// path (issue #220's Instance 2, left as follow-up) is untouched by the - /// Instance-1 `PromqlScalar` → `PromqlScalarBridge` collapse. - // `filters` (#466) round-trips, and an `Aggregate` serialized before the - // field existed still deserializes as unfiltered. - #[test] - fn aggregate_filters_serde_round_trip_and_default() { - let child = Rc::new(scan( - vec![ - col("service", DataType::Utf8, false), - col("latency", DataType::Float64, false), - ], - None, - vec![], - )); - let filtered = QueryExpr::Aggregate { - reduction: Reduction::by(vec![0]), - measures: vec![ - AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }, - AggIntent::Sum { col: Some(1) }, - ], - output_names: vec![], - filters: vec![ - Some(Predicate(Rc::new(QueryExpr::Compare { - left: Rc::new(QueryExpr::Column(1)), - op: CompareOpKind::Gt, - right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), - }))), - None, - ], - having: None, - child: Rc::clone(&child), - }; - let json = serde_json::to_value(&filtered).unwrap(); - assert_eq!( - serde_json::from_value::(json.clone()).unwrap(), - filtered - ); - - let mut legacy = json; - legacy["Aggregate"] - .as_object_mut() - .unwrap() - .remove("filters") - .expect("fixture sanity: filters was serialized"); - let decoded: QueryExpr = serde_json::from_value(legacy).unwrap(); - let QueryExpr::Aggregate { filters, .. } = &decoded else { - unreachable!() - }; - assert!(filters.is_empty()); - } - - #[test] - fn binary_op_schema_follows_the_vector_side_over_a_scalar_bridge_with_vector_match_intact() { - let vector = scan( - vec![ - col("host", DataType::Utf8, false), - col("value", DataType::Float64, false), - ], - None, - vec![], - ); - let vm = VectorMatch { - kind: VectorMatchKind::On, - labels: vec!["host".into()], - grouping: None, - }; - let op = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(vector.clone()), - rhs: Rc::new(QueryExpr::promql_scalar(1.0)), - vector_match: Some(vm.clone()), - }; - assert_eq!(op.output_schema().unwrap(), vector.output_schema().unwrap()); - let QueryExpr::BinaryOp { vector_match, .. } = &op else { - unreachable!() - }; - assert_eq!(vector_match.as_ref(), Some(&vm)); - } -} diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs deleted file mode 100644 index b4a5c87a6..000000000 --- a/crates/types/src/pre_asap/resolve.rs +++ /dev/null @@ -1,857 +0,0 @@ -//! Resolve a front-end-emitted, unresolved [`UnresolvedQueryExpr`] (`QueryExpr`) -//! into the canonical, positional [`ResolvedQueryExpr`] (`QueryExpr`). -//! -//! Both front ends (`asap-frontend-promql`, `asap-frontend-sql`) construct -//! canonical `QueryExpr` shapes directly during their own `interpret` step -//! (issue #179) — heavy-hitter `topk` recognition, the window-over-aggregate -//! fold, the `PerEntity`/`Reduce` reduction choice, and every other -//! *structural* decision happen right there, since a front end already knows -//! the answer at parse time. What's left for [`resolve_root`] is exactly the -//! "mechanical, schema-dependent substitution" #179 describes: a single -//! generic, shape-preserving walk — every [`UnresolvedQueryExpr`] variant maps to the -//! identical [`ResolvedQueryExpr`] variant — that resolves every [`ColumnRef`] to -//! the [`SchemaResolver`](super::schema_resolver::SchemaResolver)-computed positional [`ColumnId`]. -//! -//! ## Why positional `ColumnId`, not just carrying names all the way through (issue #216) -//! -//! A mature query engine can legitimately choose either design — DataFusion's -//! own logical plan (what `asap-frontend-sql` walks to build its `QueryExpr`) -//! and Calcite both keep names, with an optional table qualifier, all the way -//! through logical optimization, only going positional once they lower to a -//! physical plan. Resolving once, immediately after each front end's own -//! `interpret` step, is the better trade for *this* codebase's shape — one -//! front-end-facing DAG feeding several independent downstream passes -//! (`canonicalize`, the cost model, `dag_export`, schema/type inference, -//! `asap-aware-mapping`'s summary binding) — for three concrete reasons: -//! -//! 1. **Names collide across joins.** Not hypothetical: `join_predicate_disambiguates_shared_column_name` -//! (`crates/frontend-sql/tests/sql_lowering.rs`) exists specifically because -//! `metrics.service` and `hosts.service` are both just `"service"` once their -//! schemas are concatenated. A bare name is ambiguous the moment two sources -//! share one; `ColumnId` is what makes "the second `service`, position 4, not -//! the first" a fact recorded once, instead of a lookup redone at every use site. -//! 2. **A name's meaning changes going up the DAG.** `Project` renames/aliases, -//! `Aggregate` collapses columns and introduces synthetic ones, `Join` -//! concatenates two schemas — a name valid at a `Scan` leaf isn't -//! automatically the right binding three nodes up; it has to be reinterpreted -//! against whatever schema is in scope at that node. Resolving bottom-up -//! pins each reference to "this exact column of this exact node's -//! already-derived output schema," so nothing downstream re-derives that scope. -//! 3. **It concentrates scoping logic in one place instead of ~6.** Every -//! downstream pass just compares/indexes `ColumnId`s — O(1), unambiguous. If -//! they worked on names instead, each would need its own qualifier-aware, -//! join-collision-aware name resolver, or risk silently binding to the wrong -//! `"service"`. -//! -//! Removing this resolution step and carrying `ColumnRef` everywhere would -//! therefore be a real regression for this repo's shape, not just a rename — -//! every one of those downstream passes would have to reimplement the scoping -//! this module already centralizes. - -use std::rc::Rc; - -use thiserror::Error; - -use super::agg_intent::AggIntent; -use super::column_resolution::{ - resolve_column_ref, resolve_column_refs, resolve_expr, resolve_group_keys_promql, ResolveError, -}; -use super::expr_ir::ColumnRef; -use super::query_expr::{ - aggregate_output_schema, any_measure_filtered, ConcatDiscriminatorKey, GroupKeys, Predicate, - ProjectItem, QueryExprError, Reduction, ResolvedQueryExpr, SortKey, UnresolvedQueryExpr, -}; -use super::schema::{ColumnId, Schema}; -use super::schema_resolver::SchemaResolver; - -/// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] DAG. -#[derive(Debug, Error)] -pub enum ResolveDAGError { - /// A column reference did not resolve against its in-scope schema. - #[error("column resolution failed: {0}")] - Resolve(#[from] ResolveError), - /// Deriving the schema of an already-resolved child failed (needed to - /// resolve positional column references against it). - #[error("schema derivation failed: {0}")] - Schema(#[from] QueryExprError), -} - -/// Resolve a whole [`UnresolvedQueryExpr`] DAG rooted at `dag` into canonical -/// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the -/// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the -/// result. -pub fn resolve_root(dag: &UnresolvedQueryExpr) -> Result { - resolve_root_with_inherited(dag, &[]) -} - -/// [`resolve_root`] with label names inherited from an enclosing scope seeded -/// into the leaf schema, used when re-binding a `BinaryOp` side (issue #52). -fn resolve_root_with_inherited( - dag: &UnresolvedQueryExpr, - inherited: &[String], -) -> Result { - let fallback = SchemaResolver::new().resolve_schema_with_inherited(dag, inherited); - let l3 = resolve(dag, &fallback)?; - Ok(super::canonicalize::canonicalize(l3)) -} - -/// The generic substitution walk: converts children first (bottom-up), then -/// resolves this node's own `ColumnRef`s against the *converted child's* -/// derived output schema — so a `JOIN`'s concatenated schema and a cross- -/// series aggregate's frozen-closed output bind to the right positions. -fn resolve( - dag: &UnresolvedQueryExpr, - fallback: &Schema, -) -> Result { - use super::query_expr::QueryExpr as QE; - Ok(match dag { - QE::Scan { - source, - predicates, - schema, - } => { - let schema = schema.clone().unwrap_or_else(|| fallback.clone()); - let predicates = predicates - .iter() - .map(|Predicate(e)| Ok(Predicate(Rc::new(resolve_expr(e, &schema)?)))) - .collect::, ResolveError>>()?; - QE::Scan { - source: source.clone(), - predicates, - schema, - } - } - - // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220) sitting at this operator-DAG position — resolved through - // `resolve_expr`, same as every other scalar position (`Predicate`, - // `ProjectItem.expr`, …), not the operator walk. In practice it's - // always a `Literal`, which has no `ColumnRef` to resolve, so - // `fallback` is never actually consulted here. - QE::PromqlScalarBridge(inner) => { - QE::PromqlScalarBridge(Rc::new(resolve_expr(inner, fallback)?)) - } - QE::EvalTimestamp => QE::EvalTimestamp, - QE::CurrentTimestamp => QE::CurrentTimestamp, - - QE::PromqlVectorFromScalar(child) => { - QE::PromqlVectorFromScalar(Rc::new(resolve(child, fallback)?)) - } - QE::PromqlScalarFromVector(child) => { - QE::PromqlScalarFromVector(Rc::new(resolve(child, fallback)?)) - } - - QE::PromqlRelabel { dst, value, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::PromqlRelabel { - dst: dst.clone(), - value: Rc::new(resolve_expr(value, &child_schema)?), - child: Rc::new(child), - } - } - - QE::PromqlInfoEnrich { selector, child } => QE::PromqlInfoEnrich { - selector: selector.clone(), - child: Rc::new(resolve(child, fallback)?), - }, - - QE::PromqlSeriesSample { by, kind, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::PromqlSeriesSample { - by: resolve_group_keys(by, &child_schema)?, - kind: *kind, - child: Rc::new(child), - } - } - - QE::Filter { pred, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::Filter { - pred: Predicate(Rc::new(resolve_expr(&pred.0, &child_schema)?)), - child: Rc::new(child), - } - } - - QE::Project { - cols, - qualifier, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let cols = cols - .iter() - .map(|item| -> Result { - Ok(ProjectItem { - alias: item.alias.clone(), - expr: resolve_expr(&item.expr, &child_schema)?, - }) - }) - .collect::, _>>()?; - QE::Project { - cols, - qualifier: qualifier.clone(), - child: Rc::new(child), - } - } - - QE::Aggregate { - reduction, - measures, - output_names, - filters, - having, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let reduction = resolve_reduction(reduction, &child_schema)?; - let measures = measures - .iter() - .map(|m| resolve_agg_intent(m, &child_schema)) - .collect::, ResolveError>>()?; - // A measure filter reads the rows being aggregated, so it binds - // against the child's schema, not the aggregate's output. - let filters = filters - .iter() - .map(|f| { - f.as_ref() - .map(|Predicate(p)| Ok(Predicate(Rc::new(resolve_expr(p, &child_schema)?)))) - .transpose() - }) - .collect::, ResolveError>>()?; - // One canonical spelling of "unfiltered" (empty), so structural - // equality and CSE never split on `[]` versus `[None, None]`. - let filters = if any_measure_filtered(&filters) { - filters - } else { - Vec::new() - }; - let having = having - .as_ref() - .map(|Predicate(h)| -> Result { - let out_schema = aggregate_output_schema( - &child_schema, - &reduction, - &measures, - output_names, - )?; - Ok(Predicate(Rc::new(resolve_expr(h, &out_schema)?))) - }) - .transpose()?; - QE::Aggregate { - reduction, - measures, - output_names: output_names.clone(), - filters, - having, - child: Rc::new(child), - } - } - - QE::Dedup { cols, child } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - QE::Dedup { - cols: resolve_column_refs(cols, &child_schema)?, - child: Rc::new(child), - } - } - - QE::Concat { - children, - discriminator_unique_key, - } => { - let children: Vec<_> = children - .iter() - .map(|c| resolve(c, fallback)) - .collect::, _>>()?; - // No front end asserts this today (issue #228 shipped the - // extension point ahead of a wired call site) — resolved here - // regardless, against the first resolved branch's own output - // schema, exactly the schema `output_schema`'s `Concat` arm - // derives the merged schema from, so a future direct - // `concat_with_discriminator` caller upstream of `resolve_root` - // gets a correctly positional `ConcatDiscriminatorKey` out the - // other side. - let discriminator_unique_key = discriminator_unique_key - .as_ref() - .map(|key| -> Result<_, ResolveDAGError> { - let schema = children - .first() - .ok_or(QueryExprError::EmptyConcat)? - .output_schema()?; - Ok(ConcatDiscriminatorKey::new( - resolve_column_ref(key.discriminator(), &schema)?, - resolve_column_refs(key.inner_key(), &schema)?, - )) - }) - .transpose()?; - QE::Concat { - children, - discriminator_unique_key, - } - } - - QE::Join { - kind, - pred, - left, - right, - } => { - // Each branch is bound independently, same reasoning as `BinaryOp` - // below — different leaves / label sets. - let left = resolve_root_with_inherited(left, &[])?; - let right = resolve_root_with_inherited(right, &[])?; - let mut concat = left.output_schema()?; - concat.fields.extend(right.output_schema()?.fields); - let pred = Predicate(Rc::new(resolve_expr(&pred.0, &concat)?)); - QE::Join { - kind: kind.clone(), - pred, - left: Rc::new(left), - right: Rc::new(right), - } - } - - QE::SetOp { - kind, - all, - left, - right, - } => QE::SetOp { - kind: kind.clone(), - all: *all, - left: Rc::new(resolve_root_with_inherited(left, &[])?), - right: Rc::new(resolve_root_with_inherited(right, &[])?), - }, - - QE::Sort { - keys, - partition_by, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let keys = keys - .iter() - .map(|k| -> Result { - Ok(SortKey { - expr: resolve_expr(&k.expr, &child_schema)?, - ascending: k.ascending, - nulls_first: k.nulls_first, - }) - }) - .collect::, _>>()?; - let partition_by = resolve_group_keys(partition_by, &child_schema)?; - QE::Sort { - keys, - partition_by, - child: Rc::new(child), - } - } - - QE::Limit { n, offset, child } => QE::Limit { - n: *n, - offset: *offset, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::PromqlSubquery { - range, - resolution, - child, - } => QE::PromqlSubquery { - range: *range, - resolution: *resolution, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::TimeRange { range, child } => QE::TimeRange { - range: *range, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::TimeShift { shift, child } => QE::TimeShift { - shift: *shift, - child: Rc::new(resolve(child, fallback)?), - }, - - QE::SQLWindowFunc { - func, - args, - partition_by, - order_by, - frame, - output_name, - child, - } => { - let child = resolve(child, fallback)?; - let child_schema = child.output_schema()?; - let args = args - .iter() - .map(|a| resolve_expr(a, &child_schema)) - .collect::, _>>()?; - let partition_by = resolve_group_keys(partition_by, &child_schema)?; - let order_by = order_by - .iter() - .map(|k| -> Result { - Ok(SortKey { - expr: resolve_expr(&k.expr, &child_schema)?, - ascending: k.ascending, - nulls_first: k.nulls_first, - }) - }) - .collect::, _>>()?; - QE::SQLWindowFunc { - func: func.clone(), - args, - partition_by, - order_by, - frame: frame.clone(), - output_name: output_name.clone(), - child: Rc::new(child), - } - } - - QE::BinaryOp { - op, - lhs, - rhs, - vector_match, - } => { - // A binary op's two sides may scan different metrics with - // different label sets, so each branch resolves against its OWN - // bound schema; but an independently-bound side still has to see - // label names an *enclosing* node references (issue #52). - let own = super::schema_resolver::collect_referenced_columns(dag); - let inherited: Vec = inherited_names(fallback) - .into_iter() - .filter(|n| !own.contains(n)) - .collect(); - QE::BinaryOp { - op: op.clone(), - lhs: Rc::new(resolve_root_with_inherited(lhs, &inherited)?), - rhs: Rc::new(resolve_root_with_inherited(rhs, &inherited)?), - vector_match: vector_match.clone(), - } - } - - // The scalar variants (issue #205) are never reached here directly — - // `resolve` only ever recurses into `child`/operator positions; - // every scalar position (`Predicate`, `ProjectItem.expr`, …) goes - // through `resolve_expr` instead, at the operator arm that owns it. - other @ (QE::Column(_) - | QE::Literal(_) - | QE::Compare { .. } - | QE::BoolAnd(_) - | QE::BoolOr(_) - | QE::Not(_) - | QE::IsNull(_) - | QE::IsNotNull(_) - | QE::Cast { .. } - | QE::InList { .. } - | QE::FunctionCall { .. } - | QE::Arithmetic { .. } - | QE::Case { .. }) => { - unreachable!("resolve reached a scalar QueryExpr variant directly: {other:?}") - } - }) -} - -/// The label names an enclosing scope's schema carries beyond the `(ts, -/// value)` floor. -fn inherited_names(schema: &Schema) -> Vec { - schema - .fields - .iter() - .filter(|c| c.name != "ts" && c.name != "value") - .map(|c| c.name.clone()) - .collect() -} - -/// Resolve a name-based [`GroupKeys`] into positional -/// [`GroupKeys`], preserving its `by`/`without` mode. -fn resolve_group_keys( - keys: &GroupKeys, - schema: &Schema, -) -> Result, ResolveError> { - let ids = resolve_column_refs(keys.keys(), schema)?; - Ok(if keys.is_without() { - GroupKeys::without(ids) - } else { - GroupKeys::by(ids) - }) -} - -/// Resolve a name-based [`Reduction`] into positional -/// [`Reduction`]. -/// -/// Uses [`resolve_group_keys_promql`] rather than the strict -/// [`resolve_group_keys`], unlike every other group-key site in `resolve` -/// (`PromqlSeriesSample.by`, `Sort.partition_by`, `SQLWindowFunc.partition_by`): a key -/// absent from a **closed** schema (e.g. the output of a nested cross-series -/// aggregate that collapsed the label) is provably absent from every row, so -/// PromQL drops it from the grouping rather than rejecting the query (issue -/// #53) — `sum(sum by (group) (m)) by (job)` is the canonical case, `job` -/// absent from the inner aggregate's closed `[group, sum]` output. Applied -/// uniformly to every `Aggregate`, not just PromQL's: SQL's `GROUP BY` keys -/// are always genuinely present (DataFusion validates the plan), so the -/// "drop instead of reject" branch is simply never exercised there — the -/// lenient resolver is a no-op difference for a SQL DAG, not a behavior -/// change. -fn resolve_reduction( - reduction: &Reduction, - schema: &Schema, -) -> Result, ResolveError> { - Ok(match reduction { - Reduction::Reduce(by) => { - let ids = resolve_group_keys_promql(by.keys(), schema)?; - Reduction::Reduce(if by.is_without() { - GroupKeys::without(ids) - } else { - GroupKeys::by(ids) - }) - } - Reduction::PerEntity => Reduction::PerEntity, - }) -} - -/// Resolve a name-based [`AggIntent`] into positional -/// [`AggIntent`] — every `col: Option` resolves to -/// `Option` (`None` stays `None`, the sample-value convention); -/// every other field carries straight through unchanged. -fn resolve_agg_intent( - intent: &AggIntent, - schema: &Schema, -) -> Result, ResolveError> { - let col = |c: &Option| -> Result, ResolveError> { - c.as_ref() - .map(|r| resolve_column_ref(r, schema)) - .transpose() - }; - Ok(match intent { - AggIntent::Count { accuracy } => AggIntent::Count { - accuracy: accuracy.clone(), - }, - AggIntent::PearsonCorr { left, right } => AggIntent::PearsonCorr { - left: resolve_column_ref(left, schema)?, - right: resolve_column_ref(right, schema)?, - }, - AggIntent::Sum { col: c } => AggIntent::Sum { col: col(c)? }, - AggIntent::Min { col: c } => AggIntent::Min { col: col(c)? }, - AggIntent::Max { col: c } => AggIntent::Max { col: col(c)? }, - AggIntent::Avg { col: c } => AggIntent::Avg { col: col(c)? }, - AggIntent::StdDev { col: c, population } => AggIntent::StdDev { - col: col(c)?, - population: *population, - }, - AggIntent::Variance { col: c, population } => AggIntent::Variance { - col: col(c)?, - population: *population, - }, - AggIntent::Quantile { - col: c, - q, - accuracy, - } => AggIntent::Quantile { - col: col(c)?, - q: *q, - accuracy: accuracy.clone(), - }, - AggIntent::TopK { k, accuracy } => AggIntent::TopK { - k: *k, - accuracy: accuracy.clone(), - }, - AggIntent::Cardinality { cols, accuracy } => AggIntent::Cardinality { - cols: cols - .iter() - .map(|c| resolve_column_ref(c, schema)) - .collect::>()?, - accuracy: accuracy.clone(), - }, - AggIntent::FrequencyL2 { col: c, accuracy } => AggIntent::FrequencyL2 { - col: col(c)?, - accuracy: accuracy.clone(), - }, - AggIntent::FrequencyEntropy { col: c, accuracy } => AggIntent::FrequencyEntropy { - col: col(c)?, - accuracy: accuracy.clone(), - }, - AggIntent::Rate => AggIntent::Rate, - AggIntent::IRate => AggIntent::IRate, - AggIntent::Increase => AggIntent::Increase, - AggIntent::Changes => AggIntent::Changes, - AggIntent::Delta => AggIntent::Delta, - AggIntent::IDelta => AggIntent::IDelta, - AggIntent::Deriv => AggIntent::Deriv, - AggIntent::Resets => AggIntent::Resets, - AggIntent::PredictLinear { seconds } => AggIntent::PredictLinear { seconds: *seconds }, - AggIntent::DoubleExpSmoothing { smoothing, trend } => AggIntent::DoubleExpSmoothing { - smoothing: *smoothing, - trend: *trend, - }, - AggIntent::HistogramCount => AggIntent::HistogramCount, - AggIntent::HistogramSum => AggIntent::HistogramSum, - AggIntent::HistogramAvg => AggIntent::HistogramAvg, - AggIntent::HistogramStdDev => AggIntent::HistogramStdDev, - AggIntent::HistogramStdVar => AggIntent::HistogramStdVar, - AggIntent::HistogramFraction { lower, upper } => AggIntent::HistogramFraction { - lower: *lower, - upper: *upper, - }, - AggIntent::HistogramQuantile { q, le } => AggIntent::HistogramQuantile { - q: *q, - le: resolve_column_ref(le, schema)?, - }, - AggIntent::Math(f) => AggIntent::Math(f.clone()), - AggIntent::Absent => AggIntent::Absent, - AggIntent::AbsentOverTime => AggIntent::AbsentOverTime, - AggIntent::PresentOverTime => AggIntent::PresentOverTime, - AggIntent::TimeFn(f) => AggIntent::TimeFn(*f), - AggIntent::Group => AggIntent::Group, - AggIntent::CountValues { label } => AggIntent::CountValues { - label: label.clone(), - }, - AggIntent::LastOverTime => AggIntent::LastOverTime, - AggIntent::FirstOverTime => AggIntent::FirstOverTime, - AggIntent::MadOverTime => AggIntent::MadOverTime, - AggIntent::TsOfMinOverTime => AggIntent::TsOfMinOverTime, - AggIntent::TsOfMaxOverTime => AggIntent::TsOfMaxOverTime, - AggIntent::TsOfFirstOverTime => AggIntent::TsOfFirstOverTime, - AggIntent::TsOfLastOverTime => AggIntent::TsOfLastOverTime, - AggIntent::Extension { ext_kind, payload } => AggIntent::Extension { - ext_kind: ext_kind.clone(), - payload: payload.clone(), - }, - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::pre_asap::expr_ir::CompareOpKind; - use crate::pre_asap::query_expr::{ - BinaryOpKind, QueryExpr, Source, VectorMatch, VectorMatchKind, - }; - - // A measure filter (#466) binds positionally against the aggregate's - // input, and a vector with no set entry collapses to the empty spelling. - #[test] - fn resolve_measure_filters_against_the_child_schema() { - use crate::pre_asap::expr_ir::ScalarValue; - use crate::pre_asap::query_expr::Predicate; - use crate::pre_asap::{DataType, Field, GroupKeys}; - use crate::types::AccuracyTarget; - let scan = || UnresolvedQueryExpr::Scan { - source: Source::Table { - table_ref: "metrics".into(), - }, - predicates: vec![], - schema: Some(Schema::new(vec![ - Field::plain("service", DataType::Utf8, false), - Field::plain("latency", DataType::Float64, false), - Field::plain("bytes", DataType::Int64, false), - ])), - }; - let aggregate = |filters| UnresolvedQueryExpr::Aggregate { - reduction: Reduction::Reduce(GroupKeys::by(vec![ColumnRef::Named("service".into())])), - measures: vec![ - AggIntent::Count { - accuracy: AccuracyTarget::Exact, - }, - AggIntent::Sum { - col: Some(ColumnRef::Named("bytes".into())), - }, - ], - output_names: vec![], - filters, - having: None, - child: Rc::new(scan()), - }; - let latency_gt_one = Predicate(Rc::new(UnresolvedQueryExpr::Compare { - left: Rc::new(UnresolvedQueryExpr::Column(ColumnRef::Named( - "latency".into(), - ))), - op: CompareOpKind::Gt, - right: Rc::new(UnresolvedQueryExpr::Literal(ScalarValue::Float64(1.0))), - })); - - let resolved = resolve_root(&aggregate(vec![Some(latency_gt_one), None])).unwrap(); - let QueryExpr::Aggregate { filters, .. } = &resolved else { - unreachable!() - }; - let [Some(Predicate(first)), None] = filters.as_slice() else { - panic!("expected one filtered and one unfiltered measure, got {filters:?}"); - }; - assert!( - matches!(first.as_ref(), QueryExpr::Compare { left, .. } - if matches!(left.as_ref(), QueryExpr::Column(1))), - "latency is input column 1, got {first:?}" - ); - - let resolved = resolve_root(&aggregate(vec![None, None])).unwrap(); - let QueryExpr::Aggregate { filters, .. } = &resolved else { - unreachable!() - }; - assert!(filters.is_empty()); - } - - // Both sides resolve with qualifiers; an unknown right input is an error. - #[test] - fn resolve_pearson_corr_inputs() { - use crate::pre_asap::{DataType, Field}; - let schema = Schema::new(vec![ - Field::plain("x", DataType::Float64, true).with_table("a"), - Field::plain("x", DataType::Float64, true).with_table("b"), - ]); - let intent = AggIntent::PearsonCorr { - left: ColumnRef::Qualified { - table: "a".into(), - name: "x".into(), - }, - right: ColumnRef::Qualified { - table: "b".into(), - name: "x".into(), - }, - }; - assert_eq!( - resolve_agg_intent(&intent, &schema).unwrap(), - AggIntent::PearsonCorr { left: 0, right: 1 } - ); - let missing = AggIntent::PearsonCorr { - left: ColumnRef::Qualified { - table: "a".into(), - name: "x".into(), - }, - right: ColumnRef::Named("missing".into()), - }; - assert!(resolve_agg_intent(&missing, &schema).is_err()); - } - - // Every leg resolves independently, qualifiers included; one unknown leg - // fails rather than silently shortening the tuple. - #[test] - fn resolve_distinct_tuple_columns() { - use crate::pre_asap::{DataType, Field}; - use crate::types::AccuracyTarget; - let schema = Schema::new(vec![ - Field::plain("k", DataType::Int64, true).with_table("a"), - Field::plain("k", DataType::Int64, true).with_table("b"), - ]); - let qualified = |table: &str| ColumnRef::Qualified { - table: table.into(), - name: "k".into(), - }; - let intent = AggIntent::Cardinality { - cols: vec![qualified("b"), qualified("a")], - accuracy: AccuracyTarget::Exact, - }; - assert_eq!( - resolve_agg_intent(&intent, &schema).unwrap(), - AggIntent::Cardinality { - cols: vec![1, 0], - accuracy: AccuracyTarget::Exact, - } - ); - let missing = AggIntent::Cardinality { - cols: vec![qualified("a"), ColumnRef::Named("missing".into())], - accuracy: AccuracyTarget::Exact, - }; - assert!(resolve_agg_intent(&missing, &schema).is_err()); - } - - /// `resolve_root` over a `BinaryOp { , PromqlScalarBridge, vector_match }` - /// (issue #220): the bridged scalar operand resolves through the same - /// generic walk as every other node (its `Literal` child has no - /// `ColumnRef` to resolve, so it comes through unchanged), the vector - /// side's `ColumnRef`s resolve positionally, and the `VectorMatch` - /// modifier on the relational binary-op path survives resolution - /// untouched — Instance 2 of #220 (`BinaryOp` vs `Compare`/`Arithmetic`) - /// is out of scope for this change, so this pins that its behavior is - /// unaffected by the Instance-1 collapse. - #[test] - fn resolve_root_threads_a_scalar_bridge_operand_and_preserves_vector_match() { - let vm = VectorMatch { - kind: VectorMatchKind::Ignoring, - labels: vec!["job".into()], - grouping: None, - }; - let unresolved: UnresolvedQueryExpr = QueryExpr::BinaryOp { - op: BinaryOpKind::Compare(CompareOpKind::Gt), - lhs: Rc::new(UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { - metric: "up".into(), - }, - predicates: vec![], - schema: None, - }), - rhs: Rc::new(UnresolvedQueryExpr::promql_scalar(1.0)), - vector_match: Some(vm.clone()), - }; - - let resolved = resolve_root(&unresolved).expect("resolves"); - let QueryExpr::BinaryOp { - lhs, - rhs, - vector_match, - .. - } = &resolved - else { - panic!("expected a resolved BinaryOp, got {resolved:?}"); - }; - assert!(matches!(lhs.as_ref(), QueryExpr::Scan { .. })); - assert_eq!(rhs.as_promql_scalar(), Some(1.0)); - assert_eq!(vector_match.as_ref(), Some(&vm)); - - // Schema derivation still follows the vector side post-resolution. - assert_eq!( - resolved.output_schema().unwrap(), - lhs.output_schema().unwrap() - ); - } - - /// Issue #228 review, end-to-end: `resolve_root` over a `Concat` whose - /// discriminator column is referenced *nowhere else* in the DAG, with a - /// schema-less (usage-derived) leaf `Scan` in the first branch — exactly - /// the scenario the review flagged. Before the `schema_resolver.rs` fix, the - /// SchemaResolver's fallback schema wouldn't contain `phi` at all, and this - /// `resolve_column_ref` call would fail `NotFound` for a column the - /// caller correctly named. It must resolve cleanly, and the resolved - /// `ConcatDiscriminatorKey` must carry the *positional* `ColumnId`s of - /// the branch's own (usage-derived) schema. - #[test] - fn resolve_root_seeds_and_resolves_an_otherwise_unreferenced_discriminator_column() { - let branch = || UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { metric: "m".into() }, - predicates: vec![], - schema: None, - }; - let unresolved = UnresolvedQueryExpr::concat_with_discriminator( - vec![branch(), branch()], - ColumnRef::Named("phi".into()), - vec![ColumnRef::Named("host".into())], - ); - - let resolved = resolve_root(&unresolved).expect("resolves"); - let QueryExpr::Concat { - children, - discriminator_unique_key, - } = &resolved - else { - panic!("expected a resolved Concat, got {resolved:?}"); - }; - let schema = children[0].output_schema().unwrap(); - let key = discriminator_unique_key - .as_ref() - .expect("discriminator key survives resolution"); - assert_eq!(*key.discriminator(), schema.column_id("phi").unwrap()); - assert_eq!( - key.inner_key().to_vec(), - vec![schema.column_id("host").unwrap()] - ); - } -} diff --git a/crates/types/src/pre_asap/scalar_type_rules.rs b/crates/types/src/pre_asap/scalar_type_rules.rs deleted file mode 100644 index 44eaa250c..000000000 --- a/crates/types/src/pre_asap/scalar_type_rules.rs +++ /dev/null @@ -1,519 +0,0 @@ -//! Shared type rules for structural map scalar expressions. -//! Execution must separately implement the documented ordering/default semantics. -use super::schema::DataType; - -/// Names are resolved once against this closed builtin set; unknown functions -/// remain outside these type rules. Map access keeps the first duplicate key -/// and returns the value type's default when the key is absent. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum MapScalarFunction { - Construct, - Concat, - Access, -} -impl MapScalarFunction { - pub fn from_name(name: &str) -> Option { - match name.to_ascii_lowercase().as_str() { - "map" => Some(Self::Construct), - "mapconcat" => Some(Self::Concat), - "asap_map_access" => Some(Self::Access), - _ => None, - } - } - pub fn output_type(self, args: &[(DataType, bool)]) -> Result<(DataType, bool), String> { - match self { - Self::Construct => { - let (pairs, remainder) = args.as_chunks::<2>(); - if !remainder.is_empty() { - return Err("map construction requires key/value pairs".into()); - } - let mut key = DataType::Null; - let mut value = DataType::Null; - let mut value_nullable = false; - for pair in pairs { - if pair[0].1 || pair[0].0 == DataType::Null { - return Err("map keys must be non-null".into()); - } - key = common_type(&key, &pair[0].0)?; - value = common_type(&value, &pair[1].0)?; - value_nullable |= pair[1].1 || pair[1].0 == DataType::Null; - } - Ok(( - DataType::Map { - key: Box::new(key), - value: Box::new(value), - value_nullable, - }, - false, - )) - } - Self::Concat => { - if args.is_empty() { - return Err("map concatenation requires at least one map".into()); - } - let mut key = DataType::Null; - let mut value = DataType::Null; - let mut value_nullable = false; - for (argument, nullable) in args { - if *nullable { - return Err("nullable map containers are unsupported".into()); - } - let DataType::Map { - key: k, - value: v, - value_nullable: n, - } = argument - else { - return Err("map concatenation requires map arguments".into()); - }; - key = common_type(&key, k)?; - value = common_type(&value, v)?; - value_nullable |= *n; - } - Ok(( - DataType::Map { - key: Box::new(key), - value: Box::new(value), - value_nullable, - }, - false, - )) - } - Self::Access => { - let [(map, map_nullable), (index, index_nullable)] = args else { - return Err("map access requires a map and key".into()); - }; - if *map_nullable { - return Err("nullable map containers are unsupported".into()); - } - let DataType::Map { - key, - value, - value_nullable, - } = map - else { - return Err("map access requires a map".into()); - }; - if **key == DataType::Null { - return Err("map lookup requires a concrete map key type".into()); - } - if *index != DataType::Null && common_type(key, index)? != **key { - return Err("map lookup key requires a lossy or unsupported coercion".into()); - } - Ok(( - (**value).clone(), - *value_nullable - || *index_nullable - || *index == DataType::Null - || **value == DataType::Null, - )) - } - } - } -} -fn common_type(left: &DataType, right: &DataType) -> Result { - if left == right || *right == DataType::Null { - return Ok(left.clone()); - } - if *left == DataType::Null { - return Ok(right.clone()); - } - Err(format!( - "incompatible map scalar types: {left:?} and {right:?}" - )) -} - -#[cfg(test)] -mod tests { - use super::*; - #[test] - fn empty_map_is_bottom_typed_and_concat_resolves_it() { - let empty = MapScalarFunction::Construct.output_type(&[]).unwrap(); - assert_eq!( - empty, - ( - DataType::Map { - key: Box::new(DataType::Null), - value: Box::new(DataType::Null), - value_nullable: false - }, - false - ) - ); - assert!(MapScalarFunction::Access - .output_type(&[empty.clone(), (DataType::Utf8, false)]) - .is_err()); - let concrete = MapScalarFunction::Construct - .output_type(&[(DataType::Utf8, false), (DataType::Int64, false)]) - .unwrap(); - assert_eq!( - MapScalarFunction::Concat - .output_type(&[empty, concrete.clone()]) - .unwrap(), - concrete.clone() - ); - assert_eq!( - MapScalarFunction::Access - .output_type(&[concrete, (DataType::Utf8, false)]) - .unwrap(), - (DataType::Int64, false) - ); - } - #[test] - fn nullable_lookup_and_invalid_signatures_are_explicit() { - let map = MapScalarFunction::Construct - .output_type(&[(DataType::Utf8, false), (DataType::Int64, true)]) - .unwrap(); - assert_eq!( - MapScalarFunction::Access - .output_type(&[map, (DataType::Utf8, false)]) - .unwrap(), - (DataType::Int64, true) - ); - let nonnull = MapScalarFunction::Construct - .output_type(&[(DataType::Utf8, false), (DataType::Int64, false)]) - .unwrap(); - assert_eq!( - MapScalarFunction::Access - .output_type(&[nonnull, (DataType::Utf8, true)]) - .unwrap(), - (DataType::Int64, true) - ); - assert!(MapScalarFunction::Construct - .output_type(&[(DataType::Utf8, true), (DataType::Int64, false)]) - .is_err()); - assert!(MapScalarFunction::Concat - .output_type(&[(DataType::Int64, false)]) - .is_err()); - // ClickHouse can choose Variant(Float64, Int64), not lossless Float64. - assert!(MapScalarFunction::Construct - .output_type(&[ - (DataType::Utf8, false), - (DataType::Int64, false), - (DataType::Utf8, false), - (DataType::Float64, false), - ]) - .is_err()); - } -} - -#[cfg(test)] -mod projection_tests { - use super::*; - use crate::pre_asap::{Field, ProjectItem, QueryExpr, ScalarValue, Schema, Source}; - use std::rc::Rc; - fn project(expr: QueryExpr) -> QueryExpr { - QueryExpr::Project { - cols: vec![ProjectItem { - alias: Some("result".into()), - expr, - }], - qualifier: None, - child: Rc::new(QueryExpr::Scan { - source: Source::Table { - table_ref: "t".into(), - }, - predicates: vec![], - schema: Schema::new(vec![ - Field::plain("k", DataType::Utf8, false), - Field::plain("v", DataType::Int64, true), - ]), - }), - } - } - #[test] - fn canonical_projection_uses_map_signature_and_rejects_invalid_arity() { - let map = QueryExpr::FunctionCall { - name: "map".into(), - args: vec![QueryExpr::Column(0), QueryExpr::Column(1)], - }; - let schema = project(map.clone()).output_schema().unwrap(); - assert_eq!( - schema.fields[0].dtype, - DataType::Map { - key: Box::new(DataType::Utf8), - value: Box::new(DataType::Int64), - value_nullable: true - } - ); - assert!(!schema.fields[0].nullable); - let lookup = QueryExpr::FunctionCall { - name: "asap_map_access".into(), - args: vec![map, QueryExpr::Literal(ScalarValue::Utf8("missing".into()))], - }; - assert_eq!( - project(lookup).output_schema().unwrap().fields[0], - Field::plain("result", DataType::Int64, true) - ); - assert!(project(QueryExpr::FunctionCall { - name: "map".into(), - args: vec![QueryExpr::Column(0)] - }) - .output_schema() - .is_err()); - } -} - -/// Resolve the bounded canonical `asap_struct_field(struct, selector)` operation. -/// Selectors are positive 1-based literal ordinals or exact literal field names. -/// The existing Struct fields remain the sole authority for type/nullability. -/// Dynamic/negative/defaulted selectors and nullable containers are intentionally -/// unsupported here; this is not a claim of complete native tupleElement support. -pub fn struct_field_type( - args: &[super::QueryExpr], - schema: &super::Schema, -) -> Result<(DataType, bool), String> { - use super::{QueryExpr, ScalarValue}; - let [input, selector] = args else { - return Err("struct field access requires a struct and constant selector".into()); - }; - let (dtype, nullable) = input - .scalar_type(schema) - .map_err(|error| error.to_string())?; - if nullable { - return Err("nullable struct container access is unsupported".into()); - } - let DataType::Struct { fields } = dtype else { - return Err("struct field access requires a Struct input".into()); - }; - let field = match selector { - QueryExpr::Literal(ScalarValue::Int64(index)) if *index > 0 => usize::try_from(*index - 1) - .ok() - .and_then(|index| fields.get(index)) - .ok_or("struct field ordinal is out of bounds")?, - QueryExpr::Literal(ScalarValue::Utf8(name)) => { - let mut matches = fields.iter().filter(|field| field.name == *name); - let field = matches.next().ok_or("struct field name does not exist")?; - if matches.next().is_some() { - return Err("struct field name is ambiguous".into()); - } - field - } - _ => { - return Err( - "struct field selector must be a positive ordinal or field-name literal".into(), - ) - } - }; - Ok((field.dtype.clone(), field.nullable)) -} - -#[cfg(test)] -mod struct_field_tests { - use super::*; - use crate::pre_asap::{Field, FieldDataType, QueryExpr, ScalarValue, Schema}; - fn schema() -> Schema { - Schema::new(vec![Field::plain( - "record", - DataType::Struct { - fields: vec![ - Field::new("ts", DataType::Int64, false), - Field::new( - "values", - DataType::List { - element: Box::new(Field::new("item", DataType::Float64, true)), - }, - true, - ), - ], - }, - false, - )]) - } - fn access(selector: QueryExpr) -> QueryExpr { - QueryExpr::FunctionCall { - name: "asap_struct_field".into(), - args: vec![QueryExpr::Column(0), selector], - } - } - #[test] - fn field_access_reuses_nested_field_type_and_nullability() { - let schema = schema(); - assert_eq!( - access(QueryExpr::Literal(ScalarValue::Int64(1))) - .scalar_type(&schema) - .unwrap(), - (DataType::Int64, false) - ); - let named = access(QueryExpr::Literal(ScalarValue::Utf8("values".into()))); - let ordinal = access(QueryExpr::Literal(ScalarValue::Int64(2))); - assert_eq!( - named.scalar_type(&schema).unwrap(), - ordinal.scalar_type(&schema).unwrap() - ); - assert_eq!( - named.scalar_type(&schema).unwrap(), - ( - DataType::List { - element: Box::new(Field::new("item", DataType::Float64, true)) - }, - true - ) - ); - let roundtrip: QueryExpr = - serde_json::from_str(&serde_json::to_string(&named).unwrap()).unwrap(); - assert_eq!(roundtrip, named); - } - #[test] - fn unsupported_field_access_is_an_error_not_placeholder_typing() { - for selector in [ - QueryExpr::Column(0), - QueryExpr::Literal(ScalarValue::Int64(0)), - QueryExpr::Literal(ScalarValue::Int64(-1)), - QueryExpr::Literal(ScalarValue::Int64(3)), - QueryExpr::Literal(ScalarValue::Utf8("missing".into())), - ] { - assert!(access(selector).scalar_type(&schema()).is_err()); - } - let mut ambiguous = schema(); - if let FieldDataType::Plain(DataType::Struct { fields }) = &mut ambiguous.fields[0].dtype { - fields.push(Field::new("ts", DataType::Utf8, false)); - } - assert!(access(QueryExpr::Literal(ScalarValue::Utf8("ts".into()))) - .scalar_type(&ambiguous) - .is_err()); - let mut nullable = schema(); - nullable.fields[0].nullable = true; - assert!(access(QueryExpr::Literal(ScalarValue::Int64(1))) - .scalar_type(&nullable) - .is_err()); - } -} - -/// Canonical element lookup over a declared Map or List. Map lookup retains its -/// existing key/default contract. List lookup is one-based, supports negative -/// indices, and returns the declared element default when a dynamic index is -/// out of range. Literal zero is conservatively rejected because native array -/// behavior depends on whether the input array is constant. Nullable containers -/// are unsupported; nullable indices produce nullable results. -pub fn element_access_type( - args: &[super::QueryExpr], - schema: &super::Schema, -) -> Result<(DataType, bool), String> { - use super::{QueryExpr, ScalarValue}; - let [input, index] = args else { - return Err("element access requires a collection and index".into()); - }; - let source = input.scalar_type(schema).map_err(|e| e.to_string())?; - let key = index.scalar_type(schema).map_err(|e| e.to_string())?; - match &source.0 { - DataType::Map { .. } => MapScalarFunction::Access.output_type(&[source, key]), - DataType::List { element } => { - if source.1 { - return Err("nullable List container access is unsupported".into()); - } - if !matches!(key.0, DataType::Int64 | DataType::Null) { - return Err("List index must have integer type".into()); - } - if matches!(index, QueryExpr::Literal(ScalarValue::Int64(0))) { - return Err( - "literal zero List index is unsupported without constant-array proof".into(), - ); - } - Ok(( - element.dtype.clone(), - element.nullable || key.1 || key.0 == DataType::Null, - )) - } - _ => Err("element access requires a Map or List".into()), - } -} - -#[cfg(test)] -mod element_access_tests { - use super::*; - use crate::pre_asap::{Field, QueryExpr, ScalarValue, Schema}; - fn access(index: QueryExpr) -> QueryExpr { - QueryExpr::FunctionCall { - name: "asap_element_access".into(), - args: vec![QueryExpr::Column(0), index], - } - } - #[test] - fn list_index_preserves_nested_element_metadata() { - let element = DataType::Struct { - fields: vec![ - Field::new("ts", DataType::Int64, false), - Field::new("value", DataType::Float64, true), - ], - }; - let schema = Schema::new(vec![ - Field::plain( - "samples", - DataType::List { - element: Box::new(Field::new("item", element.clone(), false)), - }, - false, - ), - Field::plain("i", DataType::Int64, true), - ]); - for index in [1, -1, 100] { - assert_eq!( - access(QueryExpr::Literal(ScalarValue::Int64(index))) - .scalar_type(&schema) - .unwrap(), - (element.clone(), false) - ); - } - assert_eq!( - access(QueryExpr::Column(1)).scalar_type(&schema).unwrap(), - (element.clone(), true) - ); - assert!(access(QueryExpr::Literal(ScalarValue::Int64(0))) - .scalar_type(&schema) - .is_err()); - assert!(access(QueryExpr::Literal(ScalarValue::Float64(1.0))) - .scalar_type(&schema) - .is_err()); - let nested = QueryExpr::FunctionCall { - name: "asap_struct_field".into(), - args: vec![ - access(QueryExpr::Literal(ScalarValue::Int64(1))), - QueryExpr::Literal(ScalarValue::Int64(2)), - ], - }; - assert_eq!( - nested.scalar_type(&schema).unwrap(), - (DataType::Float64, true) - ); - let roundtrip: QueryExpr = - serde_json::from_value(serde_json::to_value(&nested).unwrap()).unwrap(); - assert_eq!(roundtrip, nested); - } - #[test] - fn generic_map_lookup_reuses_legacy_signature() { - let schema = Schema::new(vec![Field::plain( - "m", - DataType::Map { - key: Box::new(DataType::Utf8), - value: Box::new(DataType::Int64), - value_nullable: false, - }, - false, - )]); - let key = QueryExpr::Literal(ScalarValue::Utf8("k".into())); - let legacy = QueryExpr::FunctionCall { - name: "asap_map_access".into(), - args: vec![QueryExpr::Column(0), key.clone()], - }; - assert_eq!( - access(key).scalar_type(&schema).unwrap(), - legacy.scalar_type(&schema).unwrap() - ); - } -} - -/// Closed, namespaced contracts for PromQL pointwise float functions. -/// Date functions consume Unix seconds; `timestamp` remains a sample-selection -/// operation because its operand is a sample timestamp rather than its value. -pub fn promql_function_arity(name: &str) -> Option { - Some(match name.strip_prefix("promql_")? { - "abs" | "ceil" | "floor" | "exp" | "ln" | "log2" | "log10" | "sqrt" | "sgn" | "sin" - | "cos" | "tan" | "asin" | "acos" | "atan" | "sinh" | "cosh" | "tanh" | "asinh" - | "acosh" | "atanh" | "deg" | "rad" | "minute" | "hour" | "day_of_week" - | "day_of_month" | "day_of_year" | "month" | "year" | "days_in_month" => 1, - "round" | "clamp_min" | "clamp_max" => 2, - "clamp" => 3, - _ => return None, - }) -} diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs deleted file mode 100644 index a9afff2cb..000000000 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ /dev/null @@ -1,492 +0,0 @@ -//! The **SchemaResolver** — name resolution as an explicit pass. -//! -//! [`SchemaResolver::resolve_schema`] produces the complete, self-contained [`Schema`] every -//! `ColumnId` in the canonical DAG indexes into. [`resolve`](super::resolve) -//! then becomes purely structural: it threads the SchemaResolver's schema and -//! positional resolution downstream is **total**. -//! -//! The default [`UsageDerivedCatalog`] knows nothing — every schema is derived -//! purely from the query's own usage. That is the honest state for the -//! observability domain (metric label sets are open-ended). A registry-backed -//! `SchemaCatalog` is future work; the `SchemaResolver` pass does not change when it -//! lands, only the catalog impl swaps. - -use super::expr_ir::ColumnRef; -use super::query_expr::UnresolvedQueryExpr; -use super::schema::{DataType, Field, Schema}; - -/// The DB / source-schema metadata source — resolves a source (metric / -/// table) name to its known columns. -/// Source of truth for a source's columns — the "catalog". `SqlCatalog` backs -/// it for SQL; PromQL uses [`UsageDerivedCatalog`] (returns `None`) until a -/// registry-backed impl (returning a metric's known label set) drops in here. -/// Distinct from `Scan.schema`, which is the *resolved* binding schema this -/// feeds — the catalog is the input, the schema is the result. Even a -/// registry-backed PromQL catalog yields an **open** schema -/// ([`Schema::closed`] `= false`): a metric's -/// labels are per-series and time-varying, so the registry is a superset hint, -/// not a per-row contract. -pub trait SchemaCatalog { - /// Columns known for `source`. `None` when unknown — the [`SchemaResolver`] then - /// falls back to a usage-derived column set. - fn columns_for(&self, source: &str) -> Option>; -} - -/// The default catalog: knows nothing. Every schema the [`SchemaResolver`] produces -/// is derived purely from the query's own usage. -pub struct UsageDerivedCatalog; - -impl SchemaCatalog for UsageDerivedCatalog { - fn columns_for(&self, _source: &str) -> Option> { - None - } -} - -/// The explicit name-resolution pass. -pub struct SchemaResolver { - catalog: C, -} - -impl Default for SchemaResolver { - fn default() -> Self { - Self::new() - } -} - -impl SchemaResolver { - pub fn new() -> Self { - Self { - catalog: UsageDerivedCatalog, - } - } -} - -impl SchemaResolver { - pub fn with_catalog(catalog: C) -> Self { - Self { catalog } - } - - /// Resolve the complete [`Schema`] in scope for a query rooted at `dag`. - /// - /// Contains the time axis, the synthetic `value` column, and one column - /// per distinct name referenced anywhere in the DAG — so positional - /// `ColumnId` resolution downstream is total. - pub fn resolve_schema(&self, dag: &UnresolvedQueryExpr) -> Schema { - self.resolve_schema_with_inherited(dag, &[]) - } - - /// Like [`resolve_schema`](Self::resolve_schema), but also seeds `inherited` label names that are - /// referenced by an **enclosing** scope rather than by `dag` itself. This is - /// how an independently-bound `BinaryOp` side (each side re-binds against its - /// own sub-DAG) still sees an outer aggregate's group keys — e.g. the - /// `__name__` / `job` in `sum by (__name__)(a or b)`, which appear in neither - /// side's own matchers (issue #52). - pub fn resolve_schema_with_inherited( - &self, - dag: &UnresolvedQueryExpr, - inherited: &[String], - ) -> Schema { - let mut columns: Vec = leftmost_scan_name(dag) - .and_then(|name| self.catalog.columns_for(name)) - .unwrap_or_else(default_leaf_columns); - - // Ensure the (ts, value) floor is present. - for floor in default_leaf_columns() { - if !columns.iter().any(|c| c.name == floor.name) { - columns.push(floor); - } - } - - // Append one column per referenced-but-unknown name (group keys etc.), - // plus any inherited-from-enclosing-scope names. - let referenced = collect_referenced_columns(dag); - for name in referenced.iter().chain(inherited) { - if !columns.iter().any(|c| c.name == *name) { - columns.push(Field::plain(name.clone(), DataType::Utf8, true)); - } - } - - let time_index = columns.iter().position(|c| c.name == "ts"); - Schema { - fields: columns, - time_index, - unique_keys: Vec::new(), - // Usage-derived (schemaless PromQL): the metric's full label set is - // open and runtime-only, so this lists only what the query references. - closed: false, - } - } -} - -/// The conventional PromQL leaf shape: `(ts: Timestamp, value: Float64)`. -fn default_leaf_columns() -> Vec { - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - ] -} - -/// Push a `ColumnRef`'s bare name (the schema-seedable identifier). `Qualified` -/// collapses to its `name`; `SampleValue`/`Wildcard` carry no name. -fn push_ref_name(c: &ColumnRef, out: &mut Vec) { - match c { - ColumnRef::Named(n) => out.push(n.clone()), - ColumnRef::Qualified { name, .. } => out.push(name.clone()), - ColumnRef::SampleValue | ColumnRef::Wildcard => {} - } -} - -/// The leftmost `Scan`'s source name in a canonical (`UnresolvedQueryExpr`) DAG — -/// the [`collect_referenced_columns`] counterpart to what a dedicated -/// `Source` leaf type would carry as a method; the canonical DAG's `Scan` -/// leaf needs this walk written out instead. -fn leftmost_scan_name(dag: &UnresolvedQueryExpr) -> Option<&str> { - use UnresolvedQueryExpr as QE; - match dag { - QE::Scan { source, .. } => Some(match source { - super::query_expr::Source::TimeSeries { metric } => metric.as_str(), - super::query_expr::Source::Table { table_ref } => table_ref.as_str(), - }), - // A scalar bridge's child is a scalar-sub-language leaf (in practice - // always a `Literal`, issue #220) — never a `Scan`, same as - // `EvalTimestamp`. - QE::PromqlScalarBridge(_) | QE::EvalTimestamp | QE::CurrentTimestamp => None, - QE::PromqlVectorFromScalar(child) | QE::PromqlScalarFromVector(child) => { - leftmost_scan_name(child) - } - QE::PromqlRelabel { child, .. } - | QE::PromqlInfoEnrich { child, .. } - | QE::PromqlSeriesSample { child, .. } - | QE::Filter { child, .. } - | QE::Project { child, .. } - | QE::Aggregate { child, .. } - | QE::Dedup { child, .. } - | QE::Sort { child, .. } - | QE::Limit { child, .. } - | QE::PromqlSubquery { child, .. } - | QE::TimeRange { child, .. } - | QE::TimeShift { child, .. } - | QE::SQLWindowFunc { child, .. } => leftmost_scan_name(child), - QE::Concat { children, .. } => children.first().and_then(leftmost_scan_name), - QE::Join { left, .. } | QE::SetOp { left, .. } | QE::BinaryOp { lhs: left, .. } => { - leftmost_scan_name(left) - } - // The scalar variants (issue #205) never appear as a direct - // `leftmost_scan_name` target — every reachable one sits behind a - // wrapper field (`Predicate`, `ProjectItem`, …) this walk never - // descends into; it only follows the relational skeleton. - QE::Column(_) - | QE::Literal(_) - | QE::Compare { .. } - | QE::BoolAnd(_) - | QE::BoolOr(_) - | QE::Not(_) - | QE::IsNull(_) - | QE::IsNotNull(_) - | QE::Cast { .. } - | QE::InList { .. } - | QE::FunctionCall { .. } - | QE::Arithmetic { .. } - | QE::Case { .. } => None, - } -} - -/// Collect every distinct column name referenced anywhere in `dag` that -/// resolves positionally — every place a front end constructing -/// [`QueryExpr`](super::query_expr::QueryExpr) directly (issue -/// #179) puts a name-based reference: `Scan.predicates`, `Aggregate`'s -/// `reduction`/`having`/per-measure `col`, `Dedup.cols`, `PromqlSeriesSample.by`, -/// `Filter.pred`, `Project.cols`, `Sort.keys`/`partition_by`, -/// `SQLWindowFunc.args`/`partition_by`/`order_by`, `Join.pred`, `PromqlRelabel.value`. -/// The SchemaResolver seeds these into the usage-derived leaf so positional -/// resolution downstream is total. -pub(crate) fn collect_referenced_columns(dag: &UnresolvedQueryExpr) -> Vec { - use UnresolvedQueryExpr as QE; - fn named(expr: &UnresolvedQueryExpr, out: &mut Vec) { - for c in expr.columns_referenced() { - push_ref_name(c, out); - } - } - fn group_keys(g: &super::query_expr::GroupKeys, out: &mut Vec) { - g.keys().iter().for_each(|k| push_ref_name(k, out)); - } - fn measure_cols(measures: &[super::agg_intent::AggIntent], out: &mut Vec) { - for m in measures { - for c in m.input_cols() { - push_ref_name(&c, out); - } - } - } - fn walk(node: &UnresolvedQueryExpr, out: &mut Vec) { - match node { - QE::Scan { predicates, .. } => { - for super::query_expr::Predicate(p) in predicates { - named(p, out); - } - } - QE::Aggregate { - reduction, - measures, - filters, - having, - child, - .. - } => { - if let super::query_expr::Reduction::Reduce(by) = reduction { - group_keys(by, out); - } - measure_cols(measures, out); - for super::query_expr::Predicate(f) in filters.iter().flatten() { - named(f, out); - } - if let Some(super::query_expr::Predicate(h)) = having { - named(h, out); - } - walk(child, out); - } - QE::Dedup { cols, child } => { - cols.iter().for_each(|c| push_ref_name(c, out)); - walk(child, out); - } - QE::PromqlSeriesSample { by, child, .. } => { - group_keys(by, out); - walk(child, out); - } - QE::Filter { pred, child } => { - named(&pred.0, out); - walk(child, out); - } - QE::Project { cols, child, .. } => { - for item in cols { - named(&item.expr, out); - } - walk(child, out); - } - QE::Sort { - keys, - partition_by, - child, - } => { - for k in keys { - named(&k.expr, out); - } - group_keys(partition_by, out); - walk(child, out); - } - QE::SQLWindowFunc { - args, - partition_by, - order_by, - child, - .. - } => { - for a in args { - named(a, out); - } - group_keys(partition_by, out); - for k in order_by { - named(&k.expr, out); - } - walk(child, out); - } - QE::PromqlRelabel { value, child, .. } => { - named(value, out); - walk(child, out); - } - QE::Join { - pred, left, right, .. - } => { - named(&pred.0, out); - walk(left, out); - walk(right, out); - } - QE::EvalTimestamp | QE::CurrentTimestamp => {} - // The bridged child is a genuine scalar-sub-language position now - // (issue #220) — peel its column refs off with `named`, same as - // every other scalar-typed field (`Scan.predicates`, - // `Filter.pred`, …). In practice it's always a `Literal`, which - // references no columns, so this is a no-op today. - QE::PromqlScalarBridge(inner) => named(inner, out), - QE::PromqlVectorFromScalar(child) | QE::PromqlScalarFromVector(child) => { - walk(child, out) - } - QE::PromqlInfoEnrich { child, .. } - | QE::Limit { child, .. } - | QE::PromqlSubquery { child, .. } - | QE::TimeRange { child, .. } - | QE::TimeShift { child, .. } => walk(child, out), - QE::Concat { - children, - discriminator_unique_key, - } => { - // Same treatment as `Dedup.cols` above: an own-field - // `ColumnRef` must be seeded here too, or a discriminator - // column that isn't otherwise referenced anywhere else in - // the DAG (plausible — a raw usage-derived label, not one a - // `Project`/relabel freshly created) is absent from the - // SchemaResolver's usage-derived fallback schema, and - // `resolve.rs`'s later `resolve_column_ref` call fails with - // `NotFound` for a column the caller correctly named. - if let Some(key) = discriminator_unique_key { - push_ref_name(key.discriminator(), out); - key.inner_key().iter().for_each(|c| push_ref_name(c, out)); - } - children.iter().for_each(|c| walk(c, out)); - } - QE::SetOp { left, right, .. } => { - walk(left, out); - walk(right, out); - } - QE::BinaryOp { lhs, rhs, .. } => { - walk(lhs, out); - walk(rhs, out); - } - // The scalar variants (issue #205) never appear as a direct - // `walk` target — every reachable one is peeled off first by - // `named` at whichever operator field holds it (`Scan.predicates`, - // `Filter.pred`, `Project.cols`, …). - QE::Column(_) - | QE::Literal(_) - | QE::Compare { .. } - | QE::BoolAnd(_) - | QE::BoolOr(_) - | QE::Not(_) - | QE::IsNull(_) - | QE::IsNotNull(_) - | QE::Cast { .. } - | QE::InList { .. } - | QE::FunctionCall { .. } - | QE::Arithmetic { .. } - | QE::Case { .. } => { - unreachable!("walk reached a scalar QueryExpr variant directly: {node:?}") - } - } - } - let mut out: Vec = Vec::new(); - walk(dag, &mut out); - out.sort(); - out.dedup(); - out -} - -#[cfg(test)] -mod tests { - use std::rc::Rc; - - use super::super::query_expr::{GroupKeys, Source}; - use super::*; - - fn src(name: &str) -> UnresolvedQueryExpr { - UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { - metric: name.into(), - }, - predicates: vec![], - schema: None, - } - } - - // Both correlation inputs must seed a usage-derived schema before positional resolution. - #[test] - fn pearson_corr_inputs_seed_usage_derived_schema() { - use crate::pre_asap::{AggIntent, Reduction}; - let dag = UnresolvedQueryExpr::Aggregate { - reduction: Reduction::by(vec![]), - measures: vec![AggIntent::PearsonCorr { - left: ColumnRef::Named("x".into()), - right: ColumnRef::Named("y".into()), - }], - output_names: vec![], - filters: vec![], - having: None, - child: Rc::new(src("m")), - }; - assert_eq!(collect_referenced_columns(&dag), vec!["x", "y"]); - let schema = SchemaResolver::new().resolve_schema(&dag); - assert!(schema.column_id("x").is_some()); - assert!(schema.column_id("y").is_some()); - } - - #[test] - fn bare_source_yields_ts_value_floor() { - let schema = SchemaResolver::new().resolve_schema(&src("m")); - assert_eq!(schema.fields.len(), 2); - assert_eq!(schema.fields[0].name, "ts"); - assert_eq!(schema.fields[1].name, "value"); - assert_eq!(schema.time_index, Some(0)); - } - - #[test] - fn sort_partition_keys_land_in_schema() { - // Per-group ranking keys (`topk by (host)` → `Sort.partition_by`) must be - // seeded into the usage-derived leaf so they resolve positionally. - let dag = UnresolvedQueryExpr::Sort { - keys: vec![super::super::query_expr::SortKey { - expr: UnresolvedQueryExpr::Column(ColumnRef::SampleValue), - ascending: false, - nulls_first: false, - }], - partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), - child: Rc::new(src("hits")), - }; - let schema = SchemaResolver::new().resolve_schema(&dag); - assert!(schema.column_id("host").is_some()); - } - - /// Issue #228 review: a `Concat`'s `discriminator_unique_key` columns — - /// even one referenced nowhere else in the DAG — must be seeded into - /// the usage-derived fallback schema, exactly like `Dedup.cols`, or - /// `resolve.rs`'s later `resolve_column_ref` fails `NotFound` for a - /// column the caller correctly named. - #[test] - fn concat_discriminator_key_is_seeded_into_the_resolver_schema() { - let dag = UnresolvedQueryExpr::concat_with_discriminator( - vec![src("m")], - ColumnRef::Named("phi".into()), - vec![ColumnRef::Named("host".into())], - ); - let schema = SchemaResolver::new().resolve_schema(&dag); - assert!( - schema.column_id("phi").is_some(), - "discriminator column must be seeded" - ); - assert!( - schema.column_id("host").is_some(), - "inner_key column must be seeded" - ); - } - - #[test] - fn inherited_names_are_seeded_alongside_referenced() { - // A `BinaryOp` side re-binds against its own sub-DAG, but must still see - // an enclosing aggregate's group key (`__name__` / `job`) that appears in - // neither side's own matchers (issue #52). `resolve_schema_with_inherited` seeds it. - let schema = - SchemaResolver::new().resolve_schema_with_inherited(&src("m"), &["__name__".into()]); - assert!(schema.column_id("__name__").is_some()); - // `resolve_schema` (no inheritance) does not conjure it. - let plain = SchemaResolver::new().resolve_schema(&src("m")); - assert!(plain.column_id("__name__").is_none()); - } - - #[test] - fn custom_catalog_supplies_base_columns() { - struct FixedCatalog; - impl SchemaCatalog for FixedCatalog { - fn columns_for(&self, source: &str) -> Option> { - (source == "known").then(|| { - vec![ - Field::plain("ts", DataType::Timestamp, false), - Field::plain("value", DataType::Float64, false), - Field::plain("datacenter", DataType::Utf8, false), - ] - }) - } - } - let schema = SchemaResolver::with_catalog(FixedCatalog).resolve_schema(&src("known")); - let dc = schema - .column_id("datacenter") - .and_then(|id| schema.fields.get(id)); - assert!(matches!(dc, Some(c) if !c.nullable)); - } -} diff --git a/crates/types/src/workload.rs b/crates/types/src/workload/mod.rs similarity index 99% rename from crates/types/src/workload.rs rename to crates/types/src/workload/mod.rs index 5e1a5d48c..0289d2e38 100644 --- a/crates/types/src/workload.rs +++ b/crates/types/src/workload/mod.rs @@ -1,3 +1,6 @@ +pub mod parsed_workload; +pub mod resources; + use crate::types::AccuracyTarget; use serde::{Deserialize, Serialize}; diff --git a/crates/types/src/parsed_workload.rs b/crates/types/src/workload/parsed_workload.rs similarity index 63% rename from crates/types/src/parsed_workload.rs rename to crates/types/src/workload/parsed_workload.rs index ff955e6a7..9dcf21b83 100644 --- a/crates/types/src/parsed_workload.rs +++ b/crates/types/src/workload/parsed_workload.rs @@ -1,5 +1,5 @@ //! [`ParsedWorkload`] — a [`PlanningWorkload`] whose queries have been lowered -//! to pre-ASAP IR. +//! to the operator IR. //! //! This is the boundary between the frontend stage and the optimization stage //! (issues #429, #430). Everything downstream of lowering consumes this type @@ -10,7 +10,7 @@ use std::rc::Rc; -use crate::pre_asap::query_expr::QueryExpr; +use crate::ir::{OperatorNode, QueryRoot, ScalarExpr}; use crate::workload::{ DataWorkload, PlanningWorkload, QueryWorkload, QueryWorkloadEntry, WorkloadError, }; @@ -33,7 +33,9 @@ pub enum ParsedWorkloadError { #[derive(Debug, Clone)] pub struct ParsedWorkload { workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, + operator_indices: Vec, + scalars: Vec<(usize, ScalarExpr)>, } impl ParsedWorkload { @@ -41,16 +43,43 @@ impl ParsedWorkload { /// `i`-th entry. pub fn new( workload: PlanningWorkload, - exprs: Vec>, + exprs: Vec>, + ) -> Result { + Self::from_roots( + workload, + exprs.into_iter().map(QueryRoot::Operator).collect(), + ) + } + + pub fn from_roots( + workload: PlanningWorkload, + roots: Vec, ) -> Result { let entries = workload.query_workload.entries().count(); - if entries != exprs.len() { + if entries != roots.len() { return Err(ParsedWorkloadError::LengthMismatch { entries, - lowered: exprs.len(), + lowered: roots.len(), }); } - Ok(Self { workload, exprs }) + let mut exprs = Vec::new(); + let mut operator_indices = Vec::new(); + let mut scalars = Vec::new(); + for (index, root) in roots.into_iter().enumerate() { + match root { + QueryRoot::Operator(node) => { + operator_indices.push(index); + exprs.push(node); + } + QueryRoot::Scalar(expr) => scalars.push((index, expr)), + } + } + Ok(Self { + workload, + exprs, + operator_indices, + scalars, + }) } pub fn planning_workload(&self) -> &PlanningWorkload { @@ -65,26 +94,37 @@ impl ParsedWorkload { self.workload.data_workload.as_ref() } - pub fn exprs(&self) -> &[Rc] { + pub fn exprs(&self) -> &[Rc] { &self.exprs } pub fn len(&self) -> usize { - self.exprs.len() + self.exprs.len() + self.scalars.len() } pub fn is_empty(&self) -> bool { - self.exprs.is_empty() + self.len() == 0 } /// Normalized entries paired with their lowered expression. - pub fn entries(&self) -> impl Iterator)> + '_ { + pub fn entries(&self) -> impl Iterator)> + '_ { self.workload .query_workload .entries() + .enumerate() + .filter(|(index, _)| self.operator_indices.binary_search(index).is_ok()) + .map(|(_, entry)| entry) .zip(self.exprs.iter()) } + pub fn operator_indices(&self) -> &[usize] { + &self.operator_indices + } + + pub fn scalar_roots(&self) -> &[(usize, ScalarExpr)] { + &self.scalars + } + /// The retained workload's own validation — entry legality and data-workload /// consistency. The PromQL-specific checks it also runs were already a /// precondition of the lowering that produced `self`. diff --git a/crates/types/src/resources.rs b/crates/types/src/workload/resources.rs similarity index 100% rename from crates/types/src/resources.rs rename to crates/types/src/workload/resources.rs diff --git a/crates/types/src/resources/cache.rs b/crates/types/src/workload/resources/cache.rs similarity index 100% rename from crates/types/src/resources/cache.rs rename to crates/types/src/workload/resources/cache.rs diff --git a/crates/types/src/resources/cpu.rs b/crates/types/src/workload/resources/cpu.rs similarity index 100% rename from crates/types/src/resources/cpu.rs rename to crates/types/src/workload/resources/cpu.rs diff --git a/crates/types/src/resources/measurement.rs b/crates/types/src/workload/resources/measurement.rs similarity index 100% rename from crates/types/src/resources/measurement.rs rename to crates/types/src/workload/resources/measurement.rs diff --git a/crates/types/src/resources/physical.rs b/crates/types/src/workload/resources/physical.rs similarity index 100% rename from crates/types/src/resources/physical.rs rename to crates/types/src/workload/resources/physical.rs diff --git a/crates/types/src/resources/physical_handoff.rs b/crates/types/src/workload/resources/physical_handoff.rs similarity index 100% rename from crates/types/src/resources/physical_handoff.rs rename to crates/types/src/workload/resources/physical_handoff.rs diff --git a/crates/types/src/resources/storage.rs b/crates/types/src/workload/resources/storage.rs similarity index 100% rename from crates/types/src/resources/storage.rs rename to crates/types/src/workload/resources/storage.rs diff --git a/crates/types/tests/logical_export.rs b/crates/types/tests/logical_export.rs new file mode 100644 index 000000000..f490733a3 --- /dev/null +++ b/crates/types/tests/logical_export.rs @@ -0,0 +1,241 @@ +//! Logical transport must accept plans with no execution timing assigned, before materialization. +use asap_types::ir::schema::Schema; +use asap_types::ir::{NonASAPOp, Operator, OperatorNode}; + +#[test] +fn logical_export_accepts_unassigned_timing() { + let root = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Values { + rows: vec![vec![]], + schema: Schema::lifted(vec![], None), + })) + .unwrap(); + assert!(root.timing.is_none()); + assert!(asap_types::ir::export::compile_logical_asap_dag(&root).is_ok()); +} + +use asap_types::ir::export::{ + compile_logical_asap_dag_with_node_ids, EdgeRole, LogicalASAPDAGDocument, + LogicalASAPDAGValidationError, LogicalASAPNodeId, LogicalASAPOperatorPayload, +}; +use asap_types::ir::operator::operator_properties::Reduction; +use asap_types::ir::operator::Source; +use asap_types::ir::scalar::{ColumnRef, ScalarValue}; +use asap_types::ir::schema::{ + DataType, Field, FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, + SummaryUpdate, +}; +use asap_types::ir::{ASAPOp, OperatorResultKind, ProjectItem, ScalarExpr}; +use std::rc::Rc; + +fn values() -> Rc { + OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Values { + rows: vec![vec![ScalarExpr::Literal(ScalarValue::Float64(1.0))]], + schema: Schema::lifted(vec![Field::plain("value", DataType::Float64, false)], None), + })) + .unwrap() +} + +/// Scalar subqueries contribute real edges; repeated references export one producer. +#[test] +fn scalar_dependencies_share_one_exported_producer() { + let child = values(); + let root = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Project { + cols: vec![ProjectItem { + alias: Some("result".into()), + expr: ScalarExpr::ScalarSubquery(child.clone()), + }], + qualifier: None, + child: child.clone(), + })) + .unwrap(); + let compiled = compile_logical_asap_dag_with_node_ids(&root).unwrap(); + compiled.dag.validate().unwrap(); + assert_eq!(compiled.dag.nodes.len(), 2); + assert_eq!(compiled.dag.edges.len(), 2); + assert!(compiled + .dag + .edges + .iter() + .any(|edge| edge.role == EdgeRole::ScalarRef)); + let id = compiled.node_ids.node_id(&child).unwrap(); + assert!(Rc::ptr_eq( + compiled.node_ids.operator_node(id).unwrap(), + &child + )); + let document = LogicalASAPDAGDocument::new(compiled.dag); + let json = serde_json::to_string(&document).unwrap(); + for physical_metadata in [ + "output_state", + "data_state", + "timing", + "retention", + "window", + ] { + assert!( + !json.contains(physical_metadata), + "logical JSON contains {physical_metadata}" + ); + } + let decoded: LogicalASAPDAGDocument = serde_json::from_str(&json).unwrap(); + assert_eq!(document, decoded); + decoded.validate().unwrap(); +} + +/// Summary state identity survives logical merge export without a phase assignment. +#[test] +fn merged_summary_preserves_typed_state() { + let state = OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryAgg { + child: values(), + family: FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), + GroupingStrategy::default(), + ), + input: SummaryUpdate::column(ColumnRef::Named("value".into())), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap(); + let root = OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryMerge { + children: (0..2) + .map(|start| { + let coverage = asap_types::ir::properties::summary_coverage::SummaryCoverage { + source: Source::Table { + table_ref: "values".into(), + }, + regions: vec![ + asap_types::ir::properties::summary_coverage::CoverageRegion { + time_ms: Some(start..start + 1), + population: Default::default(), + }, + ], + }; + std::rc::Rc::new((*state).clone().with_coverage(coverage).unwrap()) + }) + .collect(), + })) + .unwrap(); + let dag = asap_types::ir::export::compile_logical_asap_dag(&root).unwrap(); + dag.validate().unwrap(); + assert_eq!(dag.nodes.len(), 4); + let root_id = dag.roots[0].operator_refs()[0]; + let merged = &dag.nodes[root_id.0 as usize]; + assert_eq!(merged.result_kind, OperatorResultKind::State); + assert_eq!(merged.output_schema, root.schema); + assert_eq!(merged.coverage, root.coverage); + assert!(matches!( + merged.payload, + LogicalASAPOperatorPayload::SummaryMerge + )); + // Transport rejects a summary producer whose required coverage was dropped. + let mut stripped = dag.clone(); + let producer = stripped + .nodes + .iter_mut() + .find(|node| matches!(node.payload, LogicalASAPOperatorPayload::SummaryAgg { .. })) + .unwrap(); + producer.coverage = None; + let id = producer.id; + assert!(matches!( + stripped.validate(), + Err(LogicalASAPDAGValidationError::InvalidCoverage(bad)) if bad == id + )); +} + +/// Malformed wire graphs fail transport integrity checks rather than reaching execution. +#[test] +fn malformed_transport_is_rejected() { + let dag = asap_types::ir::export::compile_logical_asap_dag(&values()).unwrap(); + let mut document = LogicalASAPDAGDocument::new(dag.clone()); + document.schema_version = 99; + assert!(matches!( + document.validate(), + Err(LogicalASAPDAGValidationError::UnsupportedVersion(99)) + )); + let mut duplicate = dag.clone(); + duplicate.nodes.push(dag.nodes[0].clone()); + assert!(matches!( + duplicate.validate(), + Err(LogicalASAPDAGValidationError::DuplicateNode(_)) + )); + let mut missing = dag.clone(); + missing.roots = vec![asap_types::ir::export::LogicalASAPQueryRoot::Operator( + LogicalASAPNodeId(9), + )]; + assert!(matches!( + missing.validate(), + Err(LogicalASAPDAGValidationError::MissingNode(_)) + )); + let mut unreachable = dag.clone(); + let mut extra = dag.nodes[0].clone(); + extra.id = LogicalASAPNodeId(1); + unreachable.nodes.push(extra); + assert!(matches!( + unreachable.validate(), + Err(LogicalASAPDAGValidationError::UnreachableNode(_)) + )); + let mut json = serde_json::to_value(LogicalASAPDAGDocument::new(dag)).unwrap(); + json["dag"]["nodes"][0]["output_state"] = serde_json::json!({"timing":"ingestion_time"}); + assert!(serde_json::from_value::(json).is_err()); +} + +/// Standalone constants need no fake relation, while scalar subqueries retain their producer DAG. +#[test] +fn standalone_scalar_roots_roundtrip_without_synthetic_operators() { + use asap_types::ir::{export::compile_logical_asap_query, QueryRoot}; + for root in [ + QueryRoot::Scalar(ScalarExpr::literal_f64(42.0)), + QueryRoot::Scalar(ScalarExpr::ScalarSubquery(values())), + ] { + let expected = root.as_operator().is_some(); + assert!(!expected); + let dag = compile_logical_asap_query(&root).unwrap(); + dag.validate().unwrap(); + let expected_nodes = match root { + QueryRoot::Scalar(ScalarExpr::ScalarSubquery(_)) => 1, + _ => 0, + }; + assert_eq!(dag.nodes.len(), expected_nodes); + let document = LogicalASAPDAGDocument::new(dag); + let decoded: LogicalASAPDAGDocument = + serde_json::from_str(&serde_json::to_string(&document).unwrap()).unwrap(); + decoded.validate().unwrap(); + assert_eq!(document, decoded); + } +} + +/// A batch exports as one DAG: one root per query, shared producers exported once. +#[test] +fn batch_exports_one_root_per_query_and_shares_producers() { + use asap_types::ir::{export::compile_logical_asap_workload, QueryRoot}; + let shared = values(); + let project = |alias: &str| { + OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Project { + child: shared.clone(), + cols: vec![ProjectItem { + alias: Some(alias.into()), + expr: ScalarExpr::Column(0), + }], + qualifier: None, + })) + .unwrap() + }; + let dag = compile_logical_asap_workload(&[ + QueryRoot::Operator(project("a")), + QueryRoot::Operator(project("b")), + ]) + .unwrap(); + dag.validate().unwrap(); + assert_eq!(dag.roots.len(), 2); + assert_eq!( + dag.nodes.len(), + 3, + "the shared Values node is exported once" + ); + let mut empty = dag.clone(); + empty.roots.clear(); + assert!(matches!( + empty.validate(), + Err(LogicalASAPDAGValidationError::NoRoots) + )); +} diff --git a/crates/types/tests/physical_export.rs b/crates/types/tests/physical_export.rs new file mode 100644 index 000000000..a3397e2f0 --- /dev/null +++ b/crates/types/tests/physical_export.rs @@ -0,0 +1,127 @@ +//! A timed plan exports as a PhysicalASAPDAG that keeps timing and coverage. +use asap_types::ir::export::{ + compile_physical_asap_dag, PhysicalASAPDAGDocument, PhysicalASAPDAGValidationError, +}; +use asap_types::ir::operator::{Reduction, Source}; +use asap_types::ir::properties::summary_coverage::{CoverageRegion, SummaryCoverage}; +use asap_types::ir::properties::ExecutionTiming; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{DataType, Field, FieldDataType, Schema}; +use asap_types::ir::schema::{ExactKind, ExactParams, SummaryUpdate}; +use asap_types::ir::{ + apply_materialization_timings, ASAPOp, MaterializationAssignment, NonASAPOp, Operator, + OperatorNode, TimingMemo, +}; +use std::rc::Rc; + +/// Scan(t) → SummaryAgg(sum by key) → FinalizeExactAccumulator, untimed. +fn plan() -> Rc { + let source = Source::Table { + table_ref: "t".into(), + }; + let scan = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Scan { + source: source.clone(), + predicates: vec![], + schema: Schema::new(vec![ + Field::plain("key", DataType::Utf8, false), + Field::plain("value", DataType::Float64, false), + ]), + })) + .unwrap(); + let state = OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: scan, + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + input: SummaryUpdate::column(ColumnRef::Named("value".into())), + reduction: Reduction::by(vec![0]), + grouping: Default::default(), + filter: None, + })) + .unwrap() + .with_coverage(SummaryCoverage { + source, + regions: vec![CoverageRegion { + time_ms: Some(0..60_000), + population: Default::default(), + }], + }) + .unwrap(); + OperatorNode::new_shared(Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: Rc::new(state), + })) + .unwrap() +} + +/// A summary maintained at ingestion time is read at query time. +#[test] +fn timed_plan_exports_with_timing_and_coverage() { + let timed = apply_materialization_timings( + &plan(), + &MaterializationAssignment::all_ingestion_time(), + &mut TimingMemo::new(), + ) + .unwrap(); + let dag = compile_physical_asap_dag(&timed).unwrap(); + let document = PhysicalASAPDAGDocument::new(dag.clone()); + document.validate().unwrap(); + + let timings: Vec<_> = dag.nodes.iter().map(|n| n.output_state.timing).collect(); + assert_eq!( + timings, + vec![ + ExecutionTiming::IngestionTime, + ExecutionTiming::IngestionTime, + ExecutionTiming::QueryTime + ] + ); + assert!(dag.nodes[1].coverage.is_some()); + + let decoded: PhysicalASAPDAGDocument = + serde_json::from_str(&serde_json::to_string(&document).unwrap()).unwrap(); + assert_eq!(decoded, document); +} + +/// A query-time producer cannot feed an ingestion-time consumer. +#[test] +fn query_time_input_to_ingestion_is_rejected() { + let timed = apply_materialization_timings( + &plan(), + &MaterializationAssignment::all_ingestion_time(), + &mut TimingMemo::new(), + ) + .unwrap(); + let mut dag = compile_physical_asap_dag(&timed).unwrap(); + dag.nodes[0].output_state.timing = ExecutionTiming::QueryTime; + dag.edges[0].data_state = dag.nodes[0].output_state; + assert!(matches!( + dag.validate(), + Err(PhysicalASAPDAGValidationError::QueryDependencyInIngestion { .. }) + )); +} + +/// Untimed plans cannot be exported as physical plans. +#[test] +fn untimed_plan_is_rejected() { + assert!(compile_physical_asap_dag(&plan()).is_err()); +} + +/// Two queries reading one summary state export once, with one root per query. +#[test] +fn batch_shares_the_summary_and_keeps_one_root_per_query() { + use asap_types::ir::export::compile_physical_asap_workload; + let first = plan(); + let state = first.children()[0].clone(); + let second = OperatorNode::new_shared(Operator::ASAP(ASAPOp::FinalizeExactAccumulator { + child: state, + })) + .unwrap(); + let assignment = MaterializationAssignment::all_query_time(); + let mut memo = TimingMemo::new(); + let timed: Vec<_> = [first, second] + .iter() + .map(|root| apply_materialization_timings(root, &assignment, &mut memo).unwrap()) + .collect(); + let dag = compile_physical_asap_workload(&timed).unwrap(); + dag.validate().unwrap(); + assert_eq!(dag.roots.len(), 2); + assert_eq!(dag.nodes.len(), 4, "scan and summary are exported once"); +} diff --git a/crates/types/tests/planner_vocabulary.rs b/crates/types/tests/planner_vocabulary.rs index f14a56ee8..4d98e8c7e 100644 --- a/crates/types/tests/planner_vocabulary.rs +++ b/crates/types/tests/planner_vocabulary.rs @@ -1,8 +1,6 @@ -use asap_types::post_asap::{ - validate_pane_coverage, PaneLayout, WindowEdgeCompatibility, WindowEdgeCoverage, -}; -use asap_types::pre_asap::{SchemaResolver, Source, UnresolvedQueryExpr}; -use asap_types::resources::{PhysicalHandoffBytes, PhysicalHandoffKind}; +use asap_types::ir::export::WindowEdgeCompatibility; +use asap_types::physical::{validate_pane_coverage, PaneLayout, WindowEdgeCoverage}; +use asap_types::workload::resources::{PhysicalHandoffBytes, PhysicalHandoffKind}; // Renamed pane APIs still read and emit the deployed wire contract. #[test] @@ -30,18 +28,11 @@ fn window_edge_names_preserve_wire_values() { ); } -// External consumers can use the new resolver and resource names without changing behavior. +// External consumers can use the new resource names without changing behavior. +// (The schema-resolver half moved with the resolver to `asap-frontend-common`; +// `schema_resolver::tests::bare_source_yields_ts_value_floor` covers it.) #[test] -fn renamed_schema_and_handoff_apis_are_public() { - let dag = UnresolvedQueryExpr::Scan { - source: Source::TimeSeries { - metric: "requests".into(), - }, - predicates: vec![], - schema: None, - }; - let schema = SchemaResolver::new().resolve_schema(&dag); - assert!(schema.column_id("value").is_some()); +fn renamed_handoff_apis_are_public() { let bytes = PhysicalHandoffBytes { network_bytes: 12, materialization_bytes: 4, diff --git a/crates/types/tests/schema_rebuilding.rs b/crates/types/tests/schema_rebuilding.rs index 1f1b57dbc..69c51a36d 100644 --- a/crates/types/tests/schema_rebuilding.rs +++ b/crates/types/tests/schema_rebuilding.rs @@ -1,12 +1,33 @@ +use asap_types::ir::operator::{AggIntent, Reduction, Source}; +use asap_types::ir::properties::summary_coverage::{CoverageRegion, SummaryCoverage}; +use asap_types::ir::properties::{ExecutionTiming, ResultGuarantee}; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{DataType, Field, FieldDataType, Schema}; +use asap_types::ir::schema::{ExactKind, ExactParams, GroupingStrategy, SummaryUpdate}; use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; -use asap_types::post_asap::{ - ExactKind, ExactParams, ExecutionTiming, GroupingStrategy, ResultGuarantee, SummaryUpdate, -}; -use asap_types::pre_asap::{ - AggIntent, ColumnRef, DataType, Field, FieldDataType, Reduction, Schema, Source, -}; use std::rc::Rc; +fn coverage() -> SummaryCoverage { + SummaryCoverage { + source: Source::Table { + table_ref: "t".into(), + }, + regions: vec![CoverageRegion { + time_ms: None, + population: Default::default(), + }], + } +} +/// Rewrites clear coverage; a rewriter must declare it again for summary nodes. +fn redeclare(node: OperatorNode) -> Rc { + let node = Rc::new(node); + if !node.requires_coverage() { + return node; + } + assert!(node.validate_structure().is_err()); + Rc::new((*node).clone().with_coverage(coverage()).unwrap()) +} + fn scan(key_type: DataType, name: &str) -> Rc { OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Scan { source: Source::Table { @@ -40,7 +61,12 @@ fn aggregate(child: Rc, asap: bool) -> Rc { having: None, }) }; - OperatorNode::new_shared(operator).unwrap() + let node = OperatorNode::new(operator).unwrap(); + Rc::new(if asap { + node.with_coverage(coverage()).unwrap() + } else { + node + }) } /// Rewrites follow changed input types and inherited names for either category. @@ -50,7 +76,7 @@ fn rebuilding_rederives_schema_for_both_categories() { let original = aggregate(scan(DataType::Int64, "key"), asap); original.validate_structure().unwrap(); let replacement = scan(DataType::Utf8, "new_key"); - let rebuilt = Rc::new(original.map_children(|_| replacement.clone()).unwrap()); + let rebuilt = redeclare(original.map_children(|_| replacement.clone()).unwrap()); assert_eq!(rebuilt.schema, rebuilt.operator.output_schema().unwrap()); rebuilt.validate_structure().unwrap(); } @@ -64,14 +90,14 @@ fn rebuilding_preserves_only_explicit_naming_overrides() { let mut schema = original.schema.clone(); schema.fields[0].name = "alias".into(); schema.fields[0].table = Some("result".into()); - let original = Rc::new( - OperatorNode::with_schema(original.operator.clone(), schema) - .with_guarantee(Some(ResultGuarantee::exact("fixture"))) - .with_timing(Some(ExecutionTiming::QueryTime)), - ); + let mut renamed = OperatorNode::with_schema(original.operator.clone(), schema) + .with_guarantee(Some(ResultGuarantee::exact("fixture"))) + .with_timing(Some(ExecutionTiming::QueryTime)); + renamed.coverage = original.coverage.clone(); + let original = Rc::new(renamed); original.validate_structure().unwrap(); let replacement = scan(DataType::Utf8, "new_key"); - let rebuilt = Rc::new(original.map_children(|_| replacement.clone()).unwrap()); + let rebuilt = redeclare(original.map_children(|_| replacement.clone()).unwrap()); assert_eq!(rebuilt.schema.fields[0].name, "alias"); assert_eq!(rebuilt.schema.fields[0].table.as_deref(), Some("result")); assert_eq!( @@ -110,7 +136,9 @@ fn validation_rejects_structural_overrides_for_both_categories() { schema.fields.pop(); invalid.push(schema); for schema in invalid { - let forged = Rc::new(OperatorNode::with_schema(original.operator.clone(), schema)); + let mut forged = OperatorNode::with_schema(original.operator.clone(), schema); + forged.coverage = original.coverage.clone(); + let forged = Rc::new(forged); assert!( forged.validate_structure().is_err(), "accepted structural override: {:?}", @@ -123,7 +151,7 @@ fn validation_rejects_structural_overrides_for_both_categories() { /// Passthrough rewrites derive metadata and arity, but cannot guess alias positions. #[test] fn rebuilding_updates_metadata_and_requires_new_aliases_after_arity_changes() { - use asap_types::pre_asap::GroupKeys; + use asap_types::ir::operator::GroupKeys; let input = scan(DataType::Timestamp, "key"); let original = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Limit { n: Some(10), @@ -160,8 +188,8 @@ fn rebuilding_updates_metadata_and_requires_new_aliases_after_arity_changes() { /// Maintaining membership and finalizing values preserve identity/time metadata. #[test] fn summary_transitions_preserve_structural_metadata() { - use asap_types::post_asap::maintained_population::{MaintainedPopulation, PopulationInput}; - use asap_types::pre_asap::GroupKeys; + use asap_types::ir::operator::maintained_population::{MaintainedPopulation, PopulationInput}; + use asap_types::ir::operator::GroupKeys; let mut schema = scan(DataType::Timestamp, "key").schema.clone(); schema.closed = true; schema.time_index = Some(0); diff --git a/crates/types/tests/structure_contract.rs b/crates/types/tests/structure_contract.rs index d4e164f71..68f614106 100644 --- a/crates/types/tests/structure_contract.rs +++ b/crates/types/tests/structure_contract.rs @@ -1,7 +1,9 @@ +use asap_types::ir::operator::Source; +use asap_types::ir::scalar::ScalarValue; +use asap_types::ir::schema::{DataType, Field, Schema}; use asap_types::ir::{ ExprSemantics, NonASAPOp, OperatorNode, OperatorResultKind, Predicate, ScalarExpr, }; -use asap_types::pre_asap::{DataType, Field, ScalarValue, Schema, Source}; use std::rc::Rc; fn scan() -> Rc { OperatorNode::new_shared(asap_types::ir::Operator::NonASAP(NonASAPOp::Scan { @@ -13,6 +15,19 @@ fn scan() -> Rc { })) .unwrap() } +/// Tabular coverage for a whole-table summary. +fn whole_table() -> asap_types::ir::properties::summary_coverage::SummaryCoverage { + use asap_types::ir::properties::summary_coverage::{CoverageRegion, SummaryCoverage}; + SummaryCoverage { + source: Source::Table { + table_ref: "t".into(), + }, + regions: vec![CoverageRegion { + time_ms: None, + population: Default::default(), + }], + } +} /// Resolved filters cannot hide invalid scalar types or out-of-scope columns. #[test] fn invalid_predicates_are_rejected() { @@ -52,7 +67,7 @@ fn values_contract_is_checked() { } /// Scalar typing validates every branch and never assigns placeholder types. #[test] -fn scalar_type_ruless_fail_closed() { +fn scalar_type_rules_fail_closed() { for expr in [ ScalarExpr::Column(99), ScalarExpr::FunctionCall { @@ -84,12 +99,13 @@ fn scalar_type_ruless_fail_closed() { /// A state family is not interchangeable with another sketch or a scalar field. #[test] fn state_evaluations_and_passthrough_keep_their_contracts() { - use asap_types::ir::{ASAPOp, Operator, ProjectItem}; - use asap_types::post_asap::{ + use asap_types::ir::operator::Reduction; + use asap_types::ir::scalar::ColumnRef; + use asap_types::ir::schema::{ FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SketchStatistic, SummaryUpdate, }; - use asap_types::pre_asap::{ColumnRef, Reduction}; + use asap_types::ir::{ASAPOp, Operator, ProjectItem}; let family = FieldDataType::Sketch( SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 100 }), GroupingStrategy::default(), @@ -103,6 +119,8 @@ fn state_evaluations_and_passthrough_keep_their_contracts() { grouping: GroupingStrategy::default(), filter: None, })) + .unwrap() + .with_coverage(whole_table()) .unwrap(), ); state.validate_structure().unwrap(); @@ -138,8 +156,8 @@ fn state_evaluations_and_passthrough_keep_their_contracts() { /// Phase validation checks dependencies, without declaring a computation query-only. #[test] fn execution_timing_checks_edges_not_function_names() { + use asap_types::ir::properties::ExecutionTiming::{IngestionTime, QueryTime}; use asap_types::ir::ProjectItem; - use asap_types::post_asap::ExecutionTiming::{IngestionTime, QueryTime}; let input = Rc::new((*scan()).clone().with_timing(Some(IngestionTime))); let mut project = OperatorNode::new(asap_types::ir::Operator::NonASAP(NonASAPOp::Project { child: input, @@ -168,20 +186,26 @@ fn execution_timing_checks_edges_not_function_names() { /// Both operator categories use the same fallible schema-deriving constructor. #[test] fn shared_construction_derives_both_operator_categories() { + use asap_types::ir::operator::Reduction; + use asap_types::ir::scalar::ColumnRef; + use asap_types::ir::schema::FieldDataType; + use asap_types::ir::schema::{ExactKind, ExactParams, GroupingStrategy, SummaryUpdate}; use asap_types::ir::{ASAPOp, Operator, ProjectItem}; - use asap_types::post_asap::{ExactKind, ExactParams, GroupingStrategy, SummaryUpdate}; - use asap_types::pre_asap::{ColumnRef, FieldDataType, Reduction}; let input = scan(); - let state = OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryAgg { - child: input.clone(), - family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), - input: SummaryUpdate::column(ColumnRef::Named("x".into())), - reduction: Reduction::by(vec![]), - grouping: GroupingStrategy::default(), - filter: None, - })) - .unwrap(); + let state = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: input.clone(), + family: FieldDataType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + input: SummaryUpdate::column(ColumnRef::Named("x".into())), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap() + .with_coverage(whole_table()) + .unwrap(), + ); assert_eq!(state.result_kind, OperatorResultKind::State); assert!(!state.schema.fields.last().unwrap().is_plain()); assert!(state.guarantee.is_none()); @@ -205,3 +229,76 @@ fn shared_construction_derives_both_operator_categories() { })) .is_err()); } + +/// A top-k readout derives the selected rows (partition keys, item identity, +/// Float64 `value`), the shape every executable top-k returns. +#[test] +fn topk_readout_derives_selected_rows() { + use asap_types::ir::operator::Reduction; + use asap_types::ir::scalar::ColumnRef; + use asap_types::ir::schema::{ + FieldDataType, GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, + SketchStatistic, SummaryInputExpr, SummaryUpdate, WeightDomain, + }; + use asap_types::ir::{ASAPOp, Operator}; + let rows = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "t".into(), + }, + predicates: vec![], + schema: Schema::new(vec![ + Field::plain("job", DataType::Utf8, false), + Field::plain("series", DataType::Utf8, false), + Field::plain("x", DataType::Float64, false), + ]), + })) + .unwrap(); + let state = Rc::new( + OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: rows, + family: FieldDataType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 64, + depth: 3, + heap_size: 10, + }, + ), + GroupingStrategy::default(), + ), + input: SummaryUpdate { + item: Some(SummaryInputExpr::Column(ColumnRef::Named("series".into()))), + weight: SummaryInputExpr::Column(ColumnRef::Named("x".into())), + weight_domain: WeightDomain::UnknownOrSigned, + }, + reduction: Reduction::by(vec![0]), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap() + .with_coverage(whole_table()) + .unwrap(), + ); + let topk = OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryEstimate { + summary_input: state, + query: SketchStatistic::TopK { k: 5 }, + })) + .unwrap(); + topk.validate_structure().unwrap(); + let fields: Vec<_> = topk + .schema + .fields + .iter() + .map(|f| (f.name.as_str(), f.plain_dtype().cloned())) + .collect(); + assert_eq!( + fields, + [ + ("job", Some(DataType::Utf8)), + ("series", Some(DataType::Utf8)), + ("value", Some(DataType::Float64)), + ] + ); + assert_eq!(topk.schema.unique_keys, vec![vec![0, 1]]); +} diff --git a/crates/types/tests/summary_coverage.rs b/crates/types/tests/summary_coverage.rs new file mode 100644 index 000000000..b1a8149dd --- /dev/null +++ b/crates/types/tests/summary_coverage.rs @@ -0,0 +1,148 @@ +//! Coverage composition preserves gaps and rejects duplicate observations. +use asap_types::ir::operator::operator_properties::Reduction; +use asap_types::ir::operator::Source; +use asap_types::ir::properties::summary_coverage::*; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::SummaryUpdate; +fn table(name: &str) -> Source { + Source::Table { + table_ref: name.into(), + } +} +fn coverage(start: i64, end: i64, population: &[(&str, &str)]) -> SummaryCoverage { + SummaryCoverage { + source: table("flows"), + regions: vec![CoverageRegion { + time_ms: Some(start..end), + population: population + .iter() + .map(|(k, v)| (k.to_string(), v.to_string())) + .collect(), + }], + } +} +/// Adjacent panes coalesce; gaps remain disconnected rather than becoming a hull. +#[test] +fn time_union_preserves_gaps() { + let merged = + SummaryCoverage::merge_disjoint(&[coverage(0, 1, &[]), coverage(1, 2, &[])]).unwrap(); + assert_eq!(merged.regions[0].time_ms, Some(0..2)); + assert_eq!(merged.regions.len(), 1); + let gapped = + SummaryCoverage::merge_disjoint(&[coverage(0, 1, &[]), coverage(2, 3, &[])]).unwrap(); + assert_eq!(gapped.regions.len(), 2); +} +/// Population partitions can overlap in time without sharing observations. +#[test] +fn population_and_joint_union() { + let merged = SummaryCoverage::merge_disjoint(&[ + coverage(0, 2, &[("region", "us")]), + coverage(0, 2, &[("region", "eu")]), + ]) + .unwrap(); + assert_eq!(merged.regions.len(), 2); + let joint = SummaryCoverage::merge_disjoint(&[ + coverage(0, 1, &[("region", "us")]), + coverage(1, 2, &[("region", "eu")]), + ]) + .unwrap(); + assert_eq!(joint.regions.len(), 2); + let decoded: SummaryCoverage = + serde_json::from_str(&serde_json::to_string(&joint).unwrap()).unwrap(); + assert_eq!(decoded, joint); +} +/// Intersecting predicates and windows cannot authorize once-per-observation merge. +#[test] +fn overlap_and_identity_fail_closed() { + assert_eq!( + SummaryCoverage::merge_disjoint(&[coverage(0, 2, &[]), coverage(1, 3, &[])]), + Err(CoverageError::PossibleOverlap) + ); + assert_eq!( + SummaryCoverage::merge_disjoint(&[ + coverage(0, 2, &[("region", "us")]), + coverage(0, 2, &[("tier", "premium")]) + ]), + Err(CoverageError::PossibleOverlap) + ); + assert_eq!( + SummaryCoverage::merge_disjoint(&[ + coverage(0, 2, &[("region", "us")]), + coverage(0, 2, &[("region", "us")]) + ]), + Err(CoverageError::PossibleOverlap) + ); + let mut other = coverage(1, 2, &[]); + other.source = table("other-flows"); + assert_eq!( + SummaryCoverage::merge_disjoint(&[coverage(0, 1, &[]), other]), + Err(CoverageError::SourceMismatch) + ); + assert_eq!( + coverage(2, 1, &[]).validate(), + Err(CoverageError::InvalidInterval) + ); +} + +/// Coverage is logical state metadata, and input rewrites invalidate its proof. +#[test] +fn node_coverage_is_required_checked_and_cleared_by_rewrites() { + use asap_types::ir::schema::{ + DataType, Field, FieldDataType, Schema, SketchAlgorithm, SketchKind, SketchParams, + }; + use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; + let raw = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Scan { + source: table("flows"), + predicates: vec![], + schema: Schema::new(vec![Field::plain("latency", DataType::Float64, false)]), + })) + .unwrap(); + let declared = coverage(0, 1, &[]); + assert!((*raw).clone().with_coverage(declared.clone()).is_err()); + let state = OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: raw, + family: FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), + Default::default(), + ), + input: SummaryUpdate::column(ColumnRef::Named("latency".into())), + reduction: Reduction::by(vec![]), + grouping: Default::default(), + filter: None, + })) + .unwrap(); + // Summary nodes cannot validate without coverage. + assert!(matches!( + std::rc::Rc::new(state.clone()).validate_structure(), + Err(asap_types::ir::SchemaDerivationError::Coverage( + CoverageError::Missing + )) + )); + let state = state.with_coverage(declared.clone()).unwrap(); + std::rc::Rc::new(state.clone()) + .validate_structure() + .unwrap(); + let rebuilt = state.map_children(Clone::clone).unwrap(); + assert!(rebuilt.coverage.is_none()); +} + +/// Sources without a time column declare no time bounds; such a region overlaps +/// any region it is not population-disjoint from. +#[test] +fn regions_without_time_bounds() { + let mut tabular = coverage(0, 1, &[("region", "us")]); + tabular.regions[0].time_ms = None; + let mut other = coverage(0, 1, &[("region", "eu")]); + other.regions[0].time_ms = None; + assert_eq!( + SummaryCoverage::merge_disjoint(&[tabular.clone(), other]) + .unwrap() + .regions + .len(), + 2 + ); + assert_eq!( + SummaryCoverage::merge_disjoint(&[tabular, coverage(5, 6, &[("region", "us")])]), + Err(CoverageError::PossibleOverlap) + ); +} diff --git a/crates/types/tests/summary_coverage_examples.rs b/crates/types/tests/summary_coverage_examples.rs new file mode 100644 index 000000000..5e2a7c05a --- /dev/null +++ b/crates/types/tests/summary_coverage_examples.rs @@ -0,0 +1,265 @@ +//! The examples in docs/design_docs/proposals/asap-primitive-schema.md, built as real +//! SummaryAgg -> SummaryMerge plans. Every input has the same schema +//! `(job: Utf8, state: KLL{k=200})`; only coverage differs. +use asap_types::ir::operator::{Reduction, Source}; +use asap_types::ir::properties::summary_coverage::{ + CoverageError, CoverageRegion, SummaryCoverage, +}; +use asap_types::ir::scalar::{ColumnRef, CompareOpKind, ScalarValue}; +use asap_types::ir::schema::{DataType, Field, FieldDataType, Schema}; +use asap_types::ir::schema::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate, +}; +use asap_types::ir::{ + ASAPOp, ExprSemantics, NonASAPOp, Operator, OperatorNode, Predicate, ScalarExpr, + SchemaDerivationError, +}; +use std::rc::Rc; + +const MIN: i64 = 60_000; + +fn requests() -> Source { + Source::Table { + table_ref: "requests".into(), + } +} + +fn scan() -> Rc { + OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Scan { + source: requests(), + predicates: vec![], + schema: Schema::new(vec![ + Field::plain("job", DataType::Utf8, false), + Field::plain("region", DataType::Utf8, false), + Field::plain("tier", DataType::Utf8, false), + Field::plain("latency", DataType::Float64, false), + Field::plain("size", DataType::Float64, false), + ]), + })) + .unwrap() +} + +/// `region = value`, evaluated on the scan schema. +fn region_is(value: &str) -> Predicate { + Predicate(ScalarExpr::Compare { + left: Box::new(ScalarExpr::Column(1)), + op: CompareOpKind::Eq, + right: Box::new(ScalarExpr::Literal(ScalarValue::Utf8(value.into()))), + semantics: ExprSemantics::Sql, + }) +} + +fn region(time_ms: Option>, population: &[(&str, &str)]) -> CoverageRegion { + CoverageRegion { + time_ms, + population: population + .iter() + .map(|(k, v)| (k.to_string(), v.to_string())) + .collect(), + } +} + +/// p99-ready KLL over `column`, grouped by job, with declared coverage. +fn kll_over( + column: &str, + filter: Option, + coverage: SummaryCoverage, +) -> Rc { + let node = OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: scan(), + family: FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 200 }), + GroupingStrategy::default(), + ), + input: SummaryUpdate::column(ColumnRef::Named(column.into())), + reduction: Reduction::by(vec![0]), + grouping: GroupingStrategy::default(), + filter, + })) + .unwrap(); + Rc::new(node.with_coverage(coverage).unwrap()) +} + +fn kll(time_ms: Option>, population: &[(&str, &str)]) -> Rc { + kll_over( + "latency", + None, + SummaryCoverage { + source: requests(), + regions: vec![region(time_ms, population)], + }, + ) +} + +fn merge(children: Vec>) -> Result, SchemaDerivationError> { + let schema = children[0].schema.clone(); + let merged = OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryMerge { children }))?; + merged.validate_structure()?; + // Schema never changes; only coverage does. + assert_eq!(merged.schema, schema); + Ok(merged) +} + +fn regions(node: &OperatorNode) -> Vec { + node.coverage.as_ref().unwrap().regions.clone() +} + +fn rejected(result: Result, SchemaDerivationError>, expected: CoverageError) { + match result { + Err(SchemaDerivationError::Coverage(actual)) => assert_eq!(actual, expected), + other => panic!("expected {expected:?}, got {other:?}"), + } +} + +/// Example 1: adjacent panes coalesce, gaps stay, overlapping windows are rejected. +#[test] +fn example_1_time() { + let adjacent = merge(vec![kll(Some(0..MIN), &[]), kll(Some(MIN..2 * MIN), &[])]).unwrap(); + assert_eq!(regions(&adjacent), vec![region(Some(0..2 * MIN), &[])]); + + let gapped = merge(vec![ + kll(Some(0..MIN), &[]), + kll(Some(2 * MIN..3 * MIN), &[]), + ]) + .unwrap(); + assert_eq!( + regions(&gapped), + vec![ + region(Some(0..MIN), &[]), + region(Some(2 * MIN..3 * MIN), &[]) + ] + ); + + rejected( + merge(vec![ + kll(Some(0..2 * MIN), &[]), + kll(Some(MIN..3 * MIN), &[]), + ]), + CoverageError::PossibleOverlap, + ); +} + +/// Example 2: disjoint label values merge; different labels or equal values are rejected. +#[test] +fn example_2_population() { + let t = Some(0..MIN); + let us_eu = merge(vec![ + kll(t.clone(), &[("region", "us")]), + kll(t.clone(), &[("region", "eu")]), + ]) + .unwrap(); + assert_eq!( + regions(&us_eu), + vec![ + region(t.clone(), &[("region", "eu")]), + region(t.clone(), &[("region", "us")]), + ] + ); + for other in [("tier", "premium"), ("region", "us")] { + rejected( + merge(vec![ + kll(t.clone(), &[("region", "us")]), + kll(t.clone(), &[other]), + ]), + CoverageError::PossibleOverlap, + ); + } +} + +/// Example 3: time and population stay paired; never widened to {us,eu} × [0,2). +#[test] +fn example_3_joint_regions() { + let joint = merge(vec![ + kll(Some(0..MIN), &[("region", "us")]), + kll(Some(MIN..2 * MIN), &[("region", "eu")]), + ]) + .unwrap(); + assert_eq!( + regions(&joint), + vec![ + region(Some(MIN..2 * MIN), &[("region", "eu")]), + region(Some(0..MIN), &[("region", "us")]), + ] + ); +} + +/// A table without a time column declares no time bounds. +#[test] +fn tabular_source_without_time_bounds() { + let by_region = merge(vec![ + kll(None, &[("region", "us")]), + kll(None, &[("region", "eu")]), + ]) + .unwrap(); + assert_eq!(regions(&by_region).len(), 2); + rejected( + merge(vec![ + kll(None, &[("region", "us")]), + kll(Some(0..MIN), &[("region", "us")]), + ]), + CoverageError::PossibleOverlap, + ); +} + +/// Inputs must read the same source and share the producer's update and reduction. +#[test] +fn incompatible_inputs() { + let other_source = kll_over( + "latency", + None, + SummaryCoverage { + source: Source::Table { + table_ref: "other".into(), + }, + regions: vec![region(Some(MIN..2 * MIN), &[])], + }, + ); + rejected( + merge(vec![kll(Some(0..MIN), &[]), other_source]), + CoverageError::SourceMismatch, + ); + + // Same schema (both Float64 columns), different update expression. + let size = kll_over( + "size", + None, + SummaryCoverage { + source: requests(), + regions: vec![region(Some(MIN..2 * MIN), &[])], + }, + ); + assert!(matches!( + merge(vec![kll(Some(0..MIN), &[]), size]), + Err(SchemaDerivationError::InvalidScalarSignature(message)) + if message.contains("update expression and reduction") + )); +} + +/// Summary nodes must carry coverage, and merges reject inputs without it. +#[test] +fn coverage_is_required() { + let mut missing = (*kll(Some(MIN..2 * MIN), &[])).clone(); + missing.coverage = None; + let missing = Rc::new(missing); + assert!(matches!( + missing.validate_structure(), + Err(SchemaDerivationError::Coverage(CoverageError::Missing)) + )); + rejected( + merge(vec![kll(Some(0..MIN), &[]), missing]), + CoverageError::UnknownInput, + ); +} + +/// Trusted declarations: population is not checked against the filter (#570). +/// Both states hold US data, yet the wrong declaration lets them merge. +#[test] +fn wrong_population_declaration_is_accepted_until_570() { + let declared = |value: &str| SummaryCoverage { + source: requests(), + regions: vec![region(Some(0..MIN), &[("region", value)])], + }; + let a = kll_over("latency", Some(region_is("us")), declared("eu")); + let b = kll_over("latency", Some(region_is("us")), declared("us")); + assert!(merge(vec![a, b]).is_ok()); +} diff --git a/crates/types/tests/summary_merge_structure.rs b/crates/types/tests/summary_merge_structure.rs new file mode 100644 index 000000000..27bfcf908 --- /dev/null +++ b/crates/types/tests/summary_merge_structure.rs @@ -0,0 +1,109 @@ +//! Window composition merges compatible summary states without consuming raw rows. +use asap_types::ir::operator::operator_properties::{Reduction, Source}; +use asap_types::ir::scalar::ColumnRef; +use asap_types::ir::schema::{ + DataType, Field, FieldDataType, GroupingStrategy, Schema, SketchAlgorithm, SketchKind, + SketchParams, SummaryUpdate, +}; +use asap_types::ir::{ASAPOp, NonASAPOp, Operator, OperatorNode}; +use std::rc::Rc; +fn state(k: u32) -> Rc { + let scan = OperatorNode::new_shared(Operator::NonASAP(NonASAPOp::Scan { + source: Source::Table { + table_ref: "latencies".into(), + }, + predicates: vec![], + schema: Schema::new(vec![Field::plain("value", DataType::Float64, false)]), + })) + .unwrap(); + let summary = OperatorNode::new(Operator::ASAP(ASAPOp::SummaryAgg { + child: scan, + family: FieldDataType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + Default::default(), + ), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + filter: None, + })) + .unwrap(); + std::rc::Rc::new( + summary + .with_coverage( + asap_types::ir::properties::summary_coverage::SummaryCoverage { + source: Source::Table { + table_ref: "latencies".into(), + }, + regions: vec![ + asap_types::ir::properties::summary_coverage::CoverageRegion { + time_ms: Some(0..1), + population: Default::default(), + }, + ], + }, + ) + .unwrap(), + ) +} +/// Two KLL panes compose into one typed logical state without timing assignment. +#[test] +fn compatible_panes_merge_structurally() { + let root = OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryMerge { + children: vec![state(200), shifted_state(200, 1, 2)], + })) + .unwrap(); + root.validate_structure().unwrap(); + assert_eq!(root.schema.fields.len(), 1); + assert_eq!( + root.coverage.as_ref().unwrap().regions[0].time_ms, + Some(0..2) + ); +} +/// An empty merge, raw rows and differently sized state cannot masquerade as compatible panes. +#[test] +fn incompatible_merge_inputs_fail() { + for children in [ + vec![], + vec![state(200), state(300)], + vec![state(200).children()[0].clone()], + ] { + assert!( + OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryMerge { children })).is_err() + ); + } +} + +fn shifted_state(k: u32, start: i64, end: i64) -> Rc { + let mut node = (*state(k)).clone(); + let region = &mut node.coverage.as_mut().unwrap().regions[0]; + region.time_ms = Some(start..end); + Rc::new(node) +} +/// Schema equality cannot authorize overlapping or unknown observation coverage. +#[test] +fn unsafe_coverage_merge_is_rejected() { + let mut unknown = (*state(200)).clone(); + unknown.coverage = None; + for children in [ + vec![state(200), state(200)], + vec![state(200), Rc::new(unknown)], + ] { + assert!( + OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryMerge { children })).is_err() + ); + } +} + +/// Gapped time coverage remains disconnected, and forged output metadata is rejected. +#[test] +fn merge_derives_coverage_and_validates_retained_metadata() { + let root = OperatorNode::new_shared(Operator::ASAP(ASAPOp::SummaryMerge { + children: vec![state(200), shifted_state(200, 2, 3)], + })) + .unwrap(); + assert_eq!(root.coverage.as_ref().unwrap().regions.len(), 2); + let mut forged = (*root).clone(); + forged.coverage.as_mut().unwrap().regions[0].time_ms = Some(0..2); + assert!(Rc::new(forged).validate_structure().is_err()); +} diff --git a/docs/README.md b/docs/README.md index 9f802e094..c9abe1177 100644 --- a/docs/README.md +++ b/docs/README.md @@ -19,8 +19,7 @@ Start with [ASAPPlanner input, output, and workflows](design_docs/architecture/i for the integration boundary, nested inputs, and choice of planning workflow. Use [Public library functions and examples](develop_docs/library-api.md) for -frontend lowering, workload search, ranking, optional selection and lifecycle -integration. +frontend lowering, workload search, ranking, selection and DAG assembly. ## Extend the planner diff --git a/docs/design_docs/architecture/README.md b/docs/design_docs/architecture/README.md index 8b936eafe..7f60d0932 100644 --- a/docs/design_docs/architecture/README.md +++ b/docs/design_docs/architecture/README.md @@ -8,7 +8,7 @@ deployment-level decision, and run the selected contract. For the integration workflow, start with [ASAPPlanner input, output, and workflows](input-output-workflow.md). It defines inputs, `CandidateLogicalASAPDAGs`, selection -and summary-maintenance lifecycle workflows, and future replanning support. +and assembly workflows, and future replanning support. ## Planner component flow @@ -17,16 +17,12 @@ flowchart TD W["PlanningWorkload: query demand + optional data facts"] F["Frontend dependencies: SQL catalog or PromQL time"] E["Strategy, accuracy model, and applicable evidence"] - PRE["Frontend lowering → canonical Pre-ASAP QueryExpr roots"] + PRE["Frontend lowering → canonical Pre-ASAP OperatorNode roots"] SEARCH["Whole-workload candidate search: sharing, legality, accuracy"] SPACE["CandidateLogicalASAPDAGs: compact logical candidate DAG space"] RANK["Optional cost_sorted: ranked inspection view"] SELECT["Optional global_selection + assemble_selected_dag"] DAG["Selected logical Post-ASAP DAG"] - LINPUT["Optional lifecycle inputs: horizon, rates, capabilities, costs"] - LIFE["global_selection_with_summary_maintenance_lifecycles"] - LMAT["assemble_selected_dag_with_summary_maintenance_lifecycles"] - LPLAN["SummaryMaintenanceLifecyclePlan: DAG root + lifecycle decisions"] BACKEND["Downstream: bind physical alternatives, decide deployment, compile and execute"] W --> PRE F --> PRE @@ -35,19 +31,15 @@ flowchart TD SEARCH --> SPACE SPACE --> RANK --> BACKEND SPACE --> SELECT --> DAG --> BACKEND - SPACE --> LIFE - LINPUT --> LIFE --> LMAT --> LPLAN --> BACKEND ``` `CandidateLogicalASAPDAGs` is the output of logical candidate search. Each target's candidate set holds -alternatives and rejection reasons, but no selected maintenance lifecycle. -Choose among the three branches: inspect candidates (optionally ranked), select -and assemble logical DAGs, or select and assemble with summary-maintenance -lifecycle decisions. Use the last branch when Planner owns the maintenance -decision; otherwise the backend owns it. Its first -call returns a `GlobalSelection`; the second returns a -`SummaryMaintenanceLifecyclePlan` with an assembled DAG root and lifecycle -decisions. No branch by itself deploys or executes a physical plan. +alternatives and rejection reasons, but no materialization decision. +Choose between two branches: inspect candidates (optionally ranked), or select +and assemble logical DAGs. Stage 2 materialization (#509) will decide per +sub-DAG whether to materialize and whether at ingestion or query time; until +then every summary runs at query time. No branch by itself deploys or executes +a physical plan. Known-invalid evidence rejects a logical candidate. Missing accuracy evidence leaves a constructible candidate visible in `CandidateLogicalASAPDAGs` but uncertified; default selection does not commit it without the required guarantee. Cost evidence can @@ -58,9 +50,11 @@ an unsupported physical alternative into a deployable plan. | Area | Main crate or module | Responsibility | |---|---|---| -| Shared IR | `asap-types` | Pre-ASAP and Post-ASAP expressions, schemas, workloads, guarantees, and exported plan data | +| Shared IR | `asap-types` | The unified operator IR (`ir`: one `OperatorNode` before and after ASAP optimization), schemas, workloads, guarantees, and exported plan data | +| Front-end common | `frontend-common` | Name-based `UnresolvedOp` tree shared by the front ends, and `resolve_root` into the operator IR | | Query frontends | `frontend-sql`, `frontend-promql`, `frontend-metricsql` | Parse source languages and produce canonical Pre-ASAP queries | -| ASAP-aware mapping | `asap-aware-mapping` | Candidate generation, CSE, legality, accuracy propagation, lifecycle expansion, costing, and ranking | +| ASAP-aware mapping | `asap-logical-optimizer`, `asap-physical-optimizer`, `asap-plan-selection` | #509 Stages 1–3: candidate generation, CSE, legality and accuracy propagation; physical candidates; costing and selection | +| Planner facade | `asap-planner` | Lowering dispatch, the optimization pass (`OptimizationPass`, `StagePipeline`) and `optimize` | | Developer inspection | `devtools` | Expose planner DAGs, alternatives, decisions, and explanations for inspection | | End-to-end validation | `integration-tests` | Verify behavior across frontends, mapping, and output IR | @@ -76,8 +70,7 @@ The primary output is `CandidateLogicalASAPDAGs`; `cost_sorted` derives an optio view with index-aligned costs. Downstream may inspect compatible choices across targets rather than assuming the first candidate is a feasible physical workload plan. Candidates carry logical summary algorithms, -parameters, and guarantees; selected maintenance lifecycle decisions appear -only after a summary-maintenance-lifecycle-aware helper runs. Rejection reasons +parameters, and guarantees, but no materialization decision. Rejection reasons are retained in the candidate space. ASAPQuery-backend and other downstream applications translate the candidates @@ -87,12 +80,10 @@ serving, and operational feedback. Their physical planning can reorder candidates because it has evidence that the reusable Planner does not, but it must not silently change Planner-owned semantics. -`CandidateLogicalASAPDAGs::global_selection` optionally coordinates structural choices across +`candidate_selection::global_selection` optionally coordinates structural choices across targets; `GlobalSelection::assemble_selected_dag` constructs a selected semantic DAG. -Those plain APIs do not establish physical feasibility or a -maintenance-versus-recompute decision. The lifecycle-aware selection call uses -additional workload and evidence inputs; its DAG assembly call returns a -plan with both a root and lifecycle decisions. See the [library guide](../../develop_docs/library-api.md#optional-whole-plan-selection-and-dag-assembly) +Those APIs do not establish physical feasibility or a +materialization decision. See the [library guide](../../develop_docs/library-api.md#optional-whole-plan-selection-and-dag-assembly) for the distinction. Downstream may consume candidates directly and retains responsibility for physical commitment. @@ -103,7 +94,6 @@ responsibility for physical commitment. - [Post-ASAP IR](../concepts/post-asap-ir.md) - [ASAP-aware mapping](asap-aware-mapping.md) - [Accuracy guarantees](../proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md) -- [Workload demand and summary lifecycle](../proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md) - [Physical-plan integration](physical-plan-integration.md) - [Analytical resource cost](../proposals/asap-aware-mapping/analytical-resource-cost.md) - [Searching over plans](asap-aware-plan-search.md) diff --git a/docs/design_docs/architecture/asap-aware-mapping.md b/docs/design_docs/architecture/asap-aware-mapping.md index 2600fc42d..c04fa530a 100644 --- a/docs/design_docs/architecture/asap-aware-mapping.md +++ b/docs/design_docs/architecture/asap-aware-mapping.md @@ -8,7 +8,7 @@ Given a logical query plan, the mapping layer explores alternative plans that ma Candidate search takes canonical **Pre-ASAP query roots** and produces `CandidateLogicalASAPDAGs`, a compact set of **candidate Post-ASAP DAGs**. Ranking, selection, -and summary-maintenance lifecycle decisions are subsequent operations over it; +and DAG assembly are subsequent operations over it; see [input, output, and workflows](input-output-workflow.md). For example, a percentile query might be answered by: @@ -44,7 +44,7 @@ budgets; deployment belongs to a later stage. - **Replacement Sub-DAG**: A candidate post-ASAP sub-DAG to replace a target sub-DAG. For example, a quantile aggregation may have KLL, DDSketch, and exact aggregation as alternatives. - **ReplacementStrategy**: A rule to recognize a target Sub-DAG and produces one or more valid replacement Sub-DAGs. - **Candidate Plan**: A complete post-ASAP plan formed by choosing compatible replacement alternatives across the plan. -- **Maintained population**: A multiset of qualifying records retained across evaluations and updated as members enter, change, leave or expire; multiple readouts can share this state. +- **Maintained population**: A multiset of qualifying records retained across evaluations and updated as members enter, change, leave or expire; multiple evaluations can share this state. - **Cost Model**: A model used to compare valid candidate plans according to criteria such as storage, update cost, query latency, and accuracy. The distinction between **ReplacementStrategy** and **Candidate Plan** is important. A ReplacementStrategy generates alternatives at a decision point, while a candidate plan is a complete plan that combines choices across all relevant decision points. @@ -75,7 +75,6 @@ CandidateLogicalASAPDAGs: compact candidate Post-ASAP DAGs | +--> inspect / rank +--> select and assemble logical DAGs - +--> select and assemble with summary-maintenance lifecycle decisions ``` --- @@ -104,9 +103,6 @@ The design is split into focused documents: - [ASAPPlanner planner-runtime contract](planner-runtime-contract.md) separates planner-owned search and selection from downstream physical implementation, deployment, and execution. -- [Query workloads, data workloads, and summary lifecycle maintenance](../proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md) separates - query-workload properties from data-workload properties and defines ephemeral, prepared, - shared, and continuously maintained summary-state alternatives. - [Explainability](../../develop_docs/replacement-explanations.md) describes how the planner reports available replacements using the same candidate space it optimizes. diff --git a/docs/design_docs/architecture/asap-aware-plan-search.md b/docs/design_docs/architecture/asap-aware-plan-search.md index fbc5982bc..e612740b6 100644 --- a/docs/design_docs/architecture/asap-aware-plan-search.md +++ b/docs/design_docs/architecture/asap-aware-plan-search.md @@ -84,10 +84,10 @@ sets. This avoids copying every full plan when most structure is shared. Cartesian product. `cost_sorted` returns a `RankedTargetSubDAGCandidates` view for each target. `global_selection` coordinates supported sharing and composition choices; `assemble_selected_dag(root)` assembles one selected DAG -per query root. This does not prove global physical optimality or select a -summary-maintenance lifecycle. The -[workflow design](input-output-workflow.md#workflows) explains when to use the -ordinary or summary-maintenance-lifecycle-aware path. +per query root. This does not prove global physical optimality or decide +which summaries are materialized; Stage 2 materialization (#509) will own that. +The [workflow design](input-output-workflow.md#workflows) describes the call +order. The [code architecture](../../develop_docs/asap-aware-mapping-architecture.md) describes current discovery and registry behavior; the diff --git a/docs/design_docs/architecture/evidence-dependent-candidates.md b/docs/design_docs/architecture/evidence-dependent-candidates.md index abdfbdb90..39889ef46 100644 --- a/docs/design_docs/architecture/evidence-dependent-candidates.md +++ b/docs/design_docs/architecture/evidence-dependent-candidates.md @@ -15,14 +15,14 @@ target)` returns `true`. **Uncertified** means Planner cannot make that claim: the guarantee is absent, contains unknown terms, or is known not to meet the target. An uncertified summary may still be a well-formed logical candidate; this label says nothing about whether the backend can physically execute it. -The exact `KeepPreAsap` path has an exact guarantee. +The exact path (the pre-ASAP sub-DAG kept by `retain_exact`) has an exact guarantee. | State | Planner representation | Consequence / next step | |---|---|---| | Known guarantee | `ResultGuarantee` with evaluable bound and failure probability | Planner checks whether the guarantee satisfies the query's accuracy target. If this candidate is selected, the backend checks whether its implementation can realize the selected summary; it does not re-decide the accuracy target. | | Missing accuracy/domain evidence | Symbolic `BoundExpr::Unknown` or `ProbabilityExpr::Unknown`, or `guarantee: None` on a constructible summary | Inspect `ReplacementSubDAG::has_missing_accuracy_evidence()`, obtain applicable evidence or apply explicit policy; do not claim certification. | | Missing cost | `CostModel::candidate_cost()` returns `None` for a `ReplacementSubDAG` (including a non-finite or negative legacy estimate) | Keep that logical summary/rewrite candidate in `CandidateLogicalASAPDAGs` for inspection; provide a comparable cost before selecting it by cost. This does not make it a deployable physical plan. | -| Unknown runtime support | `ReplacementSubDAG::runtime_support_evidence(model)` returns `None` | Candidate remains visible; bind a concrete implementation and confirm support before deployment. | +| Unknown runtime support | `candidate_selection::runtime_support_evidence(candidate, model)` returns `None` | Candidate remains visible; bind a concrete implementation and confirm support before deployment. | | Known invalid evidence or impossible semantics | No candidate; where supported, a `RejectedCandidate` records the error | Do not deploy. | `ResultGuarantee::has_unknown()` detects symbolic gaps. The candidate-level @@ -58,12 +58,10 @@ not emit a `RejectedCandidate` for that case. | HLL confidence | Symbolic failure probability | Reject a fully known unmet root target. | | Relative-value composition | Symbolic bound when input sign is unknown | Reject known signed input for this rule. | | Exact sum/average/extremum | Symbolic row-count probability term | Reject unsupported metric combinations. | -| Cost/rate/physical evidence | `None` cost or missing workload rate; candidate remains in `CandidateLogicalASAPDAGs` | Physical/lifecycle evaluation reports unavailable or rejected evidence. | +| Cost/rate/physical evidence | `None` cost or missing workload rate; candidate remains in `CandidateLogicalASAPDAGs` | Physical evaluation reports unavailable or rejected evidence. | | Mixed exact/summary operator | Unknown runtime support; candidate remains in `CandidateLogicalASAPDAGs` | `Some(false)` prevents construction. | -Lifecycle deployment choices are a separate output from `CandidateLogicalASAPDAGs`; their -capability/cost rejections do not erase the logical summary candidate. The -backend must still check ordinary summary family, window, and state-operation +The backend must still check summary family, window, and state-operation capabilities before deployment. - Accuracy/domain: `AccuracyEvidenceProvider` supplies quantile domains and @@ -88,7 +86,7 @@ The default `global_selection()` skips summaries that `has_missing_accuracy_evidence()` identifies as uncertified. Its `GlobalSelection::assemble_selected_dag()` result is a selected logical plan, not an instruction to deploy every candidate in `CandidateLogicalASAPDAGs`. If no alternative -is chosen at a site, DAG assembly retains the exact `KeepPreAsap` path. The +is chosen at a site, DAG assembly retains the exact pre-ASAP sub-DAG. The backend can inspect alternatives, apply its own evidence and policy, then choose a physically supported one; it must not equate candidate presence with approval. Models may explicitly opt into qualitative candidate ranking when no @@ -109,7 +107,7 @@ backend. | PromQL input | Before this PR | After this PR | |---|---|---| | `count by(job)(up)` with an ε/δ target | Hydra's shared CMS/CountSketch alternatives are absent: missing shared-grid bounds make the strategy decline the target. | Both Hydra alternatives remain in `CandidateLogicalASAPDAGs` with symbolic unknown bound/probability terms. `has_missing_accuracy_evidence()` is true; default `global_selection()` does not choose either as a certified answer. | -| `entropy_over_time(m[5m])` with an ε target | The uncalibrated frequency readout has no `SummaryEstimate` candidate. | Its `SummaryEstimate` remains inspectable with `guarantee: None`. Default selection still skips it, so candidate visibility is not an accuracy certificate. | +| `entropy_over_time(m[5m])` with an ε target | The uncalibrated frequency evaluation has no `SummaryEstimate` candidate. | Its `SummaryEstimate` remains inspectable with `guarantee: None`. Default selection still skips it, so candidate visibility is not an accuracy certificate. | | `quantile_over_time(0.9,data[5m]) / quantile_over_time(0.5,data[5m])` with an ε target | The uncertified direct DDSketch ratio is **already** visible because of #449. | Still visible with `guarantee: None`, and still skipped by default selection. This is a regression/control example, not a new candidate introduced by this PR. | For the first two rows, the observable change is the alternative set delivered @@ -121,7 +119,7 @@ evidence (for example a failure probability of `1.5`) instead produces a The corresponding reproducible checks are `cargo test -p asap-frontend-promql grouped_count_keeps_uncertified_hydra_candidates_for_backend_review`, -`cargo test -p asap-frontend-promql uncalibrated_frequency_readouts_do_not_bypass_accuracy_targets`, +`cargo test -p asap-frontend-promql uncalibrated_frequency_evaluations_do_not_bypass_accuracy_targets`, and `cargo test -p asap-integration-tests ddsketch_ratio_without_domain_proof_is_uncertified`. All three start from PromQL text and exercise frontend lowering and planning. None runs a deployed query. diff --git a/docs/design_docs/architecture/input-output-workflow.md b/docs/design_docs/architecture/input-output-workflow.md index 65ee43141..133914f21 100644 --- a/docs/design_docs/architecture/input-output-workflow.md +++ b/docs/design_docs/architecture/input-output-workflow.md @@ -18,7 +18,7 @@ Post-ASAP alternatives for the workload. | `PlanningWorkload.data_workload` | Data arrival and optional evidence about ingestion, cardinality, and distribution | No implicit default. Set `None` when unavailable for non-PromQL workloads; PromQL requires `Some(DataWorkload)` with a nonzero ingestion interval. | | Frontend-specific dependencies (outside `PlanningWorkload`) | `SqlCatalog` for SQL; `now_ms` and, when needed, `HistogramCatalog` for PromQL | `SqlCatalog` is required for SQL lowering; `now_ms` is required for PromQL lowering | | Planning models | Candidate cost/ranking and accuracy composition/checking | Used by the relevant APIs; built-in `DefaultCostModel` and `DefaultAccuracyModel` are available | -| External evidence and capabilities | Domain facts, measured costs, workload statistics, and runtime support | Supply when available and when the chosen optimization or lifecycle decision depends on them; absence is not proof | +| External evidence and capabilities | Domain facts, measured costs, workload statistics, and runtime support | Supply when available and when the chosen optimization depends on them; absence is not proof | Frontend lowering and candidate search are stages within this workflow, not additional end-to-end inputs. See [Inputs](#inputs) for the nested workload @@ -30,13 +30,14 @@ fields and [frontend dependencies](#frontend-specific-dependencies). |---|---|---| | `CandidateLogicalASAPDAGs` | The legal candidate Post-ASAP DAGs for the workload, represented compactly as canonical roots, one candidate set per target sub-DAG, and cross-target composition information | The ASAPPlanner output | -[Ranking](#ranked-view), [selection and -DAG assembly](#selection-and-dag-assembly), and -[summary-maintenance lifecycle](#summary-maintenance-lifecycle-aware-helper) APIs operate on this `CandidateLogicalASAPDAGs`. +[Ranking](#ranked-view) and [selection and +DAG assembly](#selection-and-dag-assembly) APIs operate on this `CandidateLogicalASAPDAGs`. These are alternative uses of the candidate space, not mandatory sequential -stages. `CandidateLogicalASAPDAGs` itself has no selected summary-maintenance lifecycle, and -its candidates do not choose precompute versus query-time placement: a chosen -lifecycle assignment sets each node's execution timing. +stages. Its candidates do not choose ingestion-time versus query-time +placement: a `MaterializationAssignment` sets each node's execution timing. +Stage 2 materialization (#509) will decide per sub-DAG whether to materialize +and whether at ingestion or query time; until then every summary runs at query +time. The candidate DAGs are logical planning artifacts. ASAPPlanner does **not** produce a deployed executable plan; downstream systems bind physical operators, @@ -58,7 +59,7 @@ PlanningWorkload + frontend dependencies + planning models/evidence Suppose a dashboard evaluates `count_over_time(up[5m])` once a minute, and `up` receives a sample every 15 seconds. This diagram traces the concrete -inputs and the three possible uses of the same candidate space: +inputs and the two possible uses of the same candidate space: ```mermaid flowchart TD @@ -66,16 +67,12 @@ flowchart TD D["data_workload: continuous arrival; declared ingestion interval 15 s"] T["Frontend argument: now_ms"] F["PromQL lowering"] - R["One canonical QueryExpr root"] + R["One canonical OperatorNode root"] S["Candidate search"] P["CandidateLogicalASAPDAGs: logical choices for this root"] I["cost_sorted: inspect choices"] G["global_selection + assemble_selected_dag(root)"] - L["One selected Post-ASAP DAG; exact KeepPreAsap if no optimization is selected"] - X["Extra lifecycle inputs: horizon; update rate; capabilities; comparable summary/raw costs"] - H["Summary-maintenance-lifecycle-aware selection"] - HM["Assemble one selected DAG and decide summary maintenance"] - O["SummaryMaintenanceLifecyclePlan: assembled DAG root + maintenance/recompute decision"] + L["One selected Post-ASAP DAG; the exact pre-ASAP sub-DAG if no optimization is selected"] B["Backend: bind physical operators, deploy, and execute"] Q --> F D --> F @@ -83,16 +80,13 @@ flowchart TD F --> R --> S --> P P --> I P --> G --> L --> B - P --> H - X --> H --> HM --> O --> B ``` “Predictable” says the query is known in advance; it is independent of its one-minute recurrence. The `CandidateLogicalASAPDAGs` may contain an exact count-summary -realization, but it is not a deployed query. Without the extra lifecycle -inputs, the caller can still inspect candidates or obtain a logical DAG; it -cannot conclude that maintaining a summary is cheaper than recomputing raw -results. +realization, but it is not a deployed query. The caller can inspect candidates +or obtain a logical DAG; deciding whether maintaining a summary is cheaper than +recomputing raw results belongs to Stage 2 materialization (#509). For contrast, a one-time SQL query needs a catalog but need not supply data arrival evidence merely to inspect logical alternatives: @@ -102,7 +96,7 @@ flowchart LR Q["query_batch: SELECT COUNT(*) FROM metrics; invocations 1; AdHoc"] C["SqlCatalog: resolves metrics and its columns"] F["SQL lowering"] - R["One QueryExpr root"] + R["One OperatorNode root"] P["Candidate search → CandidateLogicalASAPDAGs"] Q --> F C --> F @@ -110,8 +104,7 @@ flowchart LR ``` In this SQL example, `data_workload` can be `None` if the chosen lowering and -search rules do not consume it. The lifecycle helper is not needed merely to -inspect the `CandidateLogicalASAPDAGs`. +search rules do not consume it. --- @@ -175,9 +168,9 @@ struct BatchEntry { |---|---:|---|---| | `query` | Yes | Raw query text in `QueryWorkload.language`. | `count(up)` determines the expression to lower and plan. | | `requirements` | Yes | Accuracy and response-latency requirements. Defaults mean exact accuracy and unspecified latency. | An explicit ε target permits approximate candidates; the exact default does not. | -| `predictability` | Yes as a field; `Unknown` is allowed | Whether the query is ad hoc, known in advance, or unknown. `known_at` records when a predictable query became known. | A report known at 10:00 and scheduled for 11:00 may use a `Prepared` summary before execution. `AdHoc` or `Unknown` does not establish that eligibility. | +| `predictability` | Yes as a field; `Unknown` is allowed | Whether the query is ad hoc, known in advance, or unknown. `known_at` records when a predictable query became known. | A report known at 10:00 and scheduled for 11:00 could have its summary prepared before execution. `AdHoc` or `Unknown` does not establish that eligibility. | | `invocations` | Yes, nonzero | Number of executions in this finite batch. | Ten executions can amortize one summary build differently from one execution. | -| `execute_at` | Optional | Known execution time. | The `Prepared` case above also needs an execution time; without it Planner cannot establish a preparation window. | +| `execute_at` | Optional | Known execution time. | The prepared case above also needs an execution time; without it no preparation window can be established. | | `time_selection` | Yes | Whether the query follows current data or a historical interval, its lookback, and any fixed upper bound. | A moving five-minute window can require deletion/window support that a fixed historical interval does not. | ##### `repeating_queries: Option>` @@ -197,7 +190,7 @@ struct RepeatingEntry { | `query` | Yes | Raw query text in `QueryWorkload.language`. | `rate(up[5m])` determines the expression to lower and plan. | | `demand` | Yes | A nonzero fixed interval, fixed interval with evaluation phase, nonempty explicit schedule, or evidence-backed estimated rate. | A query every minute produces more expected reads over a horizon than one every hour. | | `requirements` | Yes | Accuracy and response-latency requirements. | An exact dashboard query cannot use an approximate summary solely because it is cheaper. | -| `predictability` | Yes as a field; `Unknown` is allowed | Records whether future executions are known in advance; independent of recurrence. | Current lifecycle code does not use this field for repeating entries; set `Unknown` if no predictability claim is available. | +| `predictability` | Yes as a field; `Unknown` is allowed | Records whether future executions are known in advance; independent of recurrence. | Current planner code does not use this field for repeating entries; set `Unknown` if no predictability claim is available. | | `time_selection` | Yes | Event-time scope, optional lookback, and optional fixed `as_of` time. | A live five-minute lookback differs from a fixed historical range when checking maintenance capabilities. | ##### Shared entry fields @@ -214,9 +207,9 @@ fields expand as follows: | `TimeSelection` | `lookback` | Optional event-time duration selected before the upper bound. | | `TimeSelection` | `as_of` | Optional fixed upper-bound timestamp; `None` means planning/evaluation time. | -Frontend lowering produces one Pre-ASAP `QueryExpr` root for each normalized +Frontend lowering produces one Pre-ASAP `Rc` root for each normalized query entry. The caller must retain each root's association with its workload -entry for later recurrence and lifecycle planning. +entry for later recurrence and materialization planning. #### `data_workload: Option` @@ -285,8 +278,8 @@ latter cannot be fabricated by one. | Accuracy model | Target-aware search takes an `AccuracyModel`; `DefaultAccuracyModel` is available. Default strategies also use it for candidate construction. | Composes candidate guarantees and checks them against requested accuracy. The model does not itself provide missing data-domain facts. | | Cost model | Candidate strategies and `cost_sorted`/`global_selection` use a `CostModel`; `DefaultCostModel` is available. | Ranks or selects candidates. The built-in model is not a measured deployment cost for every physical implementation. | | Accuracy/domain evidence | `AccuracyEvidenceProvider`; default strategies use `NoAccuracyEvidence` when no provider is supplied. | Input ranges, nonempty populations, Top-K intervals, and similar facts can certify or rule out particular approximations. Missing facts remain unknown. | -| Measured cost evidence | Supplied through a deployment-specific cost model or physical-evidence provider when cost-based physical/lifecycle comparison is needed. | CPU, memory, and I/O estimates must be comparable before claiming a summary beats raw recomputation. | -| Runtime/lifecycle capabilities | Passed to lifecycle APIs or checked by deployment-specific providers; `SummaryMaintenanceLifecycleCapabilities::default()` enables all four lifecycle shapes, so it is not proof of actual backend support. | Prevents choosing a maintenance/window operation the intended executor cannot implement. | +| Measured cost evidence | Supplied through a deployment-specific cost model or physical-evidence provider when cost-based physical comparison is needed. | CPU, memory, and I/O estimates must be comparable before claiming a summary beats raw recomputation. | +| Runtime capabilities | Checked by deployment-specific providers. | Prevents choosing a maintenance/window operation the intended executor cannot implement. | For example, the query `quantile_over_time(0.9, data[5m]) / quantile_over_time(0.5, data[5m])` does not tell Planner whether the windows @@ -305,9 +298,6 @@ domain evidence is missing; automatic `global_selection` does not choose it. See the [candidate-search reference](../../develop_docs/library-api.md#generate-and-rank-candidates) for this backend-selection path. -Additional inputs for a Planner-owned maintenance decision are listed with the -[summary-maintenance-lifecycle-aware helper](#summary-maintenance-lifecycle-aware-helper). - --- ## Output @@ -334,7 +324,7 @@ below. A future higher-level API could hide `CandidateLogicalASAPDAGs` behind th the current interface lets an integrator own them. DAG assembly connects choices after selection and does not replace this candidate interface. -Here, a **root** is the top-level `Rc` for a workload query. A +Here, a **root** is the top-level `Rc` for a workload query. A **target** is any discovered sub-DAG that may be replaced, including roots. For `count(up) + 1`, the addition is a root and `count(up)` can be an inner target. `TargetSubDAGCandidates` holds the alternatives for one such target. @@ -361,7 +351,7 @@ All paths start by lowering the workload and searching for candidates: ```text PlanningWorkload + frontend dependencies + planning models/evidence - -> frontend lowering: one QueryExpr root per normalized query entry + -> frontend lowering: one OperatorNode root per normalized query entry -> search_workload_with_targets -> CandidateLogicalASAPDAGs ``` @@ -375,12 +365,11 @@ Then choose the operation matching the caller's responsibility: | Purpose | Operation | Result | |---|---|---| | Inspect candidates or let the backend choose | [Ranked view](#ranked-view), if ranking is useful | Per-target candidate lists and costs | -| Ask Planner to choose logical computations; backend owns summary maintenance | [Selection and DAG assembly](#selection-and-dag-assembly) | One selected Post-ASAP DAG root per query | -| Ask Planner to also decide summary maintenance versus raw recomputation | [Summary-maintenance-lifecycle-aware helper](#summary-maintenance-lifecycle-aware-helper) | One plan containing a DAG root and maintenance decisions per query | +| Ask Planner to choose logical computations | [Selection and DAG assembly](#selection-and-dag-assembly) | One selected Post-ASAP DAG root per query | ### Ranked view -`CandidateLogicalASAPDAGs::cost_sorted` returns one `RankedTargetSubDAGCandidates` for each +`candidate_selection::cost_sorted` returns one `RankedTargetSubDAGCandidates` for each `TargetSubDAGCandidates` entry. Conceptually, it is the same target's alternatives in cost-model preference order where the model defines one (otherwise discovery order), with one displayed cost per alternative. It is @@ -390,7 +379,7 @@ The return type is `Vec>`; each element has thi ```rust struct RankedTargetSubDAGCandidates<'a> { - target: &'a Rc, + target: &'a Rc, consumer_count: usize, candidates: Vec<&'a ReplacementSubDAG>, costs: Vec, // costs[i] describes candidates[i] @@ -406,7 +395,7 @@ physical deployability. ### Selection and DAG assembly The input is `CandidateLogicalASAPDAGs` and a cost model. Call -`CandidateLogicalASAPDAGs::global_selection(&cost_model)` once for the workload, then +`candidate_selection::global_selection(&space, &cost_model)` once for the workload, then `GlobalSelection::assemble_selected_dag(root)` for each wanted query root. These are two public APIs, not one combined call: N roots require one selection and N assembly calls. Each successful assembly returns one DAG root; the caller @@ -430,71 +419,13 @@ the result for one query root. | **Output:** one selected logical [Post-ASAP DAG](../concepts/post-asap-ir.md) per query root | Each output DAG specifies the chosen operators, parameters, and accuracy -guarantees. Its root is represented by `Rc`; the +guarantees. Its root is an `Rc` (the same IR as the input, +with some nodes now ASAP operators) and carries no execution timing yet; the [API reference](../../develop_docs/library-api.md#api-definition-and-example) describes the function signatures and return handling. -This path selects how to compute the query, not how to maintain summary state. - -### Summary-maintenance-lifecycle-aware helper - -This workflow performs both candidate selection and DAG assembly, incorporating -summary-maintenance lifecycle costs. Use it when ASAPPlanner owns the decision -to maintain summaries versus recompute raw data. It is not needed for candidate -inspection or when the downstream backend owns that decision. - -Starting from an existing `CandidateLogicalASAPDAGs`, call these two public helpers in order; -there is no need to run the ordinary selection/assembly workflow first: - -1. `global_selection_with_summary_maintenance_lifecycles` uses the workload - binding, lifecycle capabilities, and comparable costs to choose compatible - candidates across target sub-DAGs. It returns `GlobalSelection`, not a DAG or a - deployment plan. -2. For each wanted query root, `assemble_selected_dag_with_summary_maintenance_lifecycles` - takes that selection and root, constructs a Post-ASAP DAG, compares the - selected summary's maintenance cost with raw recomputation, and returns - `Result, SummaryMaintenanceLifecycleAssemblyError>`. - When a summary does not beat a - known raw cost, or a required comparable cost is unavailable, the result - retains the exact `KeepPreAsap` root and no summary deployments. - -As in ordinary selection, one selection call serves the workload and assembly -is per root. The second helper calls `assemble_selected_dag` internally; callers -do not need a separate assembly call. Neither helper creates a materialized view -or deploys runtime state. -The output is a selected logical DAG with lifecycle decisions, not an executable -deployment plan. Any claim of optimization is relative to the supplied cost -model, evidence, and available candidates. -See the [library guide's lifecycle and capabilities section](../../develop_docs/library-api.md#lifecycle-and-capabilities) -for an API example and the capability contract. - -Across the two calls, the caller supplies these parameters: - -| Helper parameter | Source | Required | -|---|---|---:| -| `CandidateLogicalASAPDAGs` | Canonical ASAPPlanner output; passed to selection | Yes | -| `GlobalSelection` and one root | Selection result and a root in that `CandidateLogicalASAPDAGs`; passed to DAG assembly | Yes for each assembled root | -| Workload binding | `QueryWorkload` plus the workload-entry indices associated with each root | Yes | -| Planning time (`now_ms`) | Caller clock in Unix milliseconds | Yes | -| Planning horizon | Caller policy | Conditional: required for finite totals over recurring demand | -| Data arrival and update rate | `DataWorkload` evidence | Conditional: required to cost continuous maintenance | -| Lifecycle capabilities | Deployment/runtime provider | Yes for checking deployable lifecycle alternatives | -| Summary and raw cost information | Cost model and physical-evidence provider | Yes for a cost-based maintenance-versus-recompute decision | - -Recurrence and time selection are already fields of the bound `QueryWorkload`; -they are not duplicated as separate top-level inputs. Similarly, data arrival -and update rate are read from the optional `DataWorkload`. Missing required -facts remain unknown rather than being treated as zero. - -The per-query output, `SummaryMaintenanceLifecyclePlan`, **contains** the -Post-ASAP DAG rather than being a parallel representation. It records: - -* the assembled Post-ASAP DAG root (`Rc`); -* lifecycle choices for summary state; -* planning horizon and expected reads/updates; -* selected window implementation and guarantees; -* comparable summary and raw-recomputation costs; and -* whether raw recomputation was selected. +This path selects how to compute the query, not whether summary state is +materialized. --- diff --git a/docs/design_docs/architecture/metricsql-frontend.md b/docs/design_docs/architecture/metricsql-frontend.md index ddc463239..ffb45b728 100644 --- a/docs/design_docs/architecture/metricsql-frontend.md +++ b/docs/design_docs/architecture/metricsql-frontend.md @@ -9,16 +9,17 @@ on VictoriaMetrics and models MetricsQL syntax directly, including `WITH`, rollup expressions, step-relative durations, MetricsQL binary operators, aggregate limits, or-delimited matchers, and `keep_metric_names`. -The frontend walks that AST directly and emits the existing canonical -`QueryExpr`. It does not add MetricsQL fields to `QueryExpr`, SDS descriptors, -or the physical summary DAG. +The frontend walks that AST directly into the shared name-based `UnresolvedOp` +tree (`asap-frontend-common`) and calls `resolve_root`, which returns the +canonical `Rc` DAG. It does not add MetricsQL fields to the +operator IR, SDS descriptors, or the physical summary DAG. ```text MetricsQL source | MetricsqlExpr (extension semantics retained) | -canonical QueryExpr +UnresolvedOp tree --resolve_root--> canonical OperatorNode DAG | existing ASAP-aware mapping and physical Summary DAG ``` @@ -33,8 +34,8 @@ existing ASAP-aware mapping and physical Summary DAG | Common rollups: rate/increase/derivatives and statistical `*_over_time` | Existing per-entity canonical intents over the lowered range. | | PromQL arithmetic, comparison, and set binary operators without modifiers | Existing canonical `BinaryOp`. | | `default_rollup(selector[range])` | Lower to `Aggregate(LastOverTime)` over the explicit `TimeRange`. | -| `default_rollup(selector)` | Reject for exact fallback because the implicit lookbehind window depends on the runtime evaluation step, which is not a property of canonical `QueryExpr`. | -| `expr keep_metric_names` | Parsed natively, then rejected for exact fallback because canonical `QueryExpr` does not carry metric-name lineage. | +| `default_rollup(selector)` | Reject for exact fallback because the implicit lookbehind window depends on the runtime evaluation step, which is not a property of the canonical operator IR. | +| `expr keep_metric_names` | Parsed natively, then rejected for exact fallback because the canonical operator IR does not carry metric-name lineage. | | `if`, `ifnot`, `default`, aggregate `limit`, or-delimited matchers, binary match modifiers | Parsed natively and rejected until the canonical executor has the exact semantics. | | `WITH` | Expanded by the native parser; the expanded expression lowers when every resulting node is supported. | diff --git a/docs/design_docs/architecture/physical-plan-integration.md b/docs/design_docs/architecture/physical-plan-integration.md index 7fffc9fdf..2fb056412 100644 --- a/docs/design_docs/architecture/physical-plan-integration.md +++ b/docs/design_docs/architecture/physical-plan-integration.md @@ -5,14 +5,13 @@ This document defines the boundary between ASAPPlanner's logical plans, physical lowering, statistics resolution, and analytical resource estimation. It answers which representation is authoritative at each stage and prevents -the cost model from being coupled directly to either logical IR. +the cost model from being coupled directly to the logical IR. The integration pipeline is: ```text -pre-ASAP QueryExpr ─┐ - ├─ physical lowering ─> PhysicalOperator DAG -post-ASAP SummaryExpr┘ │ +logical OperatorNode DAG ─ physical lowering ─> PhysicalOperator DAG +(NonASAPOp + ASAPOp nodes) │ v OperatorStatistics │ @@ -30,8 +29,8 @@ Each representation is authoritative for a different concern: | Representation | Authoritative concern | |---|---| -| `QueryExpr` | Original exact query semantics: sources, predicates, relational and PromQL operations, and output shape. | -| `SummaryExpr` | Logical summary semantics: selected family, grouping strategy, summary composition, and summary readout. | +| `NonASAPOp` nodes | Exact query semantics: sources, predicates, relational and PromQL operations, and output shape. | +| `ASAPOp` nodes | Logical summary semantics: selected family, grouping strategy, summary composition, and summary evaluation. | | `PhysicalOperator` DAG | Selected executable algorithms, their configuration, physical identity, edges, and execution multiplicity. | | `OperatorStatistics` | Workload-dependent evidence required by each selected physical operator's resource formula. | | `ResourceEstimate` | Estimated CPU operations, peak live memory, and physical source/disk reads over one comparison scope. | @@ -39,7 +38,7 @@ Each representation is authoritative for a different concern: `PhysicalOperator` is therefore the source of truth for the operator vocabulary consumed by analytical costing. `OperatorStatistics` corresponds one-to-one with that vocabulary. It must not independently invent operator kinds or copy -all variants from either logical IR. +all variants from the logical IR. The canonical physical-plan types should live at a neutral boundary shared by lowering, costing, explanation, and downstream compilation. Their conceptual @@ -51,7 +50,7 @@ being established. One logical operation may choose between algorithms or expand into a physical sub-DAG. Conversely, one physical operator may implement nodes originating -from either logical IR. +from either operator category (`NonASAPOp` or `ASAPOp`). Examples include: @@ -61,21 +60,21 @@ Examples include: supported join algorithm. - `SummaryAgg` may lower to an exact accumulator build, CMS build, KLL build, or another physical summary algorithm selected by the candidate. -- `SummaryEstimate` must lower to a readout operator compatible with the +- `SummaryEstimate` must lower to a evaluation operator compatible with the concrete summary state it consumes. - shared logical sub-DAGs become shared physical nodes only when they refer to the same physical identity and compatible evidence. -For this reason, aligning `OperatorStatistics` directly with `QueryExpr` would -lose post-ASAP summary implementations, while aligning it directly with -`SummaryExpr` would lose raw query operators and physical algorithm choices. +For this reason, aligning `OperatorStatistics` directly with the logical +operators would lose physical algorithm choices, and with only one category +would lose either summary implementations or raw query operators. ## Lowering obligations Physical lowering is complete only when it recursively lowers the entire selected candidate DAG. It must: -1. preserve the semantics and source coverage of the logical candidate; +1. preserve the semantics and scan selection of the logical candidate; 2. select an explicit physical algorithm for every logical operation; 3. carry algorithm configuration on the physical operator rather than in a generic statistics record; @@ -91,14 +90,16 @@ its modeled descendants is invalid because it undercounts the candidate. ### Pre-ASAP lowering -`KeepPreAsap` recursively lowers its contained `QueryExpr`. Typical physical +Every `NonASAPOp` node lowers recursively, whether it is in a raw query or +kept exact inside a post-ASAP plan. Typical physical operators include scans, filters, projections, hash aggregates, joins, ordering, bounded Top-K, limits, and PromQL-specific operators. The selected physical algorithm, rather than the logical spelling, determines the formula. ### Post-ASAP lowering -Every `SummaryExpr` operation also needs explicit physical realization: +Every `ASAPOp` node, and every exact operator composed with one, also needs +explicit physical realization: | Logical summary operation | Required physical realization | |---|---| @@ -107,38 +108,24 @@ Every `SummaryExpr` operation also needs explicit physical realization: | `SummaryMerge` | merge operator over compatible concrete summary states | | `SummarySubtract` | subtract operator supported by the selected state representation | | `SummaryDelete` | physical deletion/update operator supported by the selected representation | -| `SummaryEstimate` | family- and query-specific readout operator | -| `KeepPreAsap` | recursive lowering of the contained `QueryExpr` | -| `BinaryOp` | binary evaluation preserving operand order, execution timing and any typed finite/relative-division guard | -| `ValueOperation` | concrete realization of the value operation with its required execution timing and data state | -| `RelationalJoin` | concrete row-join algorithm preserving join kind and predicate | -| `RelationalJoin` with `JoinKind::Semi` | retain left rows matching explicit right-side keys; candidate pruning carries completeness evidence and ordinary TopK ranks the result | +| `SummaryEstimate` | family- and query-specific evaluation operator | +| `FinalizeExactAccumulator` | exact-state finalization before value consumers | +| `MaintainPopulation` / `EvaluatePopulation` | maintained-population update and its aggregate or TopK-prefix evaluation | +| retained `NonASAPOp` sub-DAG | recursive lowering of the exact operators (see above) | +| `BinaryOp` | binary evaluation preserving operand order, the node's execution timing and any typed finite/relative-division guard | +| `Project` / `Filter` / `Sort` / `Limit` / `Aggregate` over a evaluation | concrete realization at the node's execution timing and data state | +| `Join` | concrete row-join algorithm preserving join kind and predicate | +| `Join` with `JoinKind::Semi` | retain left rows matching explicit right-side keys; candidate pruning carries completeness evidence and ordinary TopK ranks the result | This table is a completeness requirement, not a claim that every realization already exists. Until lowering introduces an explicit physical operator, statistics contract, validation rule, and resource formula for an operation, a candidate containing it is unavailable. -The streaming integration can consume a complete binding through -`SummaryNodeEvidence`. That binding is keyed to exact `SummaryNode` -identities and uses structured evidence for aggregate state, join, merge, -subtract, delete, readout, and retained pre-ASAP work. It is a physical -evidence boundary, not automatic physical lowering: a deployment must still -select each concrete implementation and provide all edges, resource facts, -multiplicities, source ownership, and stable physical identities. The planner -fails closed when any reachable `SummaryExpr` node lacks that binding. - -The raw/query portion of a streaming comparison remains a `PhysicalDAG` using -the canonical `PhysicalOperator` and `OperatorStatistics` pairing. Summary -evidence is kept separate only where lifecycle-driven update, retention, and -expiration multiplicities require facts beyond the query-DAG -`Once`/`PerEvaluation` schedule. It must not redefine workload, lifecycle, or -summary-family semantics. - -Lifecycle choice affects the physical DAG but does not replace it. Ephemeral, -prepared, shared, and continuously maintained alternatives determine when -build, update, readout, merge, subtract, or delete nodes execute. The physical -operators still determine how each execution consumes CPU, memory, and I/O. +Materialization choice (Stage 2, #509) affects the physical DAG but does not +replace it: it determines when build, update, evaluation, merge, subtract, or +delete nodes execute. The physical operators still determine how each execution +consumes CPU, memory, and I/O. ## Statistics contract @@ -427,7 +414,7 @@ recovering average semantics from query text. ### Candidate pruning is a sub-DAG -Candidate-based TopK uses a summary key readout, a general semi-join over +Candidate-based TopK uses a summary key evaluation, a general semi-join over explicit matching key columns, grouped Sort by the authoritative score, and grouped Limit. Sort and Limit carry the same partition keys. The join preserves authoritative left-side values and does not rank or limit @@ -440,10 +427,12 @@ fields. The phase assignment API updates producer edge states and rejects an ingestion computation that depends on query-time work. Deployment capability, storage readiness, schemas and approximation guarantees remain separate checks. -Post-ASAP DAG wire version 4 removes the special membership operator, its edge +Post-ASAP DAG wire version 4 removed the special membership operator, its edge roles and the duplicate operator phase fields without compatibility aliases. +Version 6 (current) exports one node per operator: retained exact operators are +`Relational` nodes, not embedded sub-DAGs. -Post-ASAP DAG wire version 6 adds a per-measure row predicate to the aggregate +Post-ASAP DAG wire version 7 adds a per-measure row predicate to the aggregate operators (#466): `filters` on the exact aggregate value operation, parallel to its measures, and `filter` on `SummaryAgg`, gating which rows update the summary state. The version bump makes an older reader fail loudly instead of @@ -465,38 +454,8 @@ a numeric entity key is not a score. Exact accumulator inputs are explicitly finalized before row operators consume them. None of these operations proves candidate completeness; that evidence belongs to the semi-join's pruning step. -The semantic `SummaryExpr` constructors still propose an initial execution -layout. Uniform phase assignment applies to the exported post-ASAP DAG; +The logical DAG carries no execution layout: `apply_materialization_timings` +writes each node's timing from a `MaterializationAssignment` before export (all +query time by default). Uniform phase +assignment applies to the exported post-ASAP DAG; it is not a claim that every deployment has implemented every placement. - - -## Summary cost evidence across data-arrival modes - -`SummaryMaintenanceCostModel` binds `SummaryNodeEvidence` and -`SummaryOperatorEvidence` independently of data-arrival mode. `ComparisonScope` -and the canonical `DataWorkload` determine arrival semantics; individual operator -resource records do not define another workload model. - -`SummaryMaintenanceInputs::from_workload` requires fresh snapshot cardinality. -For `AtRest`, it derives zero arrivals without requiring ingestion-rate evidence; -a fresh nonzero or invalid rate contradicts that declaration and is rejected. -For `ContinuouslyIngesting`, fresh, finite, nonnegative rate evidence remains -mandatory. Missing continuous rate evidence is never treated as zero. -Raw and summary evidence supplied directly by a provider obey the same arrival -invariant. Their source lineage, horizon, evaluation count, and snapshot dimensions -must still match. The existing lifecycle planner selects direct builds for a fixed -snapshot and charges bootstrap work, result evaluation, and retention; it charges -no arrival updates. This does not add computation-placement policy. - -`Mixed` and `Unknown` remain unsupported for analytical comparisons: the current -workload schema cannot identify separate backlog and arrival populations. The -adapter fails explicitly rather than guessing a split. The estimator version is -`summary-maintenance-resource-v2`; evidence type names drop the `Streaming` prefix -(`SummaryMaintenanceInputs`, `SummaryPhysicalInputEvidence`, `SummaryAggregateEvidence`, -`RetainedSubDAGEvidence`, `RawInputEvidence`, and the summary window/alternative -types). Update source imports; no legacy-name aliases are provided. - -Regressions cover a fixed snapshot with no rate evidence, contradictory arrival -rates, scope mismatches, missing continuous-rate/cardinality evidence, and actual -lifecycle selection of a completely costed at-rest summary against its raw scan. -The existing continuous-ingestion and mixed-arrival rejection tests remain. diff --git a/docs/design_docs/architecture/planner-runtime-contract.md b/docs/design_docs/architecture/planner-runtime-contract.md index 2d071e654..0077252ca 100644 --- a/docs/design_docs/architecture/planner-runtime-contract.md +++ b/docs/design_docs/architecture/planner-runtime-contract.md @@ -4,33 +4,27 @@ ASAPPlanner produces `CandidateLogicalASAPDAGs`, a compact logical candidate space. Integrators may select candidates downstream or ask Planner's helpers to select and assemble -DAGs. Summary-maintenance lifecycle decisions belong to Planner only when the -integration uses its lifecycle-aware workflow; physical deployment and execution -remain downstream. The [input/output/workflow design](input-output-workflow.md) -defines this boundary. - -A downstream provider can report implementation alternatives and their cost and -accuracy evidence for a Planner-owned comparison. The resulting -`SummaryMaintenanceLifecyclePlan` contains a Post-ASAP DAG root and maintenance -decisions; it is not an executable plan. Repeated provider calls do not constitute -an implemented end-to-end replanning or deployment-transition protocol. +DAGs. Stage 2 materialization (#509) will decide per sub-DAG whether to +materialize and whether at ingestion or query time; until then every summary +runs at query time. Physical deployment and execution remain downstream. The +[input/output/workflow design](input-output-workflow.md) defines this boundary. ## Three decision layers | Layer | Owner | Examples | |---|---|---| | Logical candidate semantics | ASAPPlanner | Query rewrite; summary family and parameters; grouping; accuracy guarantees when established. | -| Summary maintenance and realization selection | Planner helpers when delegated to Planner; otherwise downstream | `Ephemeral`, `Prepared`, `Shared`, `ContinuouslyMaintained`; `DirectBuild` or `Incremental`; window implementations compared using provider evidence. | +| Summary materialization and realization selection | Stage 2 materialization (#509); downstream until then | Ingestion-time maintenance or query-time computation per summary state; direct build or incremental update; window implementations compared using provider evidence. | | Concrete implementation and deployment | ASAPQuery-backend and its workload optimizer | Library and data-structure implementation, exact pane layout, placement, sharding, storage, transmission, materialization IDs, executor configuration, and workload-wide assignment. | ASAPCollector and the ASAPQuery data plane execute the compiled downstream plans. They validate capabilities and plan identities, maintain or read the specified state, and report runtime observations. They do not silently choose -a different summary, lifecycle, or realization framework. +a different summary, materialization, or realization framework. ## Incremental-maintenance example -When the integration delegates summary-maintenance decisions to Planner, +Once Stage 2 materialization (#509) owns summary-maintenance decisions, ASAPPlanner may decide that a logical summary should be incrementally maintained: new data updates existing summary state. It may also select the planner-visible window realization—such as tumbling, sliding/panes, or an @@ -43,7 +37,7 @@ pane representation, runtime operator implementation, placement, sharding, watermark behavior, and materialization identifiers. ASAPCollector maintains the compiled panes and summary state. -Thus `Incremental` describes the state-update lifecycle, while tumbling, +Thus incremental update describes how state is maintained, while tumbling, sliding, and exponential-histogram describe realization algorithms. They are distinct axes, but both can participate in ASAPPlanner's candidate space. The backend still owns how the selected algorithms are physically realized. @@ -53,7 +47,7 @@ backend still owns how the selected algorithms are physically realized. The same contract applies when ASAPPlanner selects a summary algorithm. Planner can choose KLL rather than DDSketch, while downstream chooses the concrete KLL implementation and runtime configuration that satisfies the selected parameter -and accuracy contract. Empirical KLL error, update work, state size, and readout +and accuracy contract. Empirical KLL error, update work, state size, and evaluation work observed on a particular workload can be fed back as evidence for later Planner comparisons. @@ -64,24 +58,23 @@ selected algorithm's semantics or guarantees. ## Iterative planning protocol (future integration) The sequence below is an intended integration design, not one shipped public -API or a required path for every caller. Current provider and lifecycle helpers -support a bounded planning decision; cross-run identity, migration, activation, -and rollback are not an end-to-end Planner protocol. +API or a required path for every caller. Cross-run identity, migration, +activation, and rollback are not an end-to-end Planner protocol. -1. ASAPPlanner enumerates semantically valid logical summaries, lifecycle +1. ASAPPlanner enumerates semantically valid logical summaries, materialization alternatives, and registered realization strategies. 2. A physical-plan provider maps those candidates to executor-feasible complete alternatives. Unsupported candidates are omitted or explicitly rejected. 3. The provider binds a stable alternative identity and complete evidence: - source coverage, input/output edges, operation counts, update and bootstrap + scan selection, input/output edges, operation counts, update and bootstrap fanout, retained state, CPU, memory, I/O, and accuracy facts. 4. ASAPPlanner keeps constructible candidates with missing evidence visible - in `CandidateLogicalASAPDAGs` but does not certify unknown accuracy. The - summary-maintenance-lifecycle-aware workflow compares supported alternatives - over the same workload horizon. Missing or incomparable costs do not establish + in `CandidateLogicalASAPDAGs` but does not certify unknown accuracy. + Materialization compares supported alternatives over the same workload + horizon. Missing or incomparable costs do not establish that maintaining a summary beats raw recomputation; structural scores and optimistic zeroes are not substitutes. -5. ASAPPlanner outputs the selected Post-ASAP semantics, lifecycle guarantees, +5. ASAPPlanner outputs the selected Post-ASAP semantics, materialization choices, realization contract, and chosen provider identity. 6. ASAPQuery-backend compiles that result into consistent `CollectorPlan`, `BackendPlan`, and `QueryPlan` projections and performs deployment-level and @@ -106,11 +99,8 @@ such as cache behavior, serialization overhead, compression, spill I/O, or data-distribution-dependent sketch error. Provenance and version information must accompany those facts so stale observations fail closed. -`SummaryPhysicalPlanAlternative` is the current integration point for a -complete provider-enumerated implementation. Its identity is returned with the -winning lifecycle combination. More structured planner-owned realization -contracts can refine the candidate space without moving executor -implementation into ASAPPlanner. +More structured planner-owned realization contracts can refine the candidate +space without moving executor implementation into ASAPPlanner. ## Workload-wide optimization @@ -124,7 +114,7 @@ The ASAPQuery configuration and MIP formulations can supply physical alternatives and coefficients. Their general principles also inform Planner costing: arrival rate scales ingestion work, overlapping active windows multiply update work and live state, retained windows consume memory, and -merge/subtract/readout work scales with query recurrence. Disagreement between +merge/subtract/evaluation work scales with query recurrence. Disagreement between formulations must become distinct explicit alternatives, not hidden assumptions in one cost formula. @@ -146,7 +136,7 @@ in one cost formula. automatically selected. - Shared logical nodes remain shared across the planner-runtime contract; physical sharing additionally requires compatible filters, grouping, windows, parameters, - lifecycle, and guarantees. + materialization, and guarantees. - Collector, backend, and query plans are projections of one compiled decision and cannot be optimized independently into inconsistent semantics. @@ -155,7 +145,6 @@ in one cost formula. - [Post-ASAP IR](../concepts/post-asap-ir.md) - [Physical plan integration](physical-plan-integration.md) - [Analytical resource cost](../proposals/asap-aware-mapping/analytical-resource-cost.md) -- [Workload demand and summary lifecycle](../proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md) - [ASAPCollector physical compilation](https://github.com/ProjectASAP/ASAPCollector/blob/87684f4b61514382d8b087724694f93187bfc19c/docs/design_docs/control-plane/post-asap-physical-compilation.md) - [ASAPQuery configuration formulation](https://github.com/ProjectASAP/ASAPQuery/blob/main/.design_docs/sketch-config-optimization-formulation.md) - [ASAPQuery optimizer MIP formulation](https://github.com/ProjectASAP/ASAPQuery/blob/main/.design_docs/optimizer-mip-formulation.md) diff --git a/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md b/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md index 72e626a2e..c3daefe75 100644 --- a/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md +++ b/docs/design_docs/architecture/updated_interface_with_pluggable_optimization.md @@ -9,8 +9,8 @@ What that buys: * One call in place of six across three stages. `CandidateLogicalASAPDAGs` and `GlobalSelection` no longer appear in user code. -* The root-to-entry bindings a caller used to build by hand are derived, and - their ordering contract is checked rather than assumed. +* The root-to-entry binding a caller used to build by hand is derived, and + its ordering contract is checked rather than assumed. * A new optimization algorithm can be freely implemented as a trait implementation, rather than a rule disguised to fit a two-phase pipeline it does not share. @@ -20,7 +20,6 @@ Unchanged: `CandidateLogicalASAPDAGs`, `cost_sorted`, `global_selection`, and th ```text PlanningWorkload ──lowering──▶ ParsedWorkload ──OptimizationPass──▶ PlanOutput + frontend deps + models - + lifecycle input ``` --- @@ -44,7 +43,7 @@ flowchart TD direction TB L["lowering"] O["OptimizationInput"] - PASS["OptimizationPass: MajorPass, or another implementation"] + PASS["OptimizationPass: StagePipeline, or another implementation"] L --> O --> PASS end @@ -63,9 +62,8 @@ Details of these types are provided below. |---|---| | `workload` | `&PlanningWorkload` | | `frontend_specific` | `Sql { catalog }` / `Promql { now_ms, histograms }` / `Metricsql`; fixed by `query_workload.language` | -| `models` | Cost model, accuracy model, evidence provider; `PlanningModels::builtin()` for the defaults | -| `lifecycle` | Planning clock and runtime capabilities for the maintenance-versus-recompute decision every plan carries | -| `pass` | `None` uses `MajorPass` | +| `models` | Cost model, accuracy model, evidence provider; `PlanningModels::builtin()` for the defaults. `StagePipeline` reads only the accuracy model (#580) | +| `pass` | `None` uses `StagePipeline` | ### `OptimizationInput` @@ -73,7 +71,6 @@ Details of these types are provided below. pub struct OptimizationInput<'a> { pub workload: &'a ParsedWorkload, pub models: PlanningModels<'a>, // same type UserInput uses - pub lifecycle: LifecycleInput, // same type UserInput uses } ``` @@ -83,44 +80,51 @@ pub struct OptimizationInput<'a> { ```rust pub struct PlanOutput { - pub plans: Vec, // one per workload entry, in entries() order + pub plans: Vec, // one per operator entry, in entries() order + pub scalar_roots: Vec<(usize, ScalarExpr)>, + pub selection: Option, // how the plan was chosen, if the pass says } -pub struct QueryLifecyclePlan { - pub entry_index: usize, // index into QueryWorkload::entries() - pub plan: SummaryMaintenanceLifecyclePlan, // its `root` is the DAG +pub struct QueryPlan { + pub entry_index: usize, // index into QueryWorkload::entries() + pub root: Rc, // selected post-ASAP DAG; shared nodes are the same Rc } ``` -Every plan carries the maintenance decisions, so the pass always runs -lifecycle-aware selection. A cost model that cannot price lifecycles -(`DefaultCostModel` today) makes that selection fall back to raw recompute for -every summary target; supply a model with the lifecycle cost hooks. +`StagePipeline` returns plans already timed at query time; for them +`PlanOutput::execution_timed_dag()` re-times nothing. Every summary runs at +query time until Stage 2 materialization (#509) decides per sub-DAG whether to +materialize and whether at ingestion or query time. --- ## 3. The pluggable optimization pass The optimization pass is fully pluggable, as long as the end-to-end behavior is satisfied. -The `MajorPass` described below will be used by default, which corresponds to the current optimization behavior of `ASAPPlanner`. +The `StagePipeline` described below is used by default. -### 3.1 `MajorPass` — the original optimization pass +### 3.1 `StagePipeline` — the #509 planner stages -`MajorPass` contains the original optimization algorithm the crate has always run, now behind the trait and registered under the name `major`. Its behaviour is unchanged: +`StagePipeline` runs the #509 stages and is registered under the name +`stage-pipeline`. It replaced `MajorPass`, the original replacement search +(#572); the regressions this accepted are tracked in #580. | Step | Call | |---|---| -| Build roots | `Id` is the entry's position in `entries()`; the accuracy target comes from its `requirements` | -| Candidate search | `search_workload_with_targets` with `default_strategies_with_evidence` | -| Select | `global_selection`, or `global_selection_with_summary_maintenance_lifecycles` with a `WorkloadDemand` derived from the `ParsedWorkload` | -| Assemble, per root | `assemble_selected_dag`, or its lifecycle-aware counterpart | - -Moving it behind the trait changes one thing for existing developers: -**`ReplacementStrategy` is now a concept of `MajorPass`, not of the optimization -stage.** Adding a rewrite or sharing rule to the shipped algorithm still means -implementing `ReplacementStrategy`. Replacing the algorithm means implementing -`OptimizationPass` instead — the two extension points no longer sit on top of -each other. +| Prepare roots | PromQL roots carry series identity (`with_promql_series_identity`); identical sub-DAGs merged (`share_common_sub_dags`) | +| Stage 1 | `enumerate_local_logical_candidates`: every target's local alternatives | +| Select | `plan_selection::select_plan`: a dynamic program over target nesting, priced by Stage 2 + Stage 3 | +| Build | `compose_logical_candidate`, identical producers merged, then `stage2_physical` | +| Check | Stage 3 accuracy check and price of the built plan | + +The dynamic program is exact when cost adds up per node and a target's choice +changes only its own nodes. `select_plan` checks the second for every target +and the target beneath it. When it fails, it builds every combination if +there are at most 64, and otherwise flags `Selection::method` as not +guaranteed optimal. + +`ReplacementStrategy` remains a concept of the legacy candidate search, which +the default pass no longer uses. ### 3.2 Plugging in another pass @@ -150,7 +154,7 @@ let output = e2e_plan(user_input.with_pass(&my_pass)).await?; // the whole pipe Or through an optimization pass registry: ```rust -let mut registry = PassRegistry::with_builtin(); // holds "major" +let mut registry = PassRegistry::with_builtin(); // holds "stage-pipeline" registry.register(Box::new(my_pass))?; for name in registry.names() { optimize(registry.get(name).unwrap(), optimization_input)?; @@ -159,98 +163,87 @@ for name in registry.names() { `PassRegistry` is caller-owned, not a link-time global, so two tests in one binary cannot see each other's registrations. -### 3.3 The three existing workflows, in this shape +### 3.3 The existing workflows, in this shape -[Input, output, and workflows](input-output-workflow.md) describes three ways to -use the candidate space. Only the last is what a pass produces; the other two -stay on the old interfaces. +[Input, output, and workflows](input-output-workflow.md) describes two ways to +use the candidate space. The second is what a pass produces; the first stays +on the old interfaces. | Workflow there | Here | |---|---| | Ranked view (`cost_sorted`) | Not covered by this design, you should handle it with old interfaces | -| Selection and DAG assembly | Not covered either: `search_workload_with_targets` + `global_selection` + `assemble_selected_dag` | -| Summary-maintenance-lifecycle-aware helper | `PlanOutput` | +| Selection and DAG assembly | `PlanOutput` | -The third is no longer a call sequence the caller drives. +Selection and assembly are no longer a call sequence the caller drives. Following is an example of how the old workflow maps to the new interface. ```rust -// Before — from a PlanningWorkload and a catalog, with lifecycle decisions. +// Before — from a PlanningWorkload and a catalog. // 1. Lower every normalized entry, and record which entry each root came from. // Not lower_sql_batch: it walks query_batch alone and drops repeating entries. let mut roots = Vec::new(); -let mut entry_indices = Vec::new(); for (index, entry) in workload.query_workload.entries().enumerate() { let accuracy = entry.requirements.accuracy.target(); let expr = lower_sql_dialect(&entry.query.0, &catalog, dialect.clone(), accuracy.clone()) .await?; - roots.push((index, Rc::new(expr), Some(accuracy))); - entry_indices.push(index); + roots.push((index, expr, Some(accuracy))); } // 2. Search for candidates. let strategies = default_strategies_with_evidence(&cost_model, &evidence); let space = search_workload_with_targets(roots, &strategies, &accuracy_model); -// 3. Select once for the whole workload, re-binding roots to workload entries. -let demand = WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &entry_indices, -}; -let selection = global_selection_with_summary_maintenance_lifecycles( - &space, demand, now_ms, horizon, capabilities, &cost_model)?; - -// 4. Assemble once per root. -let mut plans = Vec::new(); +// 3. Select once for the whole workload. +let selection = global_selection(&space, &cost_model); + +// 4. Assemble once per root, then share common sub-DAGs across roots. +let mut assembled = Vec::new(); for (index, root) in &space.roots { - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, root, demand, now_ms, horizon, capabilities, &cost_model)?; - plans.push((*index, plan)); + if let Some(dag) = selection.assemble_selected_dag(root)? { + assembled.push((*index, dag)); + } } +let plans = share_common_sub_dags(assembled); ``` ```rust // After. let output = e2e_plan( UserInput::new(&workload, FrontendInput::Sql { catalog: &catalog }, - PlanningModels::builtin(), - LifecycleInput::new(now_ms, capabilities).with_horizon(horizon)) + PlanningModels::builtin()) ).await?; ``` -Steps 1 and 3 are where the two bindings lived: the `Id` carried through the -roots tuple, and the `&[usize]` rebuilt for `WorkloadDemand`. Both had to agree -with `entries()` order, and nothing checked that they did. `MajorPass` still -runs all four steps; another pass need not run any of them. +Step 1 is where the binding lived: the `Id` carried through the roots tuple had +to agree with `entries()` order, and nothing checked that it did. The default +pass no longer runs these steps; another pass need not run any of them. ## 4. Code layout | Crate | What it holds | |---|---| | `asap-types` | `ParsedWorkload` | -| `asap-aware-mapping` | `OptimizationPass`, `OptimizationInput`, `PlanOutput`, `PlanningModels`, `LifecycleInput`, `optimize`, `PassRegistry`, `MajorPass` | -| `asap-planner` *(new)* | `e2e_plan`, `UserInput`, `FrontendInput`, lowering dispatch | +| `asap-plan-selection` | `PlanningModels` | +| `asap-planner` | `e2e_plan`, `UserInput`, `FrontendInput`, lowering dispatch; `OptimizationPass`, `OptimizationInput`, `PlanOutput`, `optimize`, `PassRegistry`, `StagePipeline` | ```text asap-planner ──┬──> asap-frontend-{sql, promql, metricsql} - └──> asap-aware-mapping ──> asap-types - ▲ - a pass depends only this far + └──> asap-plan-selection ──> asap-physical-optimizer ──> asap-logical-optimizer ──> asap-types ``` `asap-planner` is separate because it is the only crate depending on every frontend; before it, the sole facade re-exporting more than one was -`asap-devtools`, a developer-tools crate. `PlanningModels` and `LifecycleInput` -live in `asap-aware-mapping` because both inputs use them, and `asap-planner` -re-exports them. +`asap-devtools`, a developer-tools crate. `PlanningModels` lives in +`asap-plan-selection` because both inputs use it, and `asap-planner` +re-exports it. Since #572 the pass lives in `asap-planner`, so a pass +implementation depends on the frontends too. --- ## Related * [ASAPPlanner input, output, and workflows](input-output-workflow.md) -* [Searching over plans](asap-aware-plan-search.md) — what `MajorPass` does inside +* [Searching over plans](asap-aware-plan-search.md) — the legacy candidate search * [Planner/runtime responsibilities](planner-runtime-contract.md) * [Public library reference](../../develop_docs/library-api.md) diff --git a/docs/design_docs/concepts/accuracy-models.md b/docs/design_docs/concepts/accuracy-models.md index 5353ef9ef..39e7c468b 100644 --- a/docs/design_docs/concepts/accuracy-models.md +++ b/docs/design_docs/concepts/accuracy-models.md @@ -69,7 +69,7 @@ query text or cost estimates. flowchart TD Request[Query semantics and accuracy target] --> Generate[Generate candidates and size parameters] Evidence[Scoped source contracts and evidence] --> Generate - Generate --> Local[Derive local readout guarantees] + Generate --> Local[Derive local evaluation guarantees] Evidence --> Local Local --> Compose[Propagate guarantees through the DAG] Evidence --> Compose @@ -100,7 +100,7 @@ is ready, or that a complete deployment cost is available. ## Local estimator models and parameter sizing -A local model describes a specific readout of a specific estimator with +A local model describes a specific evaluation of a specific estimator with committed parameters and applicable assumptions. A family name or a parameter such as HLL precision is not, by itself, a confidence certificate. @@ -121,8 +121,8 @@ The built-in models currently include: | CMS | L1-normalized frequency bound from width and depth; does not by itself certify TopK membership | | CountSketch | L2-normalized frequency bound and median concentration bound, requiring valid odd depth | | KMV / Theta | Parameter-derived cardinality bounds using the registered variance/Chebyshev model at 99% confidence | -| UnivMon | Exact unit-update total for the supported readout; no universal guarantee for all its statistics | -| Other families/readouts | No default certificate where no accuracy model is registered | +| UnivMon | Exact unit-update total for the supported evaluation; no universal guarantee for all its statistics | +| Other families/evaluations | No default certificate where no accuracy model is registered | This table describes Planner's registered contracts, not independent mathematical verification of every estimator or permission to substitute @@ -214,7 +214,7 @@ an observation into a guarantee. A deployment supplies `EstimatorContract::ClassicHll` for the complete aggregate expression. It asserts the classic estimator, independent uniform bucket -hashing and an enforced maximum distinct population per readout, including +hashing and an enforced maximum distinct population per evaluation, including all merged panes. Planner combines this contract with the query or allocated local target, selects a supported precision, derives the guarantee and uses the normal propagation and selection checks. @@ -280,7 +280,7 @@ rule or a different estimator configuration would be needed in those cases. ## Organization and extension contract -The `asap-aware-mapping::accuracy` module separates these responsibilities: +The `asap-logical-optimizer::accuracy` module separates these responsibilities: ```text accuracy/ @@ -290,7 +290,7 @@ accuracy/ ├── allocation.rs # End-to-end budget allocation ├── reconciliation.rs # Accuracy coordination across consumers └── estimators/ - ├── mod.rs # Family/readout dispatch and source-contract integration + ├── mod.rs # Family/evaluation dispatch and source-contract integration ├── kll.rs ├── ddsketch.rs ├── hll.rs # Generic HLL and bounded Classic HLL @@ -308,7 +308,7 @@ share the same contract. Adding an estimator or composition requires: -1. A precisely defined error metric, estimator/readout semantics and assumptions. +1. A precisely defined error metric, estimator/evaluation semantics and assumptions. 2. Sizing behavior and a guarantee derived from the committed parameters, including unsupported parameter domains. 3. Explicit evidence requirements, population scope and provenance. 4. Propagation rules where supported; rejection or retained unknowns elsewhere. @@ -319,6 +319,6 @@ not require a new selection rule for every sketch. Deployment extensions to `AccuracyModel` remain possible, but carry the same obligation to justify metrics, assumptions and propagation. -For implementation details, see the [accuracy module](../../../crates/asap-aware-mapping/src/accuracy/mod.rs), -[guarantee representation](../../../crates/types/src/post_asap/guarantee.rs), and +For implementation details, see the [accuracy module](../../../crates/logical-optimizer/src/accuracy/mod.rs), +[guarantee representation](../../../crates/types/src/ir/properties/guarantee.rs), and [accuracy propagation companion](../../develop_docs/end-to-end-accuracy-guarantees.md). diff --git a/docs/design_docs/concepts/planner-pipeline.md b/docs/design_docs/concepts/planner-pipeline.md index 7401ae0da..c07d17775 100644 --- a/docs/design_docs/concepts/planner-pipeline.md +++ b/docs/design_docs/concepts/planner-pipeline.md @@ -14,12 +14,11 @@ over that output, not mandatory stages of candidate search. | +--> inspect candidates, optionally using cost_sorted +--> select and assemble logical DAGs - +--> select and assemble with summary-maintenance lifecycle decisions -The last two branches are alternatives: use the summary-maintenance-lifecycle-aware -workflow when Planner owns maintenance-versus-recomputation decisions; otherwise -the backend owns them. All physical binding, deployment, and execution remain -downstream responsibilities. +Stage 2 materialization (#509) will decide per sub-DAG whether to materialize +and whether at ingestion or query time; until then every summary runs at query +time. All physical binding, deployment, and execution remain downstream +responsibilities. The [input, output, and workflows](../architecture/input-output-workflow.md) document defines the public boundary and helper call order. diff --git a/docs/design_docs/concepts/post-asap-ir.md b/docs/design_docs/concepts/post-asap-ir.md index 6c9aa1461..67a5ed406 100644 --- a/docs/design_docs/concepts/post-asap-ir.md +++ b/docs/design_docs/concepts/post-asap-ir.md @@ -1,22 +1,53 @@ # Post-ASAP IR The goal of the post-ASAP IR is to represent operations using ASAP primitives -such as sketches, exact summaries, samples and wavelets. Post-ASAP IR also -retains exact Pre-ASAP sub-DAGs and supports operations over summary readouts, -since only some query operations can be satisfied using summaries. - -The lists below cover every current variant of -[`SummaryExpr`](../../../crates/types/src/post_asap/expr.rs). A node's presence -in the IR does not imply that every summary family, cost model or downstream -runtime supports it. - -## ASAP-specific nodes operated over a summary structure, not raw data - -- `SummaryAgg`: produce summary state from input data using the selected family, - parameters, update input, reduction and grouping layout. -- `SummaryEstimate`: read the requested statistic from summary state and return - query values. Exact accumulators can expose results without a separate sketch - readout. +such as sketches, exact summaries, samples and wavelets, while retaining the +exact query operators that no summary replaces, and supporting operations over +summary evaluations. + +ASAPPlanner has one operator IR before and after ASAP optimization +([`crates/types/src/ir/`](../../../crates/types/src/ir/)). A post-ASAP plan is +the same `Rc` DAG a front end produced, in which some nodes now +carry `Operator::ASAP(ASAPOp)` instead of `Operator::NonASAP(NonASAPOp)`. There +is no wrapper around retained exact work: an unreplaced `Filter`, `Join` or +`Aggregate` is the same node it was before, and either category can consume +the other's output. The node structure, the `Schema`, scalar expressions and +the catalog of non-ASAP operators are described once in the +[Pre-ASAP IR reference](../../develop_docs/pre-asap-ir.md); this document covers +what optimization adds: the ASAP operators, the accuracy guarantee, execution +timing, and the exported DAG. + +A node's presence in the IR does not imply that every summary family, cost +model or downstream runtime supports it. + +## ASAP operators + +Every variant of [`ASAPOp`](../../../crates/types/src/ir/operator/asap.rs) operates +over summary state rather than raw data. The summary family, kind/algorithm and +parameters are committed in the node; the state itself is typed by the +`FieldDataType` of the output field that carries it (`ExactAggregate`, +`Sketch`, `Sample`, `Wavelet`, `StatModel`). + +Implemented: + +- `SummaryAgg { child, family, input, reduction, grouping }`: produce summary + state from input rows using the selected family, parameters, update input, + reduction and grouping layout. Output: the grouping columns plus one `state` + field typed `family`; result kind `State`. +- `SummaryEstimate { summary_input, query }`: read the requested statistic + (`SketchStatistic`) from summary state and return query values in a row-shaped + schema. +- `FinalizeExactAccumulator { child }`: read an exact accumulator's state as + its finalized value — the maintenance-to-read boundary before query-time + operators consume it. +- `MaintainPopulation { child, population }`: maintain the full declared + population, including membership changes. +- `EvaluatePopulation { child, evaluation }`: read an aggregate or TopK prefix from a + maintained population. + +Reserved (migrated but unimplemented; schema derivation, timing and export +reject them with `UNIMPLEMENTED_ASAP_OP`): + - `SummaryMerge`: merge compatible summary states when the family supports merging. - `SummarySubtract`: subtract one summary state from another when supported by the selected representation. @@ -24,62 +55,114 @@ runtime supports it. deletion. - `SummaryJoin`: combine summary states for join estimation; this is distinct from joining ordinary rows. +- `Extension`: a deployment-defined operator. The earlier draft listed `SummaryCreate` and `SummaryInsert`. These are not -separate variants in the current IR. `SummaryAgg` describes the state-producing -computation and its update input. The -[summary-maintenance lifecycle](../proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md) -separately describes when state is created, retained, shared, updated and retired. -Physical binding and runtime execution implement the actual build and update -operations. This is not a one-to-one rename of the old nodes, and not every -summary family supports incremental maintenance. - -## Exact work and composition nodes - -- `KeepPreAsap`: retain an exact Pre-ASAP sub-DAG when it is not rewritten. -- `BinaryOp`: combine independently planned operands with the specified binary - semantics and execution timing. -- `ValueOperation`: apply aggregate, exact-function, population, projection, - filter, sort, limit or extension semantics with explicit execution timing. -- `RelationalJoin`: join row-producing children using the specified join kind - and predicate. -- Candidate pruning uses `RelationalJoin` with `JoinKind::Semi` and an explicit +separate variants. `SummaryAgg` describes the state-producing +computation and its update input. Stage 2 materialization (#509) will decide +whether and when that state is maintained. Physical binding and runtime +execution implement the actual build and update operations. Not every summary family supports incremental maintenance. + +## Exact work and composition + +Exact work is represented by the ordinary operators, unchanged: + +- A sub-DAG the planner does not rewrite keeps its `NonASAPOp` nodes. Plan + assembly marks such a sub-DAG with an exact `ResultGuarantee` + (`asap_logical_optimizer::pass1::replacement::retain_exact`); a sub-DAG with no ASAP + operator and no guarantee is a logical rewrite candidate that has not been + assessed yet (`is_logical_rewrite`). +- `BinaryOp` combines independently planned operands. Summary planning may set + its typed division guards (`checked_finite_division`, + `checked_relative_division`); the operator's timing comes from the + materialization assignment, not from the operator. +- Aggregate, projection, filter, sort and limit over a evaluation are the ordinary + `Aggregate`, `Project`, `Filter`, `Sort` and `Limit` operators reading an ASAP + node. Exact-accumulator state may pass through the projection-like + operators unchanged; a value consumer needs a `FinalizeExactAccumulator` + boundary first. +- Candidate pruning uses `Join` with `JoinKind::Semi` and an explicit equality predicate on key columns. The left input supplies authoritative - values; the right input supplies keys. Grouped Sort followed by grouped Limit ranks - and selects the joined rows. Completeness evidence belongs to pruning, not ranking. + values; the right input supplies keys. Grouped `Sort` followed by grouped + `Limit` (both with the same `partition_by`) ranks and selects the joined + rows. Completeness evidence belongs to pruning, not ranking. -A `SummaryNode` carries its expression, schema and optional result guarantee. +Every `OperatorNode` carries its schema and an optional `ResultGuarantee`. State and query values have different contracts. Exact operations over -approximate readouts still require composed accuracy guarantees. See the +approximate evaluations still require composed accuracy guarantees. See the [accuracy implementation companion](../../develop_docs/end-to-end-accuracy-guarantees.md) and [physical-plan integration](../architecture/physical-plan-integration.md) for the corresponding correctness and realization requirements. -## In-memory and exported DAG forms - -The Pre-ASAP DAG and the Post-ASAP DAG are both logical: they describe what is -computed, not which physical operators execute it. The Post-ASAP DAG has two -forms of the same content. Planning builds and shares `SummaryNode` DAGs. -`compile_post_asap_dag` converts a selected DAG into a -[`PostAsapDAG`](../../../crates/types/src/post_asap/post_asap_dag.rs) with -stable node IDs and typed edges; `PostAsapDAGDocument` is its versioned wire -envelope. Physical compilation consumes `PostAsapDAG` and produces a separate -physical DAG. - -## Execution phase +## Execution timing An operator defines what computation happens. The plan decides when it happens: **ingestion time** or **query time**. Operator identity must not imply one of these phases. Backend capability restrictions are implementation gaps, not definitions of the operator. -Every post-ASAP operator payload supports both phase assignments. Phase is -stored on the `PostAsapDAG` node, independently of its operator payload. -`PostAsapDAG::with_execution_phases` assigns a phase to every node and updates -its edges. Ingestion work cannot depend on a future query result. Default -semantic realization still proposes an initial layout; it does not restrict -which phase an operator may use. Deployments must separately check that -they have an implementation and a valid data source for the chosen placement. +The logical DAG carries no timing: `OperatorNode::timing` is `None` on every +front-end node and every candidate, and `map_children` clears it. Summary +materialization chooses a timing per summary state and records it in a +[`MaterializationAssignment`](../../../crates/types/src/ir/properties/timing.rs) (ingestion-time +maintenance or query-time computation per `SummaryAgg`). The default is +`all_query_time()`; until Stage 2 materialization (#509) decides otherwise, the +planner times every `SummaryAgg` at query time. +`apply_materialization_timings(root, &assignment, &mut TimingMemo)` then writes a +timing into every node, top-down: + +- a node of fixed kind takes its kind's timing — `SummaryEstimate` and + `EvaluatePopulation` run at query time, `MaintainPopulation` at ingestion time; +- a `SummaryAgg` takes the assignment's timing, unless something below it can + only exist at query time (a evaluation); +- every other node runs when its consumer runs: everything that feeds a + maintained state runs at ingestion time, everything above a evaluation at + query time. + +The pass then validates every edge (rows or exact-accumulator state into a +`SummaryAgg`, state into a evaluation, an ingestion-time `MaintainPopulation` under +a `EvaluatePopulation`, no ingestion work reading a query-time value) and rejects a +node reached from two consumers that need different timings; +`split_shared_by_phase` copies such a sub-DAG for one side before the +assignment is applied. `validate_maintained` and `planned_data_state` answer the +same questions for a candidate at planning time, assuming every summary is +maintained at ingestion time, without keeping anything. + +## Exported DAG + +The pre-ASAP DAG and the post-ASAP DAG are both logical: they describe what is +computed, not which physical operators execute it. Planning builds and shares +`OperatorNode` trees; +[`asap_types::ir::export::compile_post_asap_dag`](../../../crates/types/src/ir/export.rs) +converts a selected, timed tree into a `PostAsapDAG` with stable node IDs and +typed edges, and `PostAsapDAGDocument` is its versioned wire envelope +(`schema_version` = `POST_ASAP_DAG_WIRE_VERSION`, currently 6). Physical +compilation consumes `PostAsapDAG` and produces a separate physical DAG. + +Wire version 7 emits **one node per operator** — relational operators +included — with children as edges and no embedded sub-DAGs: + +- A non-ASAP node is a `Relational { operator: NonASAPOpKind }` payload: + the operator's own fields with scalar expressions mirrored as + `WireScalarExpr`, children removed. An ASAP node's payload is its variant + (`SummaryAgg`, `SummaryEstimate`, `FinalizeExactAccumulator`, + `MaintainPopulation`, `EvaluatePopulation`, …). +- Edges carry a role: `Input`, `Left`/`Right` for the two sides of a `Join`, + `SetOp`, `BinaryOp`, `SummarySubtract` or `SummaryJoin`, and `ScalarRef` + when the consumer reads the producer from inside one of its scalar + expressions (`scalar(v)`). Every edge records the intermediate schema, the + producer's data state and grouping/window compatibility. +- Each node records `output_state` (timing plus `Raw` or `SummaryState`), + `output_schema` and `guarantee`. Export reads the timing written by + `apply_materialization_timings` and rejects an untimed node + (`ExecutionDataStateError::UntimedNode`); it does not re-run data-state + validation. + +Phase is stored on the `PostAsapDAG` node, independently of its payload. +`PostAsapDAG::with_execution_phases` reassigns a phase to every node and +updates its edges; ingestion work cannot depend on a future query result. +Deployments must separately check that they have an implementation and a +valid data source for the chosen placement. ## Weighted grouped TopK @@ -90,19 +173,19 @@ the update weight is the series rate. Summing updates for one item implements the logical grouped sum without first constructing all exact grouped sums. The DAG is per-series rate → finalized values → partitioned summary construction -→ typed candidate/score readout → output projection → grouped Sort → grouped +→ typed candidate/score evaluation → output projection → grouped Sort → grouped Limit. The output count is two per job. The candidate capacity is a separate parameter, provisionally `max(k, ceil(1 / epsilon))`; this sizing choice is not a membership theorem. Missing evidence retains a logical candidate with symbolic unknown guarantees; default selection does not certify or choose it. -The row readout restores job and service identities and returns estimated sums. +The row evaluation restores job and service identities and returns estimated sums. There is no mandatory exact scoring branch or candidate semi-join in this path. The old raw counter-delta update expression is removed rather than retained as a compatibility option: counter increments are not complete windowed rate results. -The direct readout represents both score error and membership. A source provider +The direct evaluation represents both score error and membership. A source provider supplies an enforced upper bound on distinct partition/item identities for the -complete readout. Planner uses this bound to size confidence and union-bound +complete evaluation. Planner uses this bound to size confidence and union-bound score errors over adaptively selected items. Membership evidence is evaluated for the query's output count, not the candidate capacity. Score and membership failure probabilities are combined, and the score guarantee remains in the @@ -110,8 +193,8 @@ membership guarantee's child provenance. An exact request does not accept this approximate output path merely because its selected identities are certified. Deployment chooses ingestion time or query time for these operators. The -semantic constructor proposes a layout; `with_execution_phases` assigns the -placement. Either deployment must give each evaluation a complete +materialization assignment writes the placement; `with_execution_phases` can +reassign it on the exported DAG. Either deployment must give each evaluation a complete rate window and an isolated summary state, or maintain an equivalent replacement strategy. Appending successive rate snapshots to one cumulative state is invalid. An ingestion execution can compute a window before the query and store its state; diff --git a/docs/design_docs/concepts/pre-asap-ir.md b/docs/design_docs/concepts/pre-asap-ir.md index 735f522f3..0b12766b1 100644 --- a/docs/design_docs/concepts/pre-asap-ir.md +++ b/docs/design_docs/concepts/pre-asap-ir.md @@ -12,16 +12,18 @@ Only semantics that affect correctness, summary applicability, or cost become fi ### Time -- TimeRange — a PromQL range-vector lookback such as [5m]. +- TimeRange — PromQL sample selection: an instant selector's lookback, or a range selector such as [5m]. - TimeShift — moves when a selector is evaluated (offset or @). - PromqlSubquery — re-evaluates an instant-vector expression over a range. ### Relational - Scan — identifies a logical data source. +- Values — literal rows; one empty row is the input of a `SELECT` without `FROM`. +- ScalarBridge — a scalar expression at an operator position: a bare scalar query, or the scalar operand of ` op `. - Filter — restricts rows using a predicate. - Project — selects or derives output columns. -- BinaryOp — composes two inputs with arithmetic, comparison, or boolean logic. +- BinaryOp — composes two inputs with arithmetic, comparison, or boolean logic. A PromQL `bool` comparison returns 0/1 instead of filtering. - Sort — orders rows without expressing a heavy-hitter intent. - Limit — caps a row count, optionally after an offset. - Dedup — removes duplicate rows. @@ -31,10 +33,7 @@ Only semantics that affect correctness, summary applicability, or cost become fi ### PromQL-specific -- PromqlScalarBridge — holds a scalar sub-expression at an operator-DAG position. -- EvalTimestamp — provides the evaluation timestamp as a scalar. -- PromqlVectorFromScalar — promotes a scalar to a label-less instant vector. -- PromqlScalarFromVector — collapses a single-series vector to a scalar. +- PromqlVectorFromScalar — promotes a scalar to a label-less instant vector. Its inverse, PromQL `scalar(v)`, is a scalar expression that reads `v`. - PromqlRelabel — rewrites labels on each series. - PromqlInfoEnrich — enriches labels from an info metric. - PromqlSeriesSample — selects whole series without reducing them. diff --git a/docs/design_docs/decisions/concat-unique-keys.md b/docs/design_docs/decisions/concat-unique-keys.md index 9c0148abf..7d603e08f 100644 --- a/docs/design_docs/decisions/concat-unique-keys.md +++ b/docs/design_docs/decisions/concat-unique-keys.md @@ -67,9 +67,9 @@ Both current `Concat`-constructing call sites, and every consumer of test fixtures) — none of them builds a fresh `Concat` with a `Dedup` on top that this feature could remove. - **Every consumer of `Schema::unique_keys`** in the DAG, to check for a - cost beyond "a literal `Dedup` node": `pre_asap::cse::share_common_sub_dags` + cost beyond "a literal `Dedup` node": `ir::cse::share_common_sub_dags` (gates CSE producer-sharing on `Schema::has_unique_key()`) and - `asap_aware_mapping::rollup::is_legal_rollup_source` (gates rollup-source + `asap_logical_optimizer::pass1::rollup::is_legal_rollup_source` (gates rollup-source legality the same way, on an *`Aggregate`'s* own output schema). Neither case is exercised by a `histogram_quantiles` or `ROLLUP`/`CUBE`/ `GROUPING SETS` `Concat` in any current test, workload, or call site: no @@ -129,7 +129,7 @@ for why it's fine to ship unused. caller upstream of `resolve_root`, even though no such caller exists yet. - Every other match/construction site touching `Concat` across the DAG (`canonicalize.rs`, `cse.rs`, `schema_resolver.rs`, `dag_export.rs`, - `asap-aware-mapping`'s `replacement.rs`/`explanation.rs`, and every + `asap-logical-optimizer`'s `replacement.rs`/`explanation.rs`, and every test/tooling AST walker) was mechanically updated to bind or ignore the new field — most just added `, ..`; the two places that *rebuild* a `Concat` node (`cse.rs`'s `rebuild_children`, part of CSE interning) thread diff --git a/docs/design_docs/decisions/cse-cost-model.md b/docs/design_docs/decisions/cse-cost-model.md index 7ada47bc5..e4f9423a1 100644 --- a/docs/design_docs/decisions/cse-cost-model.md +++ b/docs/design_docs/decisions/cse-cost-model.md @@ -10,7 +10,7 @@ elimination) to identify eligible, structurally identical computations. A shared logical node records an opportunity for reuse; selecting maintained state or independent execution is a separate planning decision. -[`asap_types::pre_asap::cse::share_common_sub_dags`](../../../crates/types/src/pre_asap/cse.rs) +[`asap_types::ir::cse::share_common_sub_dags`](../../../crates/types/src/ir/cse.rs) (issue #223 stages 1-2, PR #235) already *detects* every structurally-identical, legally-shareable (`Schema::unique_keys`-gated) sub-DAG and shares it **unconditionally** — there is no cost gate on top of legality. This document @@ -28,7 +28,7 @@ decides the framework for stage 4, "wire workload-level CSE credit into ## Decision: cost-based (Volcano/Cascades), implemented for real This lands as an actual cost comparison, not a documented-but-unimplemented -shape. [`CostModel::cse_share_decision`](../../../crates/asap-aware-mapping/src/cost_model.rs) +shape. [`CostModel::cse_share_decision`](../../../crates/plan-selection/src/cost/cost_model.rs) compares two real, overridable cost estimates for every CSE candidate with two or more consumers: @@ -58,7 +58,7 @@ This decision does not need search infrastructure of its own. Issue #252's MEMO-based search engine (`CandidateLogicalASAPDAGs`/`TargetSubDAGCandidates` in `replacement.rs`) already enumerates and ranks the larger, workload-wide candidate space. The choice between sharing and recomputing one already-detected CSE candidate is binary, -so `CandidateLogicalASAPDAGs::cost_sorted` reuses one direct +so `candidate_selection::cost_sorted` reuses one direct `CostModel::cse_share_decision` comparison per group. This preserves the policy described here—compare costs rather than applying a fixed rule—inside the larger search engine. `search_workload_with`'s @@ -70,15 +70,15 @@ same way a real cost-based optimizer would. ## Layering constraint `share_common_sub_dags` lives in `asap-types::pre_asap` — a lower layer that -`asap-aware-mapping` (which owns `CostModel`) depends on, never the reverse. +`asap-plan-selection` (which owns `CostModel`) depends on, never the reverse. Detection therefore cannot consult cost even if it wanted to. This is why stage 1/2's detection stays unconditional (correctly, as a legality-only gate) and the cost-aware decision is applied downstream, in -`asap-aware-mapping`, after detection rather than fused into it. +`asap-plan-selection`, after detection rather than fused into it. ## Where it hooks in -[`CandidateLogicalASAPDAGs::cost_sorted`](../../../crates/asap-aware-mapping/src/replacement.rs) +[`candidate_selection::cost_sorted`](../../../crates/plan-selection/src/candidate_selection.rs) is where this hooks in today. `search_workload_with` computes each shared sub-DAG's true `consumer_count` across the whole workload up front (the same role `implement_workload_with`'s pre-pass used to play, before that function @@ -86,7 +86,7 @@ was retired along with `bind.rs` — this crate no longer commits to one physically-materialized answer at all; picking and building one final `SummaryNode` per shared sub-DAG is a downstream deployment's job, not this crate's). For a `TargetSubDAGCandidates` whose candidates are a -[`SharedSubDAGStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) +[`SharedSubDAGStrategy`](../../../crates/logical-optimizer/src/pass1/replacement.rs) share-vs-recompute pair, `cost_sorted`'s ranking step (`rank_group`/ `cse_preference`) asks `CostModel::cse_share_decision` once per group — using one representative bound `SummaryNode` built just for that comparison, not @@ -116,7 +116,7 @@ either or both, same as `size_params` already lets a deployment override ## Scope -This decision, and `cse_share_decision`'s wiring into `CandidateLogicalASAPDAGs::cost_sorted` +This decision, and `cse_share_decision`'s wiring into `candidate_selection::cost_sorted` (originally into `implement_workload_with`, before `bind.rs` was retired — see above), close out #223's stage 4 and #212's original "add CSE" tracking issue. Stage 3 (`dag_export::structural_hash` unification) landed separately diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index 274e4974c..3f448f80c 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -7,11 +7,11 @@ A Post-ASAP computation is progressively realized through four layers: ```mermaid flowchart LR L["Logical Post-ASAP DAG
What computation?"] - M["Summary Maintenance Lifecycle
How is state maintained?"] + M["Materialization
How is state maintained?"] P["Physical DAG(s)
How is it executed?"] D["Deployment Plan / DAG
How is it instantiated?"] - L -->|"Summary Maintenance
Candidate Generation"| M + L -->|"Stage 2
Materialization (#509)"| M M -->|"Physical Plan
Compiler"| P P -->|"Deployment Plan
Compiler"| D ``` @@ -19,47 +19,44 @@ flowchart LR | Layer | Defines | | --- | --- | | **Logical Post-ASAP DAG** | Computation semantics | -| **Summary Maintenance Lifecycle** | Build, retention, reuse, and window strategy | +| **Materialization** | Build, retention, reuse, and window strategy | | **Physical DAG(s)** | Supported physical candidates, executable operators and typed input boundaries | | **Deployment Plan / DAG** | Selected candidate, concrete data/state bindings and operational lifecycle | ASAPPlanner owns the first three layers and the shared physical operator implementation library. Deployment systems such as ASAPQuery and asap-fusion -own deployment compilation and operation. The lifecycle is a planning contract +own deployment compilation and operation. Materialization is a planning contract associated with the logical DAG, not a separate computation IR. -The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the +> **Status:** Stage 2 materialization (#509) will decide per sub-DAG whether to +> materialize and whether at ingestion or query time. It is not implemented yet; +> until then the planner times every summary at query time. Sections 2 and 3 +> describe the intended contract. + +The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG, the language-independent query semantics before summary selection. Both are -logical. Planning builds Post-ASAP `SummaryNode` DAGs; `compile_post_asap_dag` +logical `OperatorNode` DAGs; `compile_post_asap_dag` exports the selected DAG as a `PostAsapDAG`, which is the Physical Plan Compiler's input. Its per-node execution phase (ingestion or query time) is -decided by the selected summary maintenance lifecycle, as the layer contract -below states. +decided by a `MaterializationAssignment`, as the layer contract below states. ### Layer contract 1. **Logical Post-ASAP** (`CandidateLogicalASAPDAGs`) decides what to compute: summary families, readouts and sharing. It does not decide placement; timing that a realization strategy writes while building a candidate is provisional. -2. **Summary maintenance lifecycle** (Planner) lists the lifecycle choices for - each unique retained state: every summary state (`SummaryAgg`) and every - maintained population that does not feed a summary state. - A chosen assignment determines every node's - `ExecutionTiming`, plus window framework and retention. - `SummaryMaintenanceLifecyclePlan::execution_timed_dag` applies it: a retained - (non-`Ephemeral`) state and all of its inputs run at ingestion time; - readouts, other consumers, and `Ephemeral` states not consumed by retained - state run at query time. A population that feeds a summary state is one of - that state's inputs and follows its timing. +2. **Materialization** (Stage 2, #509) chooses ingestion or query time for each + summary state (`SummaryAgg`) and records it in a `MaterializationAssignment`. + `apply_materialization_timings` writes every node's `ExecutionTiming`: an + ingestion-time state and all of its inputs run at ingestion time; readouts, + other consumers, and query-time states run at query time. + `MaintainPopulation` always runs at ingestion time. The default assignment + is all query time, which `PlanOutput::execution_timed_dag` applies. 3. **Physical compile** (Planner) reads timing: ingestion-time nodes form the precompute DAG and the rest form the query DAG, joined by typed outputs. It does not see raw ingestion, panes, storage or stored-state readout. -4. **Backend** chooses the lifecycle assignment with its own `CostModel`: - precompute CPU (`maintenance_cost_per_update`), sketch/summary store cost - (`retention_cost_rate`), query reads (`summary_read_cost`) and per-query - builds (`build_cost`, for `Ephemeral`), counting shared state once. - `Ephemeral` requires the deployment to supply the state's raw input as a - query-time source. +4. **Backend** binds and executes the timed DAG. A query-time summary requires + the deployment to supply the state's raw input as a query-time source. ### Candidate generation and deployment selection @@ -69,7 +66,7 @@ because a deployment-independent cost estimate prefers another candidate. Logical candidates are an internal search stage, not the deployment handoff. ```text -Query semantics + accuracy and lifecycle requirements +Query semantics + accuracy and freshness requirements ↓ Planner Supported Physical DAG candidates + typed inputs/outputs + requirements ↓ backend @@ -93,9 +90,9 @@ and a feasible candidate that loses on cost. Absence is not a cost comparison. For `sum by(job)(rate(m[1m]))`, Rate remains per series before grouped Sum. `CandidateLogicalASAPDAGs` offers one such candidate, with a per-series Rate state and a grouped -Sum state. Its lifecycle assignment places it: a retained Sum state finalizes -Rate and builds Sum within a bounded precompute run; an `Ephemeral` Sum over a -retained Rate state leaves the Rate readout and Sum in the query DAG. Storing a +Sum state. Its materialization assignment places it: an ingestion-time Sum state +finalizes Rate and builds Sum within a bounded precompute run; a query-time Sum +over an ingestion-time Rate state leaves the Rate readout and Sum in the query DAG. Storing a value requires its exact evaluation window, revision, readiness and serving cadence to match the query contract. @@ -109,7 +106,7 @@ Planner's candidate space decides what to compute, not placement. For an instant-vector PromQL TopK, Planner resolves rows that carry the complete series identity and lists the current-series heap realizations per root with the other candidates, unranked. Precompute or query-time placement of Rate and grouped Sum -is not a separate Planner candidate: the summary maintenance lifecycle assigns +is not a separate Planner candidate: the materialization assignment sets each node's timing, and the physical compiler reads it. This is the target ownership contract. A backend path that still reconstructs @@ -182,10 +179,10 @@ KLLMerge p50 p99 │ - │ Summary Maintenance Candidate Generation + │ Stage 2 Materialization (#509) ▼ -2. Summary Maintenance Lifecycle +2. Materialization KLLBuild(k=200) strategy = continuously maintain @@ -239,7 +236,7 @@ Each stage adds a different class of decision while preserving the preceding contracts. Here, continuous maintenance means recurring production of pane state; the bounded build DAG does not itself implement an unbounded streaming window. -## 2. Logical Post-ASAP DAG → Summary Maintenance Lifecycle +## 2. Logical Post-ASAP DAG → Materialization The **Logical Post-ASAP DAG** defines computation semantics: @@ -257,33 +254,13 @@ Quantile(.5) Quantile(.99) It establishes that KLL with `k=200` is used and that the merge is shared by the two readouts. It does not determine when KLL states are built or retained. -**Summary Maintenance Candidate Generation** enumerates legal lifecycle choices -using workload demand, window/freshness requirements and supported physical -implementations. Backend selection uses runtime feasibility and cost after -physical compilation. The following example follows one candidate. - -Candidate generation and selection are separate steps. For every unique retained -state, enumeration reports each lifecycle (ephemeral, prepared, shared, -continuously maintained) as legal, with a Planner cost or explicitly unknown -cost, or as rejected with a reason. Planner does not remove a legal alternative -because its own estimate prefers another. A deployment prices the legal -alternatives over the whole workload, counting shared state once, and binds one -lifecycle per state. Binding checks that the choice is legal and that states on -one maintenance path share an evaluation schedule. An alternative whose cost is -unknown can be bound only when the deployment's cost model is authoritative for -complete-candidate cost; unknown cost is never treated as zero. It then yields the same -lifecycle guarantee and window framework the physical compiler consumes when -Planner selects. Planner's own cheapest-alternative selection remains available -for callers without deployment pricing. The window framework is decided for the -complete combination, not for one alternative in isolation. - -A maintained population (for example, the current series of `topk by(job)(1, m)`) -is retained state like a summary. Retaining it maintains the latest sample per -series at ingestion and leaves only the readout at query time. Choosing -`Ephemeral` rebuilds that snapshot from raw samples for each query, so the -deployment must supply the raw source at query time. The caller's `CostModel` -prices both through the same lifecycle hooks; a model without population -evidence leaves them unknown, and they are not selected. +**Stage 2 Materialization (#509)** will choose when states are built and how +long they are retained, using workload demand, window/freshness requirements +and supported physical implementations. Backend selection uses runtime +feasibility and cost after physical compilation. Retained state includes +maintained populations (for example, the current series of +`topk by(job)(1, m)`), which keep the latest sample per series at ingestion and +leave only the readout at query time. For the running example, assume it selects: @@ -303,24 +280,24 @@ reuse: one merged state serves p50 and p99 ``` -This produces the **Summary Maintenance Lifecycle**. +This is the running example's **Materialization**. -The lifecycle specifies how the selected logical summary should be maintained, +It specifies how the selected logical summary should be maintained, but not its concrete operator implementation or storage location. Physical feasibility may feed back into selection. For example, if the required -pane-based maintenance cannot be implemented, this lifecycle candidate cannot be +pane-based maintenance cannot be implemented, this materialization cannot be selected. One-minute panes alone also cannot cover an arbitrarily phased query window; that requires supported boundary handling or a different candidate. -## 3. Summary Maintenance Lifecycle → Physical DAG +## 3. Materialization → Physical DAG The **Physical Plan Compiler** consumes both computation semantics and maintenance requirements: ```text Logical Post-ASAP DAG (PostAsapDAG) -+ Summary Maintenance Lifecycle ++ Materialization + physical capabilities ↓ Physical Plan Compiler @@ -328,15 +305,14 @@ Physical Plan Compiler Physical DAG(s) ``` -For the running example, the lifecycle creates two execution boundaries. +For the running example, materialization creates two execution boundaries. These two halves are named as `PhysicalASAPDAG` names them, `precompute` -and `query`. *Maintenance* stays the lifecycle's word (section 2): it covers +and `query`. *Maintenance* stays materialization's word (section 2): it covers how state is built, retained, reused and scheduled. A precompute DAG is the -physical object that a maintenance lifecycle compiles to, so reusing -*maintenance* for it collapses two layers that the crates keep apart: -`asap-aware-mapping::summary_maintenance_*` owns the lifecycle, and -`asap-physical-operators::physical_planner` owns the DAGs. +physical object that maintenance compiles to, so reusing *maintenance* for it +collapses two layers: Stage 2 materialization (#509) owns maintenance, and +`asap-executor::physical_planner` owns the DAGs. ### Precompute Physical DAG @@ -397,17 +373,17 @@ DAG. If the required behavior cannot be realized, physical compilation fails. Materialization frontiers are Planner decisions. A candidate records both the precompute Physical DAG and the query Physical DAG, with typed outputs connecting them. The deployment compiler binds those outputs; it does not move operators. -Lifecycle timing gives the frontier: ingestion-time nodes read by query-time -nodes. For `sum by(job)(rate(m[1m]))`, the two lifecycle choices of the single +Execution timing gives the frontier: ingestion-time nodes read by query-time +nodes. For `sum by(job)(rate(m[1m]))`, two materialization choices for the single logical candidate give: ```text -Candidate A (Rate state retained, Sum Ephemeral): +Candidate A (Rate state at ingestion time, Sum at query time): precompute: counter samples → per-series Rate state materialized output: per-series Rate states for window/evaluation/revision query: stored Rate states → Rate readout → grouped Sum → result -Candidate B (Rate and Sum states retained): +Candidate B (Rate and Sum states at ingestion time): precompute: counter samples → per-series Rate → grouped Sum state materialized output: grouped Sum states for window/evaluation/revision query: stored grouped Sum states → Sum readout → result @@ -430,9 +406,9 @@ feasibility is rejected before pricing. The optimizer supplies candidate frontiers and cost evidence, including updates, retention, recurrence and sharing. `enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. -The lifecycle layer decides timing; physical compilation reads it. Lowering a +Materialization decides timing; physical compilation reads it. Lowering a node does not depend on the frontier, so each query DAG is lowered once and -different lifecycle assignments are different cuts of that lowering. +different materialization assignments are different cuts of that lowering. `compile(dag, inputs, roots)` yields the complete `CompiledPhysicalDAG`. `frontier_from_timing(&timed_dag)` reads an assignment's timed DAG (from `execution_timed_dag`) and returns its frontier: ingestion-time nodes read by @@ -448,7 +424,7 @@ pane candidates remain a separate lowering. Physical compilation opens no readers. Bounded precompute outputs become typed query inputs. Their source, filters, grouping, build window, evaluation time, readiness and -revision contracts must accompany the selected lifecycle and be checked during +revision contracts must accompany the selected materialization and be checked during deployment binding. Type compatibility alone does not establish reuse legality. The Planner integration test executes both candidates through the shared runtime @@ -463,7 +439,7 @@ The **Deployment Plan Compiler** binds the Physical DAGs to the concrete deploym ```text Physical DAGs -+ Summary Maintenance Lifecycle ++ Materialization + deployment catalog/state + sources/materializations + operational policy @@ -507,7 +483,7 @@ InputSlot[5 panes] ``` The Deployment Plan Compiler establishes bindings and checks that their contracts -satisfy the physical inputs and selected lifecycle, including KLL parameters, +satisfy the physical inputs and selected materialization, including KLL parameters, source, filters, grouping, window coverage and revision scope. The deployment engine resolves request-specific states and checks their actual coverage, revisions and readiness at execution time. A compiled plan cannot establish future readiness. @@ -523,8 +499,8 @@ The complete example makes the ownership boundary explicit: | Stage | KLL example decision | | --- | --- | | **Logical Post-ASAP DAG** | Use `KLL(k=200)` with shared merge for p50/p99 | -| **Summary Maintenance Candidate Generation** | Maintain 1-minute panes and reuse them for aligned five-minute queries | -| **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements and each node's execution timing | +| **Stage 2 Materialization (#509)** | Maintain 1-minute panes and reuse them for aligned five-minute queries | +| **Materialization** | Record pane/window/freshness/reuse requirements and each node's execution timing | | **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | | **Physical DAG** | Define precompute and query DAGs with typed input/output boundaries | | **Deployment Plan Compiler** | Bind raw input and KLL state slots to concrete sources/materializations | @@ -534,7 +510,7 @@ The complete example makes the ownership boundary explicit: Logical: "Use KLL for p50/p99." -Lifecycle: +Materialization: "Maintain reusable 1-minute KLL panes." Physical: @@ -547,7 +523,7 @@ Deployment: ``` The deployment engine executes the bound Physical DAGs through ASAPPlanner's -shared physical operator implementation library, `asap-physical-operators`, and +shared physical operator implementation library, `asap-executor`, and its DAG runtime. The merge executes once per run for both consumers. Execution does not introduce additional planning decisions. @@ -564,17 +540,11 @@ snapshots as separate inputs. ## 6. Executable acceptance coverage -The tests cover optimizer-selected lifecycle execution alongside independent +The tests cover physical candidate execution alongside independent operator/runtime fixtures: | Test | Contract exercised | | --- | --- | -| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | -| `summary_maintenance_lifecycle_e2e::chosen_lifecycle_timing_decides_precompute_contents` | PromQL workload → enumerated lifecycles → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the state in precompute, Ephemeral leaves precompute empty and reads the raw source at query time; both return the same p99 | -| `summary_maintenance_lifecycle_e2e::lifecycle_timing_cuts_one_compilation` | KLL quantile and grouped Rate→Sum: one compilation cut by the ContinuouslyMaintained and Ephemeral timed DAGs equals `compile_candidate` for each; the frontier is the retained state or empty | - -| `summary_maintenance_lifecycle_e2e::chosen_population_lifecycle_decides_precompute_contents` | PromQL `topk by(job)` over a maintained population → explicit choice → timed DAG → compiled candidate; ContinuouslyMaintained stores the population in precompute, Ephemeral rebuilds it from raw samples at query time; both rank alike | -| `summary_maintenance_lifecycle_e2e::planner_lifecycle_selection_reproduces_strategy_timing` | For PromQL summary fixtures, the timed DAG from Planner's retained selection equals the DAG realization strategies produce | | `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | | `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | | `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | diff --git a/docs/design_docs/proposals/asap-aware-mapping/README.md b/docs/design_docs/proposals/asap-aware-mapping/README.md index a58d3b5dd..814d50134 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/README.md +++ b/docs/design_docs/proposals/asap-aware-mapping/README.md @@ -9,4 +9,3 @@ extensions. Each status note identifies the implemented scope and remaining work - [Shared maintained populations](maintained-populations.md) - [Optimization dimensions](optimizations.md) - [Summary properties](summary-properties.md) -- [Workload demand and summary lifecycle](workload-demand-and-summary-lifecycle.md) diff --git a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md index 520501731..1524bb811 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md +++ b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md @@ -1,8 +1,8 @@ # Analytical resource cost model > Status: implemented model with explicit support limits. The -> [analytical estimator](../../../../crates/asap-aware-mapping/src/analytical_cost.rs) -> and physical/streaming adapters implement supported evidenced comparisons. +> [analytical estimator](../../../../crates/plan-selection/src/cost/analytical_cost.rs) +> and physical-plan adapter implement supported evidenced comparisons. > Unsupported operators, arrival modes and missing evidence remain unavailable; > proposed extensions are not implied by the implemented formulas. @@ -17,23 +17,19 @@ plans. These are separate concerns: evidence to estimate CPU work, peak memory, and source/disk I/O. The physical-resource estimator itself is independent of the arrival mode. -Two planner adapters currently lower work into it. `PhysicalPlanCostModel` -compares complete at-rest plans. `SummaryMaintenanceCostModel` resolves -`DataArrival::ContinuouslyIngesting` over a finite horizon, including -bootstrap, arriving updates, retained state, and query readout. Evidence from -one arrival mode must not be reused for the other. `Mixed` and `Unknown` -remain unavailable until their distinct data regions are modeled. - -Both entry points replace dimensionless plan-node counts with estimates +One planner adapter currently lowers work into it: `PhysicalPlanCostModel` +compares complete at-rest plans. Continuously-ingesting, `Mixed` and `Unknown` +comparisons are unavailable; costing summary maintenance over arriving data +belongs to Stage 2 materialization (#509). Evidence from one arrival mode must +not be reused for another. + +The adapter replaces dimensionless plan-node counts with estimates derived from operator complexity, cardinality, row width, and concrete summary parameters. The estimates are predictions; they are not measurements reported by a physical executor. The model does not decide semantic legality. Ordinary summary guarantees are -composed before costing. A window framework that itself introduces error must, -however, carry a typed composed guarantee in the same complete evidence bundle; -the streaming adapter checks that guarantee against every bound workload -accuracy target before the candidate can be ranked. Missing evidence produces +composed before costing. Missing evidence produces an unavailable estimate, never an assumed zero or a structural-cost fallback. The implementation keeps five layers distinct: @@ -64,21 +60,12 @@ physical_operator_statistics.rs ──────┤ physical evidence contract ▼ analytical_cost.rs ─────────── operator formulas and CPU/memory/I/O composition │ - ├──────────────► physical_plan_cost_model.rs - │ at-rest raw/rewrite/summary comparison adapter - │ - └──────────────► summary_maintenance_cost/ - evidence.rs authoritative summary evidence - estimator.rs complete maintenance-DAG resources - window.rs window assignment and accuracy - model.rs lifecycle/alternative ranking adapter + └──────────────► physical_plan_cost_model.rs + at-rest raw/rewrite/summary comparison adapter ``` -Raw query plans and incrementally maintained summary plans share -`EvidenceBackedPhysicalDAG`; there is no streaming-only duplicate of the -physical DAG or operator-statistics contract. Summary-maintenance modules add -only the evidence and scheduling semantics that do not exist for an ordinary -query plan. +Raw query plans and summary plans share `EvidenceBackedPhysicalDAG`; there is +no separate physical DAG or operator-statistics contract for summaries. An estimate has physical dimensions: @@ -134,7 +121,7 @@ required: the result cache is filled by evaluations in this horizon. Buffer residency is a steady-state capacity/working-set model; it does not model cold-start warming or access order. -`asap_types::resources` is the single definition site for `CacheProfile`, +`asap_types::workload::resources` is the single definition site for `CacheProfile`, `CacheEvidence`, and `CacheCapacityEvidence` (implemented in `resources/cache.rs`). These schemas describe cache assumptions, not additive CPU or byte consumption. The mapping module re-exports the same types for existing import paths, while @@ -236,14 +223,14 @@ normalized workload, lowered query IR, and freshness-aware statistics: `DataWorkload` does define whether input is streaming: its `arrival` field is `AtRest`, `ContinuouslyIngesting`, `Mixed`, or `Unknown`, and a continuous arrival rate comes from fresh `ingestion_rate` evidence. These facts describe -how source data arrives. They do not choose a lifecycle or a window framework: +how source data arrives. They do not choose materialization or a window framework: `Incremental` describes how a selected summary state is updated, while tumbling, sliding, and exponential histogram describe how that state is organized over time. Evidence is read through `Evidence::value_at(planning_time)`. Stale, future, or improperly time-bounded evidence remains unknown. Costing follows -the same freshness rule as accuracy and lifecycle planning. +the same freshness rule as accuracy checking. The current workload schema does not yet contain every physical statistic. The missing facts have explicit ownership: @@ -306,7 +293,7 @@ physical operators and matching statistics variants. Until then, a candidate containing such an unlowered operation is unavailable rather than partially costed. -## Workload horizon and lifecycle +## Workload horizon Every alternative must cover the same source data and query horizon. The `DataArrival::AtRest` physical-DAG comparison is build-once, read-many: @@ -327,9 +314,7 @@ incremental updates are a one-time snapshot build. The at-rest summary alternative scans the selected source snapshot once and retains state. Its raw alternative recomputes from that snapshot for every -query read. The continuously-ingesting entry point separately charges -bootstrap, updates, summary operations, retained state, and raw evaluations; -its lifecycle rules are defined below. +query read. ### Comparable source and workload scope @@ -365,11 +350,11 @@ coverage set must equal the scope source set: a Scan query with an empty scope, or a source-free query with a non-empty scope, fails closed. Empty snapshot identifiers, invalid recurrence, or a zero horizon also fail closed. -Every reachable physical `Scan` carries one exact `SourceCoverage` copied from +Every reachable physical `Scan` carries one exact `ScanSelection` copied from this scope. That coverage includes the existing `Source`, its provider-owned snapshot ID, and canonical ordinary predicates or info-metric matchers. A scan with no coverage, or coverage not present in `ComparisonScope.sources`, makes the plan unavailable. Other -operators cannot declare source coverage. This prevents a DAG over source B +operators cannot declare scan selection. This prevents a DAG over source B from being estimated under source A's comparison scope. ## General DAG costing @@ -554,7 +539,7 @@ counts, releases transient output after its last consumer, and keeps retained state live. Consequently a shared scan is charged once per execution and a fan-out's memory includes the outputs that really coexist. -Each estimate independently requires the semantic set of source coverages on +Each estimate independently requires the semantic set of scan selections on its reachable Scan nodes to equal `ComparisonScope.sources`. Multiple physical Scans may repeat one coverage, but no scope source may be omitted and no Scan may add another coverage. This invariant is enforced by the estimator itself, @@ -596,7 +581,7 @@ It consumes the existing query and physical-operator enums; it does not introduce a parallel logical operator vocabulary. For every occurrence, the lowerer sends a `PhysicalNodeRequest` containing the logical node, selected existing `PhysicalOperator`, occurrence and synthetic-role metadata, already-lowered -child physical IDs, and any source coverage to a +child physical IDs, and any scan selection to a `PhysicalNodeEvidenceProvider`. The provider atomically returns its own stable `physical_id`, the authoritative `OperatorStatistics`, and explicit `output_buffer_bytes`; logical edge bytes are never substituted for an @@ -604,14 +589,14 @@ allocation. Missing evidence makes the entire query unavailable. The returned `EvidenceBackedPhysicalDAG` snapshots this evidence so costing does not re-read a live catalog after lowering. -Each lowered Scan is bound to exactly one `SourceCoverage` in the comparison +Each lowered Scan is bound to exactly one `ScanSelection` in the comparison scope by the existing source and canonical predicate values. The bound value therefore also supplies the provider-owned snapshot ID. Zero matches fail as outside scope; multiple matching coverages fail as ambiguous rather than choosing an arbitrary snapshot. When a predicate-bearing logical Scan expands to Scan → Filter, the synthetic Scan has its own physical ID, statistics, and buffer evidence and carries that exact coverage; the Filter has separate -evidence and no source coverage. +evidence and no scan selection. `ComparisonScope.sources` is an order-independent set of semantic coverages; duplicates are invalid. After lowering, every reachable physical Scan must use a member of that set and every member must be used by at least one Scan. @@ -733,157 +718,14 @@ input. ## Summary operator formulas -### Incremental single-summary foundation - -For `DataArrival::ContinuouslyIngesting`, the incremental estimator accepts -one selected lifecycle and one unique logical `SummaryAgg`. This deliberately -narrow contract prevents one flat evidence record from being reused across -several summary nodes with different input cardinalities, algorithms, or state -sizes. Complete multi-node streaming alternatives require per-node physical -evidence. - -The canonical workload supplies fresh bootstrap cardinality, ingestion rate, -query recurrence, planning time, and a finite horizon. Physical evidence adds -logical/bootstrap bytes, physical bootstrap scan bytes, active and retained -window counts, the number of concrete summary-state instances per window, and -bytes per state instance. Names use `summary`, not `sketch`, because an exact -aggregate or another non-sketch state is equally valid. - -For bootstrap rows `B`, arrivals `U`, simultaneously updated windows `A`, -query evaluations `Q`, physical summary instances `P`, and state bytes `S`: - -```text -insert invocations = (B + U) × A -retained memory = (A + retained_windows) × P × S -``` - -Each input row is routed to its matching summary instance; it is not inserted -into every group. Merge, subtract, and readout work may operate over all `P` -instances. Delete work follows the same routed window updates rather than -multiplying every update by every possible group. - -An empty bootstrap is valid and has zero logical bytes and zero source reads. -A non-empty bootstrap requires positive logical and physical source bytes. -Active window count, summary-instance count, state width, horizon, and query -evaluation count must be positive; retained-window count may be zero for a new -stream. Required per-operation CPU evidence must be finite and positive. - -Lifecycle retention and the planning horizon are different quantities. A -short retained window may be maintained throughout a much longer planning -horizon, so the estimator does not require `retention >= horizon`. Lifecycle -legality and query time-coverage checks establish whether the retained window -can answer the query. - -### Comparing single-summary lifecycle alternatives - -For one logical `SummaryAgg`, the analytical lifecycle adapter converts the -same physical evidence into the existing lifecycle planner's five cost terms: - -| Lifecycle term | Resource basis | -|---|---| -| Initial build | Bootstrap rows routed to every bootstrap-active window, plus the bootstrap source read. | -| Maintenance per update | One arriving row routed to every currently active window. | -| Summary read | Readout of every physical summary instance needed by one query evaluation. | -| Retention rate | All active and retained state bytes calibrated over the finite comparison horizon. | -| Retirement | Zero only for releasing modeled memory; an actual delete, expiration, or rebuild requires explicit operation evidence. | - -The existing lifecycle model—not this adapter—enumerates `Ephemeral`, -`Prepared`, `Shared`, and `ContinuouslyMaintained`, checks workload and runtime -legality, and multiplies per-update and per-read terms by the normalized -workload rates. Missing any required term leaves that alternative unavailable. - -`Ephemeral` is a direct build, not incremental maintenance. For every query -evaluation, it rebuilds from the snapshot visible at that evaluation, charges -that evaluation's complete source read, and releases its state afterward. -Its state contributes to peak transient memory but not persistent retention. - -The raw side is supplied as a complete `ResourceEstimate` for one execution of -the raw physical DAG. The lifecycle planner applies the same recurrence and -horizon. This deliberately avoids reconstructing raw work with a special-case -`input_rows × cpu_per_row` formula that would omit joins, windows, sorts, or -other operators. - -Flat single-summary evidence is bound to the exact `SummaryNode` and raw -`QueryExpr` identities for which it was produced. It cannot be reused for a -structurally similar node or for multiple summary states. A complete -multi-summary `SummaryExpr` DAG requires per-node physical evidence and -physical-identity deduplication. - -### Complete bound streaming summary DAGs - -The multi-node streaming path accepts a complete, already-bound -`SummaryExpr` DAG. It does not guess physical implementations. The provider must -provide evidence for every reachable node: - -| Logical node | Required physical evidence | -|---|---| -| `KeepPreAsap` | One retained preprocessing operator with output edge, horizon CPU, workspace, and output buffer. | -| `SummaryAgg` | Input/output edges, insert CPU, concrete state count and width, bootstrap/update window fanout, and explicit source-read ownership. | -| `SummaryMerge` | Typed merge evidence with total CPU, workspace, output buffer, I/O, and execution multiplicity. | -| `SummarySubtract` | Typed subtract evidence with the same resource dimensions. | -| `SummaryDelete` | Typed delete evidence plus expiration/retraction rate, routing fanout, and the exact state owner. | -| `SummaryEstimate` | Typed readout evidence with total resource use per execution. | -| `SummaryJoin` | Ordered input/output edges and total physical join CPU, workspace, output buffer, I/O, and multiplicity. | - -The merge/subtract/delete/readout evidence is an enum structured by operation -kind. Delete-only rate and routing fields therefore cannot be attached to a -merge or readout. Join CPU is the total build, probe, match-production, and -output work of the selected algorithm; matched output pairs alone are not a -valid join cost. - -Every parent input edge must equal the corresponding child output edge. -Provider-owned `physical_id` values deduplicate a shared operator only when -its complete evidence and physical child identities also agree. The cost model -holds owning `Rc` references for bound target and summary roots, so pointer -keys cannot become stale and alias a later allocation. - -A `SummaryAgg` that reads storage declares `source_coverage_index = Some(i)`, -a non-empty bootstrap-read identity, and positive physical source bytes. An -aggregate over an already-materialized summary edge declares `None`, an empty -read identity, and zero source bytes. Its logical input rows and bytes remain -positive when the intermediate is non-empty. This prevents nested aggregates -from charging the original source scan repeatedly. - -For streaming raw recomputation, `planning_time_input_rows`, -`planning_time_input_bytes`, and `planning_time_source_scan_bytes` describe the -initial snapshot. Logical bytes per arriving row and physical source bytes per -arriving row are separate. The recurrence determines every evaluation offset; -the provider supplies one once-counted physical DAG whose statistics aggregate -those evolving evaluations over the complete horizon. Marking its nodes -`PerEvaluation` would multiply the already-aggregated evidence again and is -rejected. Validation follows only nodes reachable from the physical root. If -the raw algorithm intentionally reads the same semantic source more than once, -each reachable scan carries the same evolved source statistics and is charged -separately; equal source coverage does not deduplicate physical I/O. - -This raw-evolution contract currently supports exactly one distinct source -coverage. A multi-source streaming target is unavailable until per-source -arrival rates and widths are supplied. Target lineage includes ordinary -predicates and PromQL info selectors; extra, missing, or mismatched source -coverage makes both sides incomparable. - -Lifecycle enumeration considers only alternatives legal for the canonical -workload and runtime. A `Prepared` state must cover every scheduled evaluation -it serves. `Shared.retention` describes data/window coverage, not the planning -horizon, so a shorter retention value is not rejected merely because the -optimizer horizon is longer. Missing node evidence, zero required CPU, -unknown I/O, inconsistent edges, or an unsupported lifecycle combination -makes the complete candidate unavailable; partial per-state costs are never -used as a fallback. +Costing incremental maintenance of continuously-ingested summaries was removed +with the summary maintenance lifecycle; Stage 2 materialization (#509) will +define it. The formulas below give per-operation work and state size. ### Ranking complete physical implementations A logical summary candidate can be bound to more than one complete physical -implementation. Each alternative has a non-empty, provider-owned identity and -a complete `StreamingNodeEvidence` bundle. The planner evaluates every legal -lifecycle combination against every bound physical implementation over the -same `ComparisonScope`, excludes alternatives whose evidence is incomplete or -invalid, and returns both the least calibrated cost and its physical-plan -identity. Duplicate identities are rejected because they would make the -selection result ambiguous. If no explicit alternatives are registered, the -candidate's single canonical evidence bundle is used. - -Physical evidence is alternative-specific: window fanout, retained state, +implementation. Physical evidence is alternative-specific: window fanout, retained state, operation costs, and source reads must describe that implementation as a whole. The planner does not mix individual nodes from different alternatives. @@ -903,7 +745,7 @@ cpu_ops = bootstrap_rows × bootstrap_window_count × insert_ops(params) scan_bytes = source_read_bytes for the build ``` -Merge, subtract, and delete add their own invocation counts described above; +Merge, subtract, and delete add their own invocation counts; they are never folded into the simple formula implicitly. Concrete accuracy-sized parameters determine state and work: @@ -927,7 +769,7 @@ from logical group count alone. Summary merge, subtract, delete, and readout are separate physical operators. Their CPU and memory use the concrete summary state size and number of input states. A plan using one of these operations is unavailable until the -corresponding formula and required lifecycle evidence are present. +corresponding formula and required evidence are present. Summary construction uses physical-input realization rules before it emits a `SummaryAgg`. The default rule consumes the logical aggregate's immediate @@ -962,14 +804,14 @@ bound physical DAG for a `SummaryExpr` candidate. The deployment implements `PlannerPhysicalPlanProvider`: query-node evidence is consumed atomically by the generic query lowerer, while summary binding returns a complete `EvidenceBackedPhysicalDAG`, including embedded raw work, build/read operators, retained -state, execution multiplicity, and source coverage. The adapter calls +state, execution multiplicity, and scan selection. The adapter calls `estimate_physical_dag_comparison`; it never calls `DefaultCostModel` or a structural-node-count fallback for final cost. A candidate is exposed to global selection only when both complete DAGs are valid and its calibrated cost is strictly below the raw baseline. Missing or stale evidence, an unknown physical algorithm, invalid edges, incomplete -source coverage, or a candidate that is not cheaper yields `None`. When no +scan selection, or a candidate that is not cheaper yields `None`. When no candidate remains, `chosen = None` preserves the raw pre-ASAP target. Logical CSE share/recompute rewrites are not complete physical alternatives: @@ -988,9 +830,7 @@ mismatches and arithmetic overflow also fail closed. ### Downstream physical-planning boundary This cost model consumes resource evidence for a physical implementation, but -ASAPPlanner does not own or select that implementation. It does select the -abstract per-summary `SummaryWindowFramework` assignment by comparing complete -`StreamingWindowFrameworkCandidate` evidence bundles. Component ownership, +ASAPPlanner does not own or select that implementation. Component ownership, including the distinction between a window primitive and its concrete runtime implementation, is defined in [ASAPPlanner planner-runtime contract](../../architecture/planner-runtime-contract.md). @@ -1064,7 +904,7 @@ candidate. The intended end-to-end selection pipeline is: 1. enumerates semantically valid alternatives; -2. checks end-to-end accuracy and lifecycle legality; +2. checks end-to-end accuracy; 3. derives fresh workload and operator statistics; 4. sizes physical summary parameters; 5. estimates the complete candidate DAG; @@ -1072,34 +912,11 @@ The intended end-to-end selection pipeline is: The query lowerer and physical estimator cover the supported raw-query shapes listed above. `PhysicalPlanCostModel` executes this pipeline for every -candidate supplied to `CandidateLogicalASAPDAGs::global_selection`. Logical rewrites are +candidate supplied to `candidate_selection::global_selection`. Logical rewrites are lowered recursively. Summary candidates participate only after the deployment has bound their complete `SummaryExpr` DAG; there is no optimistic generic -summary fallback. The streaming adapter connects raw recomputation and -primitive summary lifecycle costs to the existing global lifecycle-selection -hooks. -The lifecycle planner enumerates compatible lifecycle combinations for the -unique `SummaryAgg` deployments and invokes -`complete_summary_candidate_estimate` -for each combination before selecting the minimum. The hook receives explicit -node-to-guarantee bindings plus the horizon and expected reads. Each logical -occurrence is looked up by exact `Rc` identity, while every -evidence record also carries a provider-owned physical identity. Equal physical -identities deduplicate work and retained state only when their logical summary, -selected window framework, operator facts, edge statistics, lifecycle -guarantee, and physical child identities agree; -conflicts make the candidate unavailable. Thus heterogeneous states are costed -independently and genuinely shared deployments once. Merge, subtract, delete, -readout, and join participate in automatic -candidate ranking. Exhaustive whole-root scoring is capped at 4,096 lifecycle -combinations because an arbitrary whole-candidate hook cannot be soundly -pruned by primitive costs; a larger space is unavailable rather than consuming -exponential planner time. If the root needs unavailable operation evidence, the -hook returns unavailable. Global selection then excludes that summary and -materialization retains the raw expression. A missing raw estimate also forces -raw fallback, because no public selection/materialization path may publish an -uncompared summary. The planner never falls back to the partial `SummaryAgg` -sum. +summary fallback. Choosing between a maintained summary and raw recomputation +belongs to Stage 2 materialization (#509). Before applying the following arithmetic, callers validate exact equality of the raw and selected alternative's `ComparisonScope`, and use the same diff --git a/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md b/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md index f71dde2ea..8bb452783 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md +++ b/docs/design_docs/proposals/asap-aware-mapping/ddsketch-quantile-ratios.md @@ -14,7 +14,7 @@ The final guarantee records both input ranges and their contract identifiers. Th ## Candidate generation without evidence -The default `SketchAlgorithmStrategy` permits a direct DDSketch quantile-ratio +The default `ASAPStrategies` permits a direct DDSketch quantile-ratio candidate when domain evidence is absent, but leaves the root guarantee unset. This is useful for the v1 integration path; it does not turn missing evidence into evidence. Other approximate divisions still require their own composition diff --git a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md index 005670db8..91382160a 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md +++ b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md @@ -1,7 +1,7 @@ # Shared maintained population rule > Status: planner rule implemented; deployment support is conditional. See -> [MaintainedPopulationStrategy](../../../../crates/asap-aware-mapping/src/maintained_population.rs). +> [MaintainedPopulationStrategy](../../../../crates/logical-optimizer/src/pass1/maintained_population.rs). > A deployment must provide the membership, freshness, state and operation > capabilities described below. Planner representation alone does not implement > population maintenance in a runtime. @@ -51,7 +51,7 @@ column and grouping determine whether consumers refer to the same population. ## Rule: share one population across compatible readouts **Realization:** `MaintainedPopulationStrategy`, an opt-in `ReplacementStrategy` -in [maintained_population.rs](../../../../crates/asap-aware-mapping/src/maintained_population.rs). +in [maintained_population.rs](../../../../crates/logical-optimizer/src/pass1/maintained_population.rs). **Target sub-DAGs:** @@ -70,7 +70,7 @@ columns and multi-measure aggregates need additional rules. ```text KeepPreAsap(input) - -> MaintainPopulation { input, max_k, quantiles } [lifecycle-timed] + -> MaintainPopulation { input, max_k, quantiles } [ingestion time] -> ReadPopulation { Quantile(q1) } [read] -> ReadPopulation { Quantile(q2) } [read] -> ReadPopulation { TopK(k1) } [read] @@ -116,8 +116,7 @@ because their source names or numeric values happen to agree. ## Validation, selection and execution responsibilities Planner validates the declared input, the query-time readout and readout -compatibility; the population's lifecycle decides whether it is maintained at -ingestion or rebuilt per query. Its intended guarantee is exact membership and exact readout; +compatibility; `MaintainPopulation` always runs at ingestion time. Its intended guarantee is exact membership and exact readout; a physical implementation still must preserve the language's numeric and empty-input semantics. In particular, SQL global COUNT over an empty population returns a row with zero, while PromQL COUNT over an empty vector returns an empty vector. diff --git a/docs/design_docs/proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md b/docs/design_docs/proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md deleted file mode 100644 index 483e04503..000000000 --- a/docs/design_docs/proposals/asap-aware-mapping/workload-demand-and-summary-lifecycle.md +++ /dev/null @@ -1,758 +0,0 @@ -# Design: Query Workloads, Data Workloads, and Summary Lifecycle Maintenance - -> Status: partially implemented design. Workload types and -> [lifecycle planning APIs](../../../../crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs) -> implement the bounded planning path described below. The current-support and -> future-work sections distinguish available behavior from broader search, -> forecast integration and runtime deployment work. - -## Audience and context - -This document is for ASAPPlanner designers, architects, researchers, and -developers working on workload-aware plan selection. It defines how the -planner should describe query workload, data workload, and the lifecycle of -summary state. It is a design contract, not a description of the current -public Rust API. - -The terminology follows the ProjectASAP -[glossary](https://github.com/ProjectASAP/internal-docs/blob/03e1c70f5af3ae9221471898541067eee7f86338/glossary.md). -That glossary is authoritative for the meanings of data workload, query -workload, ad-hoc and predictable queries, one-time and repeated queries, -real-time and longitudinal queries, output cardinality, and lookback window. -This document maps those concepts into planner responsibilities and records -where the current model is incomplete. - -This design is orthogonal to -[end-to-end accuracy guarantees](end-to-end-accuracy-guarantees.md). Accuracy -decides whether a candidate is correct enough. Workload demand and state -lifecycle decide whether building, maintaining, sharing, or recomputing that -candidate is worthwhile. Neither decision may override the other. - -## Problem and why now - -A summary operator does not imply one execution lifecycle. The same exact or -approximate summary can be: - -- built once from data at rest and discarded after one query; -- prepared before a known future query and retired afterward; -- shared across a bounded set of requests; or -- maintained incrementally as data continues to arrive. - -Likewise, an exact stateless operator may run once over a batch, once per -update in an incremental pipeline, or once per readout. Operator statefulness, -execution schedule, and output representation are separate properties. - -The query expression alone cannot determine those properties. The same query -may arrive unexpectedly during exploration, run once at a scheduled time, or -repeat every ten seconds on a dashboard. Planning summary state from syntax -alone either misses reuse or invents reuse that the workload does not justify. - -The current `PlanningWorkload` separates query demand from data arrival. -Query entries include predictability, recurrence or invocation count, accuracy, -and time selection; `DataWorkload` contains arrival and empirical facts. -These fields do not themselves select a summary-maintenance lifecycle. -The [input/output/workflow design](../../architecture/input-output-workflow.md) -is authoritative for current fields, defaults, and public call sequences. - -## Inputs, outputs, and end-to-end behavior - -For the broader lifecycle design, four categories of information matter -(these are not four current top-level Rust fields): - -1. logical queries, which define query semantics; -2. query workload, including per-query accuracy and latency requirements, - predictability, recurrence, and queried time scope; -3. data-workload characteristics, including arrival, volume, cardinality, and - distribution; -4. existing summaries and the lifecycle actions available to the deployment. - -Candidate search outputs `CandidateLogicalASAPDAGs`. The implemented lifecycle-aware workflow -then returns a `SummaryMaintenanceLifecyclePlan` per query root, containing the -Post-ASAP DAG and maintenance decisions. It can choose exact raw recomputation -when summary maintenance does not beat raw cost or comparable costs are missing. A -state deployment states whether a summary is ephemeral, prepared, shared for a -bounded period, or continuously maintained. It retains costs, assumptions, and -structured rejection reasons. Exporting full input provenance remains a later -integration. - -```text - logical queries ---+ - query workload -----+ - data workload ---+--> candidate plans - available summaries ---+ -> semantic and accuracy legality - -> lifecycle alternatives - -> horizon-normalized cost - -> selected plan + deployments -``` - -For an unpredictable one-time query, the planner may read an existing summary, -build an ephemeral summary, or recompute from raw data. It must not assume -future reuse. For a predictable one-time query, it may additionally compare -preparing state in advance with building or recomputing at execution time. For -repeated queries, it may amortize build and maintenance cost across reads over -an explicit horizon. - -### End-to-end decision order - -```text -normalize query and data workloads - -> derive recurrence, time-scope, and data evidence - -> enumerate semantic plan alternatives - -> enumerate legal execution contracts and state lifecycles - -> validate summary capabilities and phase constraints - -> derive and check accuracy guarantees - -> normalize one-time and rate costs over an explicit horizon - -> rank legal alternatives and compare the selected summary deployment - with raw recomputation - -> emit plan, deployments, assumptions, and rejected alternatives -``` - -## Goals and non-goals - -### Goals - -- Represent glossary-defined query-workload and data-workload concepts without - collapsing independent axes into one enum. -- Separate an operator's statefulness from its execution schedule and the - lifecycle of the state it produces. -- Make unknown demand explicit and fail closed rather than treating it as zero - or infinite reuse. -- Compare one-time and rate-valued costs only through an explicit horizon. -- Explain why a selected plan builds, reuses, maintains, or avoids summary - state. -- Preserve a minimal path from the current batch/repeating workload and - recurrence profile to the proposed model. - -### Non-goals - -- Scheduling jobs, assigning machines, admission control, or executing queries. -- Predicting future query text inside ASAPPlanner. -- Defining a sketch runtime or state-storage protocol. -- Choosing a concrete forecasting algorithm for uncertain demand. -- Changing accuracy targets or guarantee algebra. - -## Heilmeier questions - -- **What are we trying to do?** Choose whether summary state should be built, - maintained, shared, reused, or avoided for different query workloads - and data workload. -- **How is it done today, and what are the limits?** The planner distinguishes - one-shot counts, fixed repeating intervals, and an ingest-rate proxy. It - cannot distinguish an unexpected exploratory query from a scheduled one-time - report, or data at rest from continuous ingestion as an explicit mode. -- **What is new, and why will it succeed?** Orthogonal workload axes and an - explicit state lifecycle let the existing recurrence formulas compare the - same summary under different deployment choices without changing query - semantics. -- **Who cares?** Users need predictable latency and cost; operators need to - know what state will exist and for how long; planner developers need demand - assumptions to be auditable. -- **What are the risks and costs?** More inputs can make planning harder to - configure, forecasts may be stale, and a large lifecycle search space can - increase planning cost. -- **What are the checks for success?** The acceptance cases below must produce - different lifecycle alternatives and cost terms for identical query syntax - under different workload contracts. - -## Proposed design - -### Authoritative concepts and ownership - -| Concept | Authoritative layer | Reason | -| --- | --- | --- | -| Query meaning | Pre-ASAP query IR | Workload metadata must not change semantics | -| Accuracy requirement | Query workload (per-query) | The required result fidelity may be explicit or supplied by the normalization default | -| Response-latency requirement | Query workload (per-query) | The optional end-to-end response-time bound belongs to one query execution | -| Query workload | Workload input | Arrival and recurrence are not inferable from syntax | -| Data workload | Workload input | Ingestion and distribution describe the data, not query workload | -| Summary capability | Summary properties | Merge, delete, and update support constrain legal lifecycles | -| State lifecycle | Physical planning decision | Lifecycle is selected, not declared by `SummaryAgg` | -| Cost | Cost model and explanation | Cost consumes all inputs but does not define their meaning | - -### Query workload - -Accuracy and latency are separate per-query requirements within the query -workload. They constrain different planner decisions and must not be collapsed -into one SLA value: - -```rust -enum AccuracyRequirement { - /// The caller supplied the required result fidelity. - Explicit(AccuracyTarget), - /// The source omitted accuracy; normalization applies the exact default. - ImplicitExact, -} - -enum LatencyRequirement { - /// Maximum permitted end-to-end latency for one query execution. - ExplicitMax(Duration), - /// The caller supplied no latency bound. - Unspecified, -} - -struct QueryRequirements { - accuracy: AccuracyRequirement, - response_latency: LatencyRequirement, -} -``` - -An omitted accuracy field is not an unknown accuracy target and does not permit -arbitrary approximation: the current normalization policy makes it -`ImplicitExact`. Keeping that variant distinct from `Explicit(Exact)` preserves -whether the caller chose exactness or inherited the default. An unspecified -response-latency requirement imposes no response-time constraint; it is not a -zero-duration bound or evidence that every latency is acceptable. Accuracy is -checked as a legality constraint. The normalized model preserves response -latency, but the current planner does not yet reject plans against that bound. - -#### Classification axes - -The glossary classifications must be modeled independently. - -##### Predictability - -```rust -enum Predictability { - /// The query shape is not known before arrival. - AdHoc, - /// The query or parameterized template is known before execution. - Predictable { - known_at: Option, - }, - /// The caller supplied no reliable classification. - Unknown, -} -``` - -`AdHoc` does not mean repeated or one-time. It means the query shape was not -known in advance. The glossary currently places exploratory/ad-hoc queries in -the one-time category, so the MVP should accept `AdHoc + OneTime` and reserve -other combinations until a concrete use case establishes their semantics. - -##### Recurrence - -```rust -enum QueryRecurrence { - OneTime { - invocations: u64, - execute_at: Option, - }, - Repeated { - demand: RepeatedDemand, - }, - Unknown, -} - -enum RepeatedDemand { - FixedInterval(Duration), - Scheduled(Vec), - EstimatedRate(DemandEstimate), -} - -struct DemandEstimate { - /// Time range over which the demand was measured or forecast. - observation_window: ObservationWindow, - /// Expected demand, expressed in exactly one form. - expected: ExpectedDemand, - /// Highest expected invocation rate within the observation window. - peak_rate: Option, - /// Highest expected number of simultaneously executing invocations. - max_concurrency: Option, - /// Confidence in this estimate, in the inclusive range [0.0, 1.0]. - confidence: Confidence, - source: EvidenceSource, - observed_at: Option, - valid_for: Option, -} - -enum ExpectedDemand { - /// Expected total invocations over `observation_window`. - InvocationCount(u64), - /// Expected average invocations per second over `observation_window`. - AverageRate(Rate), -} - -struct ObservationWindow { - start: Timestamp, - end: Timestamp, -} - -struct Confidence(f64); -``` - -One-time means no recurrence is expected for that workload entry. Several -one-time consumers may still share a subplan within a submitted workload. -Repeated means the same query expression over its selected data is evaluated -over time, matching the glossary. Parameterized templates require an explicit -equivalence policy before their executions count as the same query. - -Query-workload volume is more than an average rate. Cost and latency can differ -for the same total request count when requests arrive in bursts or concurrently. -`ExpectedDemand` makes invocation count and average rate alternative -representations, preventing conflicting values in one estimate. The observation -window must be non-empty, rates must be finite and non-negative, and -`Confidence` must be between zero and one. Fixed intervals and explicit -schedules are declarations rather than estimates and do not need fabricated -confidence. The MVP may cost only invocation count and evaluation rate, but it -must preserve unsupported volume characteristics for explanation rather than -silently discarding them. - -##### Queried time scope - -```rust -enum QueryTimeScope { - RealTime, - Longitudinal, - Mixed, - Unknown, -} -``` - -`QueryTimeScope` is not a response-latency requirement. It classifies the event -time of the data selected by the query; `LatencyRequirement` constrains the -wall-clock time allowed to produce the result. They are independent: a -longitudinal query over archived data may require a 100 ms response, while a -real-time query over the latest data may permit a 30 second response. - -This classification is not derived only from a numeric lookback. A five-minute -lookback over recent data is real-time; the same duration over archived data is -not. Planning input should therefore carry the classification and the concrete -time selection separately: - -```rust -struct TimeSelection { - scope: QueryTimeScope, - lookback: Option, - as_of: Option, -} -``` - -For example, the same five-minute lookback has a different scope depending on -whether it is anchored at the current planning time or at a historical time: - -```rust -// The last five minutes: real-time. -TimeSelection { - scope: QueryTimeScope::RealTime, - lookback: Some(Duration::minutes(5)), - as_of: None, -} - -// A five-minute interval from archived data: longitudinal. -TimeSelection { - scope: QueryTimeScope::Longitudinal, - lookback: Some(Duration::minutes(5)), - as_of: Some(timestamp!("2024-01-01T12:05:00Z")), -} -``` - -`lookback` is a query property already represented by temporal query nodes in -some frontends. The normalized workload should reference or derive it rather -than introduce a second conflicting value. - -### Data workload is separate from query workload - -```rust -enum DataArrival { - AtRest, - ContinuouslyIngesting, - Mixed, - Unknown, -} - -/// Statistical distribution of keys in the input data. -enum DataDistribution { - /// A small number of keys account for most observations. - Zipf, - /// Keys are approximately equally likely. - Uniform, - /// Observations arrive in bursts with a temporarily concentrated key set. - Bursty, -} - -struct DataWorkload { - arrival: DataArrival, - ingestion_volume: Evidence, - ingestion_rate: Evidence, - input_cardinality: Evidence, - distribution: Evidence, -} -``` - -`DataDistribution` reuses the existing ASAPPlanner classification. It describes -the key-frequency shape used by summary accuracy and cost models, not whether -data arrives continuously. An unavailable or unsupported distribution is -represented by `Evidence.value = None` rather than by assuming the default -distribution. - -The former `DataCharacteristics` was a stale, continuous-ingestion-specific -case built around series count and samples per second. `DataWorkload` replaces -it as the normalized input rather than embedding that special case in the -general model. Data at rest may have row count and scan statistics without a -nonzero ingestion rate. Unknown arrival must not be interpreted as continuously -ingesting or at rest. - -Every empirical value uses an evidence wrapper conceptually containing: - -```rust -struct Evidence { - value: Option, - source: EvidenceSource, - observed_at: Option, - valid_for: Option, -} -``` - -This reuses the provenance and freshness principles from empirical summary -parameter configuration. Missing, stale, or future-dated evidence remains -unknown. - -### Output cardinality is a derived or evidenced cost input - -Output cardinality depends on input cardinality and grouping columns. The -planner may derive it analytically, accept a catalog estimate, or leave it -unknown. The source and freshness metadata must be preserved because output -cardinality affects summary size, read cost, post-processing cost, and network -cost. It is not a query correctness requirement. - -### Separate operator state, schedule, and output - -The physical design must not use `SummaryAgg` as shorthand for incremental -maintenance. - -```rust -enum OperatorState { - Stateless, - Stateful { - mergeable: bool, - deletable: bool, - }, -} - -enum EvaluationSchedule { - OneShot, - PerUpdate, - OnRead, -} - -enum OutputRepresentation { - PlainRows, - SummaryState, - FinalizedValue, -} -``` - -A one-shot sketch builder is stateful while it consumes its input, but it does -not imply long-lived incremental maintenance. A stateless transform can run -`PerUpdate` before a downstream maintained summary. These types describe an -execution contract; they do not replace semantic operators in the post-ASAP IR. - -### State lifecycle is a plan alternative - -```rust -enum StateLifecycle { - Ephemeral, - Prepared { - activate_at: Timestamp, - retire_at: Timestamp, - }, - Shared { - retention: Duration, - }, - ContinuouslyMaintained, -} -``` - -- `Ephemeral` builds state for one submitted workload and discards it afterward. -- `Prepared` builds or begins maintaining state before a predictable query and - retires it after the known need ends. -- `Shared` retains state for multiple consumers over a bounded lifetime. -- `ContinuouslyMaintained` applies data updates until an explicit later - deployment decision retires the state. - -The summary family and its properties constrain which lifecycles are legal. -For example, an append-only sketch may support continuous inserts but not a -sliding-window lifecycle requiring deletion. Lifecycle legality is checked -before cost ranking, like accuracy legality. Deployments provide these -per-summary properties through `summary_lifecycle_capabilities`; moving -real-time windows require deletion support as well as incremental updates. - -### Existing summaries are planning input - -An ad-hoc query cannot justify creating permanent state from unknown future -demand, but it may use compatible state that already exists. The planning -problem therefore needs a state catalog describing identity, parameters, -coverage, freshness, accuracy guarantee, lifecycle, and ownership. Catalog -integration is a separate implementation increment; this design only requires -that "reuse existing" and "create new" remain distinguishable alternatives. - -### Cost over a horizon - -`H` is the optimization horizon: the future wall-clock duration over which the -planner compares one-time and recurring costs. The existing cost model -represents it in seconds: - -```rust -/// A finite, strictly positive optimization duration, in seconds. -struct Horizon(f64); -``` - -The horizon is not the query lookback, the queried time scope, or the response -latency bound. It answers only "over how much future execution time should -these alternatives be costed?" All alternatives in one decision must use the -same `H`. `reads(H)` is the number of query evaluations expected or scheduled -within that horizon; for a fixed evaluation rate it is -`H * evaluation_rate`, plus any separately modeled one-time invocations. -Who supplies `H`, and whether a deployment may default it, remains an explicit -architecture decision below. If no horizon is available, the planner must not -compare a one-time cost with a rate-valued cost. - -For a stateful incremental alternative over horizon `H`: - -```text -total(H) = build_cost - + H * update_rate * maintenance_cost_per_update - + reads(H) * summary_read_cost - + H * retention_cost_rate - + retirement_cost -``` - -For repeated raw recomputation: - -```text -total(H) = reads(H) * raw_recompute_cost -``` - -The current lifecycle-aware materialization sums the selected summary -deployments and can replace that plan with raw recomputation when the raw cost -is lower or the summary lifecycle is uncostable. Jointly reconsidering every -sibling semantic candidate under lifecycle costs remains a later optimizer -integration; this document does not claim that broader search is implemented. - -For an ephemeral summary: - -```text -total = invocations * (build_cost + summary_read_cost + retirement_cost) -``` - -`retirement_cost` consistently means the one-time cost of ending a summary -state lifecycle, including deallocation or other cleanup. For ephemeral state, -retirement happens immediately after each invocation; for prepared, shared, or -continuously maintained state, it happens when that deployment is retired. - -For prepared state, update and retention terms apply only between activation -and retirement. Existing state does not pay a new build cost, but its catalog -provenance must establish that assumption. - -The existing `Cost`, `CostRate`, `EvaluationRate`, `UpdateRate`, `Horizon`, and -`total_cost` types are the minimum viable foundation. The implementation should -extend their explanations and lifecycle coverage instead of creating a second -recurrence cost system. - -### Unknown and uncertain demand - -Unknown demand is not zero demand and is not evidence of future reuse. The MVP -policy is: - -- do not select newly created long-lived state solely on unknown future reuse; -- allow raw recomputation, ephemeral build, and reuse of already available - compatible state; -- retain an explicit explanation of the missing demand evidence; and -- require an explicit planning objective before using an estimated demand - distribution. - -Future uncertain-demand support may add expected-cost, percentile-cost, -worst-case, or regret objectives. Those policies must consume a typed estimate -with confidence and provenance; they are not implicit behavior of -`Predictability::Unknown`. - -## Review against the ProjectASAP glossary - -The glossary review found the following required coverage and current gaps. - -| Glossary concept | Current ASAPPlanner representation | Missing design support | -| --- | --- | --- | -| Data at rest vs continuously ingesting | `DataArrival` is explicit | Runtime/catalog-specific arrival discovery remains external | -| Ingestion volume | `DataWorkload::ingestion_volume` carries evidence | A concrete time basis for volume remains deployment-specific | -| Ingestion rate | Evidenced independently from query evaluation rate | Preserve richer unit/provenance metadata when integrations require it | -| Input cardinality | Evidenced workload-level cardinality feeds accuracy | Per-dataset/metric/column scoping remains future work | -| Data distribution | Evidenced built-in enum | Permit deployment-specific distributions later | -| Ad-hoc vs predictable | `Predictability` is independent from recurrence | Parameterized-template equivalence remains open | -| One-time vs repeated | One-time, fixed, scheduled, estimated, and unknown recurrence | Forecast-policy integration remains future work | -| Query volume and characteristics | Estimates preserve average/count, peak, concurrency, confidence, and freshness | Peak and concurrency are not yet consumed by cost or latency models | -| Real-time vs longitudinal | `TimeSelection` carries scope, lookback, and `as_of` | Conflict policy with temporal IR remains open | -| Output cardinality | May be inferred locally; no common evidenced input | Add derived/evidenced value and provenance for costing | -| Lookback window | Represented in temporal query shapes/frontends | Establish query IR as authority and expose it to workload costing | -| CTSA pipeline | Not explicitly modeled | Keep as architectural context; planner consumes collect/store/analyze facts but does not model transmission topology in the MVP | -| CSP(F) | Cost and fidelity partly modeled | Treat scale/performance/fidelity as objectives and constraints; do not collapse fidelity into cost | - -Two terminology constraints apply: - -1. A repeated query is not inherently a streaming-data workload. It may - repeatedly query data at rest. -2. A one-time query is not inherently stateless. A predictable one-time query - may justify prepared state, while an ephemeral summary is stateful during - its one execution. - -## Minimal complexity - -The minimum input model is determined by the downstream applications selected -for integration, not by a context-free notion of the fewest possible fields. -Each supported use case must contribute the workload facts that can change -plan legality, accuracy, lifecycle, or cost: - -- Time-series metric queries require queried time scope and lookback. -- Repeated dashboard queries, including an ASAPQuery integration, require - recurrence and evaluation frequency so the planner can cost reuse and - maintenance across executions. -- Batch queries over data at rest require an explicit at-rest arrival mode and - must not be assigned a fabricated ingestion rate. -- Summary techniques whose accuracy depends on the input distribution require - evidenced distribution characteristics; omitting them must produce unknown - accuracy or a conservative fallback rather than a favorable assumption. - -The initial implementation should include the union of fields required by its -committed integrations. Additional workload dimensions should be added when a -new downstream use case demonstrates that they affect a planning decision. - -The simplest alternative is to extend `BatchEntry` with optional schedule and -classification fields and extend `RepeatingEntry` with time scope. That is a -reasonable serialization migration, but it is not a sufficient conceptual -model: it continues to make predictability and recurrence mutually exclusive -container choices, and it has no place for data arrival or state lifecycle. - -The minimum new conceptual layers are therefore: - -1. orthogonal query-demand metadata, required because glossary categories are - not one taxonomy; -2. data-workload metadata, required because ingestion does not describe query - recurrence; -3. state lifecycle as a physical alternative, required because one summary - operator can be deployed ephemerally or incrementally. - -No separate scheduler, forecasting framework, or replacement cost model is -introduced. Existing query IR, summary properties, accuracy model, and -recurrence cost types remain authoritative in their domains. - -## Alternatives and decisions - -### Encode workload class as one enum - -Rejected. Variants such as `AdHoc`, `OneShot`, and `Repeated` overlap: -predictability and recurrence are different facts, and time scope is a third. - -### Infer demand from query syntax or submitted root count - -Rejected. Syntax contains no evidence of future arrival, and several roots in -one request establish only current structural sharing. - -### Treat every summary as continuously maintained - -Rejected. It excludes ephemeral construction over data at rest and overcharges -one-time plans. It also hides deployment lifetime from explanations. - -### Treat every one-time query as raw recomputation - -Rejected. An ephemeral summary may reduce memory or network cost during one -execution, an existing summary may already answer the query, and a predictable -future query may justify preparation. - -### Fold fidelity into a scalar cost - -Rejected. Accuracy and semantic correctness are constraints checked before -ranking. A cheaper plan cannot purchase permission to violate fidelity. - -### Extend the existing recurrence profile only - -Partially accepted for implementation reuse, rejected as the whole model. -`RecurrenceProfile` is an aggregated cost context for a target. It should remain -the derived input to cost decisions, while normalized workload metadata retains -predictability, time scope, provenance, and lifecycle information needed before -and after aggregation. - -## Quality attributes and evidence - -- **Understandability:** explanations use glossary terms and show each axis - separately. Proxy: reviewers can distinguish repeated queries from continuous - ingestion in exported plan evidence. -- **Debuggability:** selected and rejected lifecycle alternatives record costs, - horizon-derived decisions, assumptions, and typed rejection reasons. Full - demand/data provenance in exported explanations remains future work. -- **Maintainability:** current recurrence types remain the cost authority; - normalized workload types remain the source authority. No duplicate formula - system is introduced. -- **Extensibility:** scheduled and estimated recurrence fit without changing - query semantics. Forecasting policies remain pluggable planning objectives. -- **Performance:** lifecycle enumeration expands the candidate space. The MVP - should generate only capability-compatible alternatives and deduplicate - equivalent deployments before ranking. -- **Operability:** every long-lived state has activation, retention or retirement - semantics and ownership in output. Concrete runtime APIs are future work. -- **Security and privacy:** query logs and empirical distributions may be - sensitive. Provenance must identify a source without requiring raw query-log - contents to be embedded in exported plans. - -## Acceptance and test design - -Realization acceptance is defined by identical logical queries producing -different legal lifecycle choices under different workload contracts: - -1. **Unpredictable one-time query:** offers raw recomputation, compatible - existing state, and ephemeral build; does not justify new continuous state. -2. **Predictable scheduled one-time query:** may offer prepared state with a - bounded activation and retirement period. -3. **Repeated query over continuously ingesting data:** compares incremental - maintenance and repeated recomputation using distinct update and evaluation - rates over an explicit horizon. -4. **Repeated query over data at rest:** uses evaluation rate without inventing - maintenance updates. -5. **Real-time and longitudinal queries with the same expression:** preserve - different time selections and may receive different scan, retention, and - summary alternatives. -6. **Unknown demand:** remains unknown in explanation and cannot make a newly - created long-lived state win through assumed reuse. -7. **Mixed one-time and repeated consumers:** requires an explicit horizon and - accounts for shared build cost once. -8. **Accuracy failure:** rejects a lifecycle regardless of favorable workload - cost. - -Focused unit tests should cover normalization, invalid combinations, evidence -freshness, lifecycle capability checks, and dimensional cost arithmetic. -End-to-end tests should cover cases 1–8 through candidate selection and exported -explanations. A reviewer who did not implement the workload types should design -or review at least the unknown-demand and mixed-consumer cases; that independent -review has not occurred for this design document. - -## Risks, rollout, and exit criteria - -The implementation should roll out additively: - -1. add normalized metadata and explanations while preserving current - batch/repeating behavior; -2. derive the existing `RecurrenceProfile` from the richer model; -3. add ephemeral and existing-state alternatives; -4. add prepared and continuously maintained lifecycle selection; -5. integrate empirical demand and state catalogs only when provenance and - freshness contracts are available. - -Compatibility requires old workloads to normalize without changing their -current decisions when no new metadata is supplied. Unknown new fields must -take the documented conservative path rather than acquire optimistic defaults. - -Open decisions requiring architecture or product input: - -- whether predictable parameterized query templates count as the same repeated - query and under which equivalence relation; -- who supplies the optimization horizon and whether a deployment may define a - default for purely repeated workloads; -- which planning objective governs uncertain demand; -- how state ownership, quota, and retirement requests cross the planner/runtime - boundary; -- whether real-time versus longitudinal is supplied by the caller, derived by a - policy using `as_of` and lookback, or both with conflict diagnostics; and -- the minimum evidence freshness required before empirical workload data may - affect selection. - -The design exits draft status when these decisions have owners, the normalized -input has a compatibility plan, and acceptance cases 1–8 can be expressed in -fixtures without runtime-specific assumptions. diff --git a/docs/design_docs/proposals/asapquery-rule-coverage.md b/docs/design_docs/proposals/asapquery-rule-coverage.md index e496c9fe9..9ba999292 100644 --- a/docs/design_docs/proposals/asapquery-rule-coverage.md +++ b/docs/design_docs/proposals/asapquery-rule-coverage.md @@ -21,11 +21,11 @@ cost, and selection rules under `optimizer/`. The reviewed source is | Temporal aggregate functions | Lowering covered; realization varies | `Aggregate(PerEntity)` over `TimeRange` represents the full family. Sum, count, min, max, quantile, rate, and increase have summary realizations; `avg_over_time` is currently exact `PassThrough`, matching ASAPQuery's exact-only multi-stat fallback rather than claiming a maintained summary. | | Spatial aggregate functions | Lowering covered; realization varies | `Aggregate(Reduce(GroupKeys))` is shared by SQL and PromQL. Supported single accumulators and ordinary `by(...)` avg rewrites generate candidates; shapes such as `avg without(...)` retain the same exact raw fallback that ASAPQuery uses for multi-stat AQEs. | | Collapsible temporal + spatial aggregates | Semantic-equivalent rewriting | The existing rewrite strategy uses accumulator algebra: sum∘sum, sum∘count, min∘min, and max∘max. It rejects all other pairs and requires identical output schemas. | -| Sketch alternatives and exact fallback | Covered more generally | `SketchAlgorithmStrategy` enumerates legal summary realizations. The enclosing memo group always retains the original raw expression as the exact fallback; the strategy does not falsely label an approximate sketch as exact. | +| Sketch alternatives and exact fallback | Covered more generally | `ASAPStrategies` enumerates legal summary realizations. The enclosing memo group always retains the original raw expression as the exact fallback; the strategy does not falsely label an approximate sketch as exact. | | Subpopulation label placement | Covered more generally | `HydraGroupingStrategy` and `GroupingStrategy` express per-subpopulation and shared multi-subpopulation realizations. | | Shared computation | Covered more generally | workload-wide CSE and `SharedSubDAGStrategy` operate on physical DAG identity rather than AQE names. | | Average decomposition | Semantic-equivalent rewriting | The same rewrite strategy exposes independently optimizable sum/count accumulators when null semantics and schema permit it. | -| Merge/delete legality | Covered | Summary-family capabilities and lifecycle validation determine which maintenance operations are legal. | +| Merge/delete legality | Covered | Summary-family capabilities determine which maintenance operations are legal. | | Window-framework selection | Separate physical-planning work | Window selection must compare an extensible set of implementations, including tumbling, sliding, PromSketch-style exponential-histogram windows, and other window frameworks. This audit does not introduce a closed window enum or choose among them. | | Retention/cleanup scheduling | Outside planner scope | The audit deliberately does not import ASAPQuery's Arroyo-specific cleanup thresholds, timers, or failure workarounds. The planner may declare a selected summary's required retention horizon and cost it, but the runtime/storage layer owns when and how expired physical state is reclaimed. | | Empirical per-sketch atomic costs | Covered through evidence | Analytical statistics and deployment profiles provide cost evidence; benchmark tables should be ingested as calibrated evidence rather than compiled into matching rules. | @@ -38,22 +38,22 @@ does not create a new strategy category. | Decision | Existing owner | |---|---| -| Which summary algorithm can implement one aggregate intent | `SketchAlgorithmStrategy` | +| Which summary algorithm can implement one aggregate intent | `ASAPStrategies` | | How grouping/subpopulation state is laid out | `HydraGroupingStrategy` | | Whether an equivalent logical expression exposes better accumulators | `SemanticEquivalentRewriteStrategy` (the broadened existing avg rewrite; `AvgToSumOverCountStrategy` remains a compatibility name) | | Whether identical physical work is shared | `SharedSubDAGStrategy` | | Whether a finer grouping can answer a coarser grouping | `RollupStrategy` | | Whether tighter accuracy can answer a looser request | `AccuracyReconciliationStrategy` | | Whether a larger Top-K result can answer a smaller limit | `TopKLimitReuseStrategy` | -| Which maintenance lifecycle is legal | the summary-maintenance lifecycle planner | +| Whether and when a summary is materialized | Stage 2 materialization (#509) | | Which window framework implements a range | the physical deployment/window-selection planner | | How expired physical state is cleaned up | runtime/storage lifecycle management, not a planner strategy | Accordingly, ASAPQuery's four collapsible temporal/spatial patterns extend the existing semantic-rewrite owner. Temporal and spatial function recognition is already front-end lowering into `AggIntent`; sketch compatibility remains in -`SketchAlgorithmStrategy`; labels remain in `HydraGroupingStrategy`; and -maintenance lifecycle legality remains in the lifecycle planner. Window +`ASAPStrategies`; labels remain in `HydraGroupingStrategy`; and +materialization decisions belong to Stage 2 materialization (#509). Window framework selection is separate physical-planning work. None of these become a parallel syntax-oriented `PatternStrategy`. @@ -69,7 +69,7 @@ planner. Rules match typed operators and declared capabilities, never parser spellings. A rule that composes operators states the algebraic law it relies on and preserves the original output schema. Unknown pairs, missing statistics, or -unsupported lifecycle operations produce no candidate; they never silently +unsupported maintenance operations produce no candidate; they never silently fall back to an optimistic estimate. Window choices should follow the same principle without assuming that every diff --git a/docs/design_docs/proposals/error-resource-profile.md b/docs/design_docs/proposals/error-resource-profile.md index d4ccd4a04..f7ef87206 100644 --- a/docs/design_docs/proposals/error-resource-profile.md +++ b/docs/design_docs/proposals/error-resource-profile.md @@ -1,16 +1,16 @@ # Error–Resource Profile (ERP) -> Status: implemented ERP v1 with proposed extensions. The -> [ERP module](../../../crates/asap-aware-mapping/src/erp.rs) consumes discrete -> empirical profiles; benchmark observations are not worst-case accuracy proofs. -> See the status, supported selection modes and remaining work below. +> Status: ERP v1 was implemented in `asap-aware-mapping::erp` and deleted under +> #572 because the planner never called it. Benchmark observations are not +> worst-case accuracy proofs. ## Status ERP v1 is a discrete, shape- and distribution-conditioned profile exchanged between `sketch-bench` and ASAPPlanner. It is an empirical planning input, not a proof -of a worst-case sketch guarantee. The implementation lives in -`asap-aware-mapping::erp`; `approxbench erp` exports the producer artifact. +of a worst-case sketch guarantee. The planner-side implementation +(`asap-aware-mapping::erp`) was deleted under #572; `approxbench erp` exports +the producer artifact. ## Motivation diff --git a/docs/design_docs/proposals/planner-layering-example1-acceptance.md b/docs/design_docs/proposals/planner-layering-example1-acceptance.md new file mode 100644 index 000000000..dc4b83d74 --- /dev/null +++ b/docs/design_docs/proposals/planner-layering-example1-acceptance.md @@ -0,0 +1,256 @@ +# Planner layering, Example 1: acceptance spec (MVP) + +Audience: planner designers and the Phase C implementer. +Source: [planner-layering.md](planner-layering.md), "Example 1: Aggregation over +dimensions". Tests: `crates/integration-tests/tests/planner_layering_example1.rs`. +Viewer fixture shape: `tools/dag-viewer/examples/planner-layering-example1.expected.json`. + +This spec was written by a test designer who did not implement the stages. +The candidate counts were later changed to follow the planner's output (user +decision, see [Candidate counts](#candidate-counts)). + +## Scope + +| Stage | Doc (#509) | MVP (this spec) | +|---|---|---| +| 0. Frontends | 1 workload `LogicalDAG` | same | +| 1. Logical ASAP | Pass 1 (3) × identical-expression rule (2) × window-composition rule (3 × 3) = **54** | Pass 1 (32) × identical-expression rule (2) = **64**. Window composition (blocked on #511: #518, #522) is not implemented. | +| 2. Physical ASAP | Materialization options per window form = **156** | Physical operator implementation only, no materialization = **64** (the doc's Raw/Raw plans, separate or with a shared input) | +| 3. Selection | 1 plan | 1 plan | + +Every MVP candidate is one of the doc's candidates: in Stage 1, the one with no +window summary for either query; in Stage 2, the one where both queries are +**Raw** (state rebuilt from the last 1 min of raw samples at every refresh). + +## Workload + +Shared data workload: `continuously_ingesting`, 15 s ingestion interval, +1,000,000 series, about 66,667 samples/s, `zipf`, volume unknown. + +| Query | Repeats | `lookback` | `as_of` | Accuracy | Latency | +|---|---|---|---|---|---| +| Q1 `sum by (job) (rate(http_requests_total[1m]))` | 10 s | 1 m | evaluation time | exact | none | +| Q2 `topk by (job) (10, sum_over_time(http_requests_total[1m]))` | 10 s | 1 m | evaluation time | ε = 0.01, δ = 0.001 | ≤ 100 ms | + +## Stage 0: 1 candidate + +| Query | Chain | +|---|---| +| Q1 | scan `http_requests_total` → range 1m → rate (per series) → sum by (job) | +| Q2 | scan `http_requests_total` → range 1m → sum_over_time (per series) → topk by (job) (10) | + +Invariants: + +* Exactly one candidate, with one root per query, in workload entry order. +* No summary node (sketch `summary_agg`, `summary_estimate`, `summary_merge`). +* Q1 and Q2 share no node. Sharing is a Stage 1 decision. + +## Candidate counts + +The original spec expected 6 Stage 1 candidates: 3 Q2 options (exact, +Count-Min + heap, Hydra) × separate or shared input. It did not account for +Pass 1's exact-accumulator options: each of Q1's `rate` and `sum` and Q2's +`sum_over_time` may stay a raw aggregate or become an exact accumulator +(`SummaryAgg` → `FinalizeExactAccumulator`). Pass 1 also offers +CountSketch + heap for Q2's top-k, and a **whole-expression** Count-Min or +CountSketch + heap: one heap sketch per `job` over the raw samples of the +range, keyed by series identity and weighted by the sample value, which +realizes `topk by (job) (10, sum_over_time(…))` as a whole and absorbs the +`sum_over_time` (no exact per-series sum is computed, so it has no choice of +its own). Pass 1 therefore yields 2 × 2 (Q1) × (3 × 2 + 2) (Q2) = **32** +combinations, and Pass 2's identical-expression rule adds a shared-input +variant of each: **64** candidates. Hydra is not produced yet; its test stays ignored, naming the +missing feature. + +## Stage 1: 64 candidates + +Every combination of Q1 `rate` {raw, Rate acc} × Q1 `sum` {raw, Sum acc} × Q2 +(top-k {exact, Count-Min + heap, CountSketch + heap} × `sum_over_time` {raw, +Sum acc}, or whole-expression {Count-Min + heap, CountSketch + heap}), twice: +L1–L32 with each query reading its own `http_requests_total[1m]` input, and +L33–L64 (labelled "· shared input") with +one scan → range 1m read by both queries. Pass 2 adds the shared variant only +because sharing merges nodes; Stage 3 chooses between the variants by cost. + +Invariants: + +* Exactly 64 candidates: each combination once with separate inputs and once + with the shared input. No duplicates. +* Every candidate covers both queries. +* Q1 never reaches a sketch node. +* Only the scan and range nodes may be shared. No summary is shared. +* For each Q2 option, both an independent and a shared variant are present. +* A whole-expression sketch reads the range directly; no `sum_over_time` is + computed in its candidates. +* Pending (ignored test): the sketch families in Q2 are {`CmsWithHeap`, + `CountSketchWithHeap`} per subpopulation, over the per-series sums or whole + expression, and Hydra `HydraCms` (no Hydra alternative yet). + +## Stage 2: 64 candidates + +`Pn` implements `Ln`. Exact Q2 top-k becomes sort (partition by job) → limit +10; a sketch Q2 is a heap-sketch build → top-10 estimate, which returns the +selected rows (`job`, series identity, `value`); every Q2 option has this +Stage 1 schema. Everything runs at query time. + +Invariants: + +* Exactly one physical candidate per logical candidate. `from_logical` is a + bijection onto the Stage 1 ids, so no valid candidate is dropped before + Stage 3. +* Each physical candidate keeps its logical Q2 option and input sharing. +* Exact TopK is implemented as a sort followed by a limit (16 candidates). +* A summary Q2 is a sketch build node feeding an estimation node. There is no + merge node, because the MVP has no window summaries. +* No node runs at ingestion time, because the MVP has no materialization. +* The runtime's physical planner compiles every candidate Stage 3 finds + valid, and rejects the 24 Count-Min + heap candidates for the reason Stage 3 + gives: their update weights are not proven non-negative. + +### Candidates for manual review + +Generated from `tools/dag-viewer/examples/planner-layering-example1.json`. +"acc" is an exact accumulator; "raw" keeps the relational aggregate. + +| Id | Label | Q1 choice | Q2 choice | Input | Stage 3 outcome | Reason | +|---|---|---|---|---|---|---| +| P1 (L1) | Q1 exact · Q2 exact | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 86.401 vs 52.201 cpu ms | +| P2 (L2) | Q1 exact · Q2 exact (Sum acc) | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 83.401 vs 52.201 cpu ms | +| P3 (L3) | Q1 exact · Q2 CMS+heap | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P4 (L4) | Q1 exact · Q2 CMS+heap (Sum acc) | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P5 (L5) | Q1 exact · Q2 CountSketch+heap | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 198.401 vs 52.201 cpu ms | +| P6 (L6) | Q1 exact · Q2 CountSketch+heap (Sum acc) | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 195.401 vs 52.201 cpu ms | +| P7 (L7) | Q1 exact · Q2 whole-expression CMS+heap | rate: raw; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P8 (L8) | Q1 exact · Q2 whole-expression CountSketch+heap | rate: raw; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 568.401 vs 52.201 cpu ms | +| P9 (L9) | Q1 exact (Rate acc) · Q2 exact | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 83.401 vs 52.201 cpu ms | +| P10 (L10) | Q1 exact (Rate acc) · Q2 exact (Sum acc) | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 80.401 vs 52.201 cpu ms | +| P11 (L11) | Q1 exact (Rate acc) · Q2 CMS+heap | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P12 (L12) | Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc) | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P13 (L13) | Q1 exact (Rate acc) · Q2 CountSketch+heap | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 195.401 vs 52.201 cpu ms | +| P14 (L14) | Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 192.401 vs 52.201 cpu ms | +| P15 (L15) | Q1 exact (Rate acc) · Q2 whole-expression CMS+heap | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P16 (L16) | Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 565.401 vs 52.201 cpu ms | +| P17 (L17) | Q1 exact (Sum acc) · Q2 exact | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 85.401 vs 52.201 cpu ms | +| P18 (L18) | Q1 exact (Sum acc) · Q2 exact (Sum acc) | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 82.401 vs 52.201 cpu ms | +| P19 (L19) | Q1 exact (Sum acc) · Q2 CMS+heap | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P20 (L20) | Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc) | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P21 (L21) | Q1 exact (Sum acc) · Q2 CountSketch+heap | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 197.401 vs 52.201 cpu ms | +| P22 (L22) | Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 194.401 vs 52.201 cpu ms | +| P23 (L23) | Q1 exact (Sum acc) · Q2 whole-expression CMS+heap | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P24 (L24) | Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 567.401 vs 52.201 cpu ms | +| P25 (L25) | Q1 exact (Sum acc, Rate acc) · Q2 exact | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | separate | valid, costlier | 82.401 vs 52.201 cpu ms | +| P26 (L26) | Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | separate | valid, costlier | 79.401 vs 52.201 cpu ms | +| P27 (L27) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P28 (L28) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P29 (L29) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | separate | valid, costlier | 194.401 vs 52.201 cpu ms | +| P30 (L30) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | separate | valid, costlier | 191.401 vs 52.201 cpu ms | +| P31 (L31) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | separate | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P32 (L32) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | separate | valid, costlier | 564.401 vs 52.201 cpu ms | +| P33 (L33) | Q1 exact · Q2 exact · shared input | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 59.201 vs 52.201 cpu ms | +| P34 (L34) | Q1 exact · Q2 exact (Sum acc) · shared input | rate: raw; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 56.201 vs 52.201 cpu ms | +| P35 (L35) | Q1 exact · Q2 CMS+heap · shared input | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P36 (L36) | Q1 exact · Q2 CMS+heap (Sum acc) · shared input | rate: raw; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P37 (L37) | Q1 exact · Q2 CountSketch+heap · shared input | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 171.201 vs 52.201 cpu ms | +| P38 (L38) | Q1 exact · Q2 CountSketch+heap (Sum acc) · shared input | rate: raw; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 168.201 vs 52.201 cpu ms | +| P39 (L39) | Q1 exact · Q2 whole-expression CMS+heap · shared input | rate: raw; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P40 (L40) | Q1 exact · Q2 whole-expression CountSketch+heap · shared input | rate: raw; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 541.201 vs 52.201 cpu ms | +| P41 (L41) | Q1 exact (Rate acc) · Q2 exact · shared input | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 56.201 vs 52.201 cpu ms | +| P42 (L42) | Q1 exact (Rate acc) · Q2 exact (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 53.201 vs 52.201 cpu ms | +| P43 (L43) | Q1 exact (Rate acc) · Q2 CMS+heap · shared input | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P44 (L44) | Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P45 (L45) | Q1 exact (Rate acc) · Q2 CountSketch+heap · shared input | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 168.201 vs 52.201 cpu ms | +| P46 (L46) | Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: Rate acc; sum: raw | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 165.201 vs 52.201 cpu ms | +| P47 (L47) | Q1 exact (Rate acc) · Q2 whole-expression CMS+heap · shared input | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P48 (L48) | Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap · shared input | rate: Rate acc; sum: raw | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 538.201 vs 52.201 cpu ms | +| P49 (L49) | Q1 exact (Sum acc) · Q2 exact · shared input | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 58.201 vs 52.201 cpu ms | +| P50 (L50) | Q1 exact (Sum acc) · Q2 exact (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | valid, costlier | 55.201 vs 52.201 cpu ms | +| P51 (L51) | Q1 exact (Sum acc) · Q2 CMS+heap · shared input | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P52 (L52) | Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P53 (L53) | Q1 exact (Sum acc) · Q2 CountSketch+heap · shared input | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 170.201 vs 52.201 cpu ms | +| P54 (L54) | Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: raw; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 167.201 vs 52.201 cpu ms | +| P55 (L55) | Q1 exact (Sum acc) · Q2 whole-expression CMS+heap · shared input | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P56 (L56) | Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap · shared input | rate: raw; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 540.201 vs 52.201 cpu ms | +| P57 (L57) | Q1 exact (Sum acc, Rate acc) · Q2 exact · shared input | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: raw | shared | valid, costlier | 55.201 vs 52.201 cpu ms | +| P58 (L58) | Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: exact (sort → limit); sum_over_time: Sum acc | shared | **selected** | cheapest valid (52.201 cpu ms) | +| P59 (L59) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap · shared input | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: raw | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P60 (L60) | Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: Count-Min + heap; sum_over_time: Sum acc | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P61 (L61) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap · shared input | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: raw | shared | valid, costlier | 167.201 vs 52.201 cpu ms | +| P62 (L62) | Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input | rate: Rate acc; sum: Sum acc | top-k: CountSketch + heap; sum_over_time: Sum acc | shared | valid, costlier | 164.201 vs 52.201 cpu ms | +| P63 (L63) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap · shared input | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression Count-Min + heap over raw samples | shared | invalid | q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative | +| P64 (L64) | Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap · shared input | rate: Rate acc; sum: Sum acc | top-k and sum_over_time: whole-expression CountSketch + heap over raw samples | shared | valid, costlier | 537.201 vs 52.201 cpu ms | + +Count-Min + heap stays invalid, whole-expression or not: Q2 ranks +`sum_over_time` of raw samples, and nothing in the workload declares +`http_requests_total` non-negative (no metric type), so neither existing proof +(`UnitCount`, `ResetAwareCounterDerivative`) applies. CountSketch admits signed +weights. + +## Stage 3: 1 plan + +Invariants: + +* One selected id. Every other candidate is listed once as rejected, with a + reason, and marked invalid (accuracy, latency or capability) or valid but + costlier. +* Each priced candidate's cost has one entry per DAG node, and `total` is + their sum. A shared node is therefore charged once, for all its consumers. + Stage 3 prices valid candidates only. +* The selected plan is no more expensive than any valid candidate. +* For each Q2 option, the shared-input candidate costs no more than its + separate counterpart. +* A shared-input candidate is selected, matching the doc's "Raw with a shared + input" winner. It saves exactly one scan and one range node over the same + choices with separate inputs. + +Outcome with the built-in models: P58 (all exact, with exact accumulators for +Q1's rate and sum and Q2's sum_over_time, over the shared input) at 52.201 cpu +ms, against 79.401 for the same choices with separate inputs (P26). The scan +(23.2) and range (4.0) are priced once instead of twice. The cheapest +whole-expression CountSketch + heap plan, P64, costs 537.201: its sketch +updates 4,000,000 raw samples × (depth 125 + 1 heap update) = 504.0, against +19.001 for Q2's exact Sum accumulator (5.0) and sort + limit (14.001). The doc's other typical winners need +window forms or materialization and are out of MVP scope. + +## Ambiguities and MVP deviations + +1. **Stage 2 is materialization-only in the doc.** Example 1 lists only + window-form or materialization options. The MVP reads "physical operator + implementation" from the Stage 2 section: "TopK as a sort followed by a + limit". This gives one physical candidate per logical candidate. The doc + names no alternative implementation (for example hash versus sort + aggregation) for this example. +2. **Where exact TopK becomes sort + limit.** The Pass 1 diagram already + draws exact Q2 as "sort + limit 10 per job", but the Stage 2 section calls + this a physical choice. Today the frontend emits `aggregate[top_k]`. The + spec requires sort → limit only by Stage 2. +3. **Minimal Pass 2.** Only the identical-expression rule applies. The + window-composition rule adds the rest of the doc's 54 candidates, and its + tumbling and sliding variants wait on #511. Sharing the range selector + needs its rows to have a provable key, which CSE's legality rule requires: + a PromQL series has at most one sample per timestamp, so (series identity, + timestamp) is declared as the scan's key. +4. **"Shared summary charged once"** does not apply literally: the doc says no + summary is shared in Example 1. The tests check the general form (each node + charged once) on the shared input node. +5. **Workload DAG with two roots.** `LogicalASAPDAG` has one `root`, but + Example 1 needs one DAG for both queries. The stubs carry `query_roots` + until the export supports several roots. +6. **Where cost is computed.** The viewer contract puts `cost` on Stage 2 + candidates, but the doc says only Stage 3 uses the cost model. The spec has + Stage 3 produce costs, and the viewer document attaches them to Stage 2 + entries. +7. **Hydra's inner sketch.** The doc says "Hydra over the whole `job` + column" without naming the inner sketch. The spec assumes `HydraCms`, the + construction with the proven guarantee for frequencies. +8. **Other top-*k* families.** The code also offers `CountSketchWithHeap`. + The doc lists only Count-Min + heap and Hydra; the tests follow the + planner and include it. +9. **Latency.** The doc says an exact top 10 rebuilt at every refresh "may + miss" 100 ms. Whether exact-Q2 candidates are rejected is left to the cost model. +10. **Accuracy of Count-Min heap merges** matters only with tumbling windows, + which are out of MVP scope. +11. **Whole-expression top-*k*.** The doc draws Q2's sketch options over + "input". The planner offers both a heap sketch over the exact per-series + sums and one over the raw samples of the range (sum_over_time is additive, + so per-series weights add up in the sketch). Both are kept; Stage 3 ranks + them by cost. diff --git a/docs/design_docs/proposals/planner-layering-example2-acceptance.md b/docs/design_docs/proposals/planner-layering-example2-acceptance.md new file mode 100644 index 000000000..753e67131 --- /dev/null +++ b/docs/design_docs/proposals/planner-layering-example2-acceptance.md @@ -0,0 +1,149 @@ +# Planner layering, Example 2: acceptance spec + +Audience: planner designers and the implementers of the summary-capability +rule, UnivMon sizing and accuracy, and SQL `Entropy`/`L2` recognition. +Source: [planner-layering.md](planner-layering.md), "Example 2: One summary for +several computations". Tests: +`crates/integration-tests/tests/planner_layering_example2.rs` (helpers in +`tests/planner_layering_common/`). + +This spec was written by a test designer who does not implement the stages. +Tests run the library pipeline `plan_selection::plan_stages` (Stage 1 → 2 → 3). +A test that needs a missing feature is `#[ignore]`d with the feature's name. + +## Workload + +Three SQL dashboard panels over `flows`, every 10 s, `lookback` 1 m, `as_of` +evaluation time, no latency requirement. + +| Query | Computation | SQL (abridged) | Accuracy | +|---|---|---|---| +| Q1 | `Distinct(src_ip)` | `SELECT COUNT(DISTINCT src_ip) FROM flows WHERE ts >= now() - INTERVAL '1 minute'` | ε = 0.02, δ = 0.01 | +| Q2 | `Entropy(src_ip)` | `SELECT -SUM(p * LN(p)) FROM (SELECT COUNT(*) * 1.0 / SUM(COUNT(*)) OVER () AS p … GROUP BY src_ip)` | ε = 0.05, δ = 0.01 | +| Q3 | `L2(src_ip)` | `SELECT SQRT(SUM(c * c)) FROM (SELECT src_ip, COUNT(*) AS c … GROUP BY src_ip)` | ε = 0.01, δ = 0.01 | + +Data workload: the shared one (`continuously_ingesting`, about 66,667 +samples/s, `zipf`, volume unknown) with `input_cardinality` = 10,000,000 and no +`data_ingestion_interval`. The catalog is `flows(ts, src_ip)`; the doc names +no other column. + +**Stand-in.** The SQL frontend lowers Q1 to a distinct count but Q2 and Q3 to +exact relational plans (the doc's own TODO). Stages 1–3 therefore run on a +PromQL stand-in with the same structure, three statistics of one input over +the same 1-min window: `distinct_over_time(flows_src_ip[1m])`, +`entropy_over_time(flows_src_ip[1m])` and `l2_over_time(flows_src_ip[1m])`, +with Q1–Q3's targets and a 15-s ingestion interval (PromQL needs one). + +## Expected candidates per stage + +| Stage | Doc | Today (stand-in) | +|---|---|---| +| 0 | One summary-free DAG: `Distinct`, `Entropy`, `L2` over `flows.src_ip`, last 1 m | Same for the stand-in. SQL: Q1 is `aggregate:cardinality`; Q2 and Q3 are `count` per `src_ip` → window function / projection → `sum` (not recognized). | +| 1, Pass 1 | Each computation: exact, a specialized summary, UnivMon. 3 × 3 × 3 = **27** | Q1: exact, HLL, Theta, KMV, UnivMon. Q2, Q3: exact, UnivMon (no specialized entropy or norm summary). 5 × 2 × 2 = **20** | +| 1, Pass 2 | Summary-capability rule: + 1 (one UnivMon for all three, sized for ε = 0.01) + 9 (3 pairs share a UnivMon × 3 options of the third). Independent kept. **37**. Window composition left out. | Identical-expression rule only: a shared-input variant of every combination, **40**. In that variant, UnivMons chosen by several queries merge into one build because Pass 1 gives every UnivMon the same parameters. | +| 2 | Same options as Example 1: no window form → rebuilt from raw samples each refresh. | One physical candidate per logical candidate, everything at query time. | +| 3 | One UnivMon sized for ε = 0.01 vs three separate summaries; the shared one usually wins (one update per flow record instead of three). | HLL rejected (its bound has no failure probability for the (ε, δ) target); every UnivMon rejected (no accuracy model). An all-exact shared-input plan is selected. | + +## Invariants and tests + +### Stage 0 + +| Invariant | Test | Status | +|---|---|---| +| The SQL workload is valid, in entry order | `workload_encodes_example2` | passes | +| SQL Q1 lowers to a distinct count over `flows` | `stage0_sql_q1_lowers_to_a_distinct_count` | passes | +| SQL Q2 and Q3 lower to `FrequencyEntropy` and `FrequencyL2` | `stage0_sql_q2_q3_lower_to_entropy_and_l2` | ignored: SQL frontend recognition | +| The stand-in lowers to the three statistics over one 1-min range | `stage0_standin_lowers_to_three_statistics_of_one_input` | passes | +| One summary-free DAG; the queries share no node | `stage0_one_summary_free_dag_without_sharing` | passes | + +### Stage 1 + +| Invariant | Test | Status | +|---|---|---| +| Each statistic has an exact and a UnivMon option | `stage1_pass1_offers_exact_and_univmon_for_each_statistic` | passes | +| Q1 has a specialized distinct-count summary (HLL, Theta or KMV) | `stage1_pass1_offers_a_distinct_count_summary` | passes | +| Q2 and Q3 have a specialized entropy and norm summary | `stage1_pass1_offers_specialized_entropy_and_l2_summaries` | ignored: Pass 1 families | +| Every combination of the queries' own options is kept, with no shared summary | `stage1_keeps_every_independent_combination` | passes | +| A candidate has one UnivMon build read by all three queries, feeding a distinct-count, an entropy and an L2 estimate | `stage1_summary_capability_adds_one_univmon_for_all_three` | passes (see note) | +| For each pair, a candidate shares one UnivMon between the two while the third takes each of its own options | `stage1_summary_capability_adds_pairwise_shared_univmons` | ignored: summary-capability rule | +| A shared UnivMon has the parameters Pass 1 gives its strictest consumer alone | `stage1_shared_univmon_is_sized_for_the_strictest_consumer` | passes (see note) | +| Pass 1 sizes UnivMon per target: Q3's (ε = 0.01) is larger than Q2's (ε = 0.05) | `stage1_univmon_is_sized_per_accuracy_target` | ignored: UnivMon sizing | +| Only a UnivMon is ever shared across statistics | `stage1_only_univmon_is_shared_across_statistics` | passes | +| Candidates are valid DAGs with unique ids | `stage1_candidates_are_valid_and_uniquely_named` | passes | + +Note: the all-three UnivMon exists today only because Pass 1's UnivMon +parameters ignore ε, so the identical-expression rule merges the three builds +in the shared-input variant. Once UnivMon is sized per target, these two tests +fail until the summary-capability rule sizes one UnivMon for the strictest +consumer. + +### Stage 2 + +| Invariant | Test | Status | +|---|---|---| +| One physical candidate per logical candidate | `stage2_keeps_every_logical_candidate` | passes | +| The shared UnivMon stays one build read by all three queries | `stage2_keeps_the_shared_univmon_as_one_build` | passes | + +Materialization and window variants are out of scope: the doc leaves out the +window-composition variants for this example, and a summary with no window +form is rebuilt at every refresh (Example 1's table). + +### Stage 3 + +| Invariant | Test | Status | +|---|---|---| +| One selected id; every other candidate rejected once with a reason | `stage3_selects_one_and_explains_the_rest` | passes | +| The selected plan is the cheapest valid one | `stage3_selects_cheapest_valid` | passes | +| Each node is charged once (a shared summary is costed once) | `stage3_charges_each_node_once` | passes | +| UnivMon candidates are judged by an accuracy model, not rejected for lacking one | `stage3_judges_univmon_with_an_accuracy_model` | ignored: UnivMon accuracy model | +| One UnivMon for all three costs no more than three separate UnivMons | `stage3_shared_univmon_costs_no_more_than_three` | ignored: UnivMon accuracy model | +| Selection direction: the selected plan has at most one UnivMon, because one shared UnivMon dominates several over the same input | `stage3_selected_plan_has_at_most_one_univmon` | passes (vacuously today: no UnivMon is valid) | + +The doc's "the shared candidate usually wins" depends on the cost model, so +it is tested only as the dominance above: one build sized for the strictest +consumer costs no more than three builds that include one of that size. Whether +it also beats the exact plans is the model's call. + +## Ambiguities and conflicts with the planner + +1. **SQL recognition.** Q2 and Q3 lower to exact relational plans, so the doc's + workload has no `Entropy` or `L2` intent. Hence the stand-in. The stand-in's + statistics are per series (`PerEntity`), the doc's over the whole column; + with one `flows_src_ip` series they coincide. +2. **Pass 1 options.** The doc gives each computation three options. The + planner gives Q1 five (three specialized distinct-count summaries) and Q2, + Q3 two (no specialized entropy or norm summary). The tests check presence, + not counts. +3. **Candidate count.** The doc's 37 excludes the identical-expression rule's + shared-input variant, which the planner adds (Example 1 counts it). Tests do + not pin a count. +4. **Independent UnivMons on a shared input.** The doc keeps independent + candidates. The planner keeps them only with separate inputs: in the + shared-input variant, identical UnivMon builds always merge. +5. **Sizing.** "Sized for the strictest requirement, ε = 0.01" presumes + UnivMon sizing per ε; Pass 1 uses fixed UnivMon parameters. +6. **Accuracy model.** Stage 3 has no UnivMon accuracy model and rejects every + UnivMon candidate. It also rejects HLL against an (ε, δ) target because the + HLL bound has no failure probability; the doc treats a distinct summary as + valid. +7. **Pairwise count.** "The third keeps any of its own 3 options" includes its + own UnivMon, sized for itself. The test requires the third's full Pass 1 + option set. +8. **Data workload.** Only `input_cardinality` and `data_ingestion_interval` + change, so the ingestion rate stays about 66,667/s with 10,000,000 keys. +9. **SQL candidates that fail to build.** On `flows(ts, src_ip)`, 180 of the + 500 SQL candidates fail in Stage 1: sketch alternatives for `COUNT(*) … + GROUP BY src_ip` reference `ColumnRef::SampleValue`, and the schema has no + `value` column. With extra numeric columns they build, apparently reading + another column. This is a Pass 1 defect independent of this example. + +## Devtool + +No `--example planner-layering-2` yet. It needs: + +* SQL `Entropy` and `L2` recognition (frontend), so the demo shows the doc's + workload rather than exact relational plans; +* the Pass 1 fix in ambiguity 9: today `stage_pipeline` aborts exporting the + first unbuildable candidate; +* a SQL lowering path with the `flows` catalog in `stage_pipeline` (devtool + only; it lowers PromQL only today). diff --git a/docs/design_docs/proposals/planner-layering-example3-acceptance.md b/docs/design_docs/proposals/planner-layering-example3-acceptance.md new file mode 100644 index 000000000..e99314710 --- /dev/null +++ b/docs/design_docs/proposals/planner-layering-example3-acceptance.md @@ -0,0 +1,115 @@ +# Planner layering, Example 3: acceptance spec + +Audience: planner designers and the implementer of Pass 2's +window-composition rule. +Source: [planner-layering.md](planner-layering.md), "Example 3: Aggregation over +windows" and the Pass 2 section. Tests: +`crates/integration-tests/tests/planner_layering_example3.rs` (helpers and the +shared Example 3/4 workloads in `tests/planner_layering_common/`). Devtool: +`stage_pipeline --example planner-layering-3a` and `planner-layering-3b`. + +This spec was written by a test designer who does not implement the stages. +Tests run `plan_selection::plan_stages`. A test that needs a missing feature +is `#[ignore]`d with the feature's name. + +## Workload + +Shared data workload, except Pattern A's `arrival` is `mixed`. + +**Pattern A.** One `query_batch`, `invocations: 1` at T, `ad_hoc`, ε = 0.005, +δ = 0.01. Tests use T = 2026-01-01T00:00:00Z and a 365-day year. + +| Query | `lookback` | `as_of` | +|---|---|---| +| q1 `quantile_over_time(0.99, latency_ms[5y])` | 5 y | T | +| q2 `quantile_over_time(0.99, latency_ms[1y])` | 1 y | T | +| q3 `quantile_over_time(0.99, latency_ms[1y] offset 1y)` | 1 y | T − 1 y | +| q4 `quantile_over_time(0.99, latency_ms[1y] offset 2y)` | 1 y | T − 2 y | +| q5 `quantile_over_time(0.99, latency_ms[3y] offset 2y)` | 3 y | T − 2 y | + +**Pattern B.** `quantile_over_time(0.99, latency_ms[5m])`, every 1 min, +`lookback` 5 m, `as_of` evaluation time, ε = 0.01, δ = 0.01, ≤ 200 ms. + +## Expected candidates per stage + +| Stage | Doc | Today | +|---|---|---| +| 0 | A: five scan → range → quantile chains. B: one. | Same; q3–q5 also have a `time_shift`. | +| 1, Pass 1 | A: exact or KLL over its own interval per query, 2⁵ = 32. B: exact, KLL. | A: exact, KLL (k = 547), DDSketch per query, 3⁵ = 243. B: exact, KLL (k = 269), DDSketch. | +| 1, Pass 2 | A: window composition adds one EH of KLLs over [T − 5 y, T], one merge + p99 estimate per query, plus a candidate for every grouping of two or more queries onto shared window summaries ("hundreds"). B: each Pass 1 option × {none, sliding L = 5 min s = 1 min (5 active windows), 1-min tumbling (merge 5)} = 6. Independent kept. | Identical-expression rule only. A: a shared-input variant (shared scan, and the 2-y shift read by q4 and q5) of every combination, 486. B: nothing to share, 3. No window summaries. | +| 2 | See Example 4. | One physical candidate per logical candidate, at query time. | +| 3 | Not stated for Example 3. | A: five independent KLLs. B: KLL. | + +## Invariants and tests + +### Pattern A + +| Invariant | Test | Status | +|---|---|---| +| Both workloads are valid; Pattern B repeats every 1 min over 5 min | `workload_encodes_example3` | passes | +| Each query lowers to scan → (time shift) → range → quantile | `stage0_a_lowers_each_query_to_its_interval` | passes | +| One summary-free Stage 0 DAG; the queries share no node | `stage0_a_one_summary_free_dag_without_sharing` | passes | +| Each query has an exact and a KLL option | `stage1_a_pass1_offers_exact_and_kll_per_query` | passes | +| Five independent KLLs, each read by one query, all the same size | `stage1_a_keeps_five_independent_klls` | passes | +| The identical-expression rule shares only raw input (scan, range, shift) and keeps the unshared variant | `stage1_a_identical_expression_rule_shares_only_raw_input` | passes | +| One EH of KLLs over ≥ 5 y is read by all five queries; each has its own merge feeding a p99 estimate | `stage1_a_window_composition_adds_one_eh_for_all_five` | ignored: window composition (EH) | +| The EH candidate coexists with the five independent KLLs | `stage1_a_keeps_independent_and_shared_window_summaries` | ignored: window composition (EH) | +| Every pair of queries shares a window summary in some candidate | `stage1_a_window_composition_groups_every_pair` | ignored: window composition (partial groupings) | +| One selected id, the rest explained; the selected plan is the cheapest valid one | `stage3_a_selects_cheapest_valid` | passes | +| Each node is charged once | `stage3_a_charges_each_node_once` | passes | +| For the same local choices, sharing the scan costs no more than separate scans | `stage3_a_shared_scan_is_not_costlier` | ignored: Stage 3 row estimates (see ambiguity 7) | + +### Pattern B + +| Invariant | Test | Status | +|---|---|---| +| The query lowers to scan → range 5m → quantile, no summary | `stage0_b_lowers_to_one_range_quantile` | passes | +| Exact and KLL options | `stage1_b_pass1_offers_exact_and_kll` | passes | +| Each summary option appears with no window, a 5-min sliding window with a 1-min slide, and 1-min tumbling windows | `stage1_b_window_composition_adds_sliding_and_tumbling_per_option` | ignored: window composition (sliding, tumbling) | +| Window parameters are legal: L divides W; s divides L and the evaluation interval; a tumbling length divides W and the interval | `stage1_b_window_parameters_are_legal` | passes (vacuously today) | +| Sliding with L = W has no merge; tumbling merges before the estimate | `stage1_b_merge_only_where_the_window_form_needs_it` | ignored: window composition | +| A window form that merges uses a mergeable summary (KLL, DDSketch), in both patterns | `stage1_window_merges_use_mergeable_summaries` | passes (vacuously today) | +| One selected, cheapest valid, each node charged once | `stage3_b_selects_cheapest_valid` | passes | + +**Reading window summaries.** The IR has no window summary yet, so tests read +it through `planner_layering_common::window_form(dag, build)`, which returns +`WindowForm::None` today. The window-composition implementer makes it return +`Sliding { length_ms, slide_ms }`, `Tumbling { length_ms }` or +`ExponentialHistogram { horizon_ms }` from the new IR. Merges are read as +`SummaryMerge` nodes between the build and the `SummaryEstimate`. + +## Ambiguities and conflicts with the planner + +1. **Pass 1 families.** The doc offers exact and KLL; the planner also offers + DDSketch. Tests check presence of exact and KLL, and require window forms + for every summary option. +2. **Exact in window forms.** Pattern B counts 2 × 3 = 6, so the exact option + also gets sliding and tumbling forms. Tumbling needs a mergeable exact + state for quantiles (the retained values), which the planner has no + accumulator for. Tests do not require window forms for exact. +3. **Groupings.** "One candidate for every way of grouping two or more queries + onto shared window summaries" leaves open whether a grouping is a set + partition (several shared summaries at once) and how ungrouped queries + combine with their Pass 1 options. Tests require only every pair and all + five. +4. **Yearly tumbling windows** "would also work" for Pattern A. The doc does + not say whether Pass 2 generates them. Tests neither require nor forbid + them. +5. **EH accuracy.** A boundary inside an EH bucket is approximate. Whether the + accuracy model charges that against ε = 0.005 is not stated; not tested. +6. **`as_of` and `offset`.** The table gives `as_of` = T − 1 y and the query + text has `offset 1y`. We read `as_of` as the window's end (the result of + the offset at T), not as a second shift. The frontend lowers the offset + from the text. +7. **Stage 3 statistics.** Today the shared-scan variant of Pattern A costs + more than separate scans. The time-shifted scans of q3–q5 are priced as + 4,000,000 samples (one minute of data) instead of years, and a 1-y range + over the shared 5-y scan passes all of its rows. Both are cost-model + estimate defects, so the invariant test is ignored on them. +8. **Identical-expression sharing.** Not mentioned by the doc. The planner + shares the `latency_ms` scan, and the 2-y time shift read by q4 and q5 + (the shift sits below the range). +9. **Additional sliding forms.** L shorter than W (merged) is allowed by the + Pass 2 rules but not listed for Pattern B. Tests allow extra forms. +10. **Latency.** Pattern B's ≤ 200 ms is not checked by any test; Stage 3's + latency model is not part of this example. diff --git a/docs/design_docs/proposals/planner-layering-example4-acceptance.md b/docs/design_docs/proposals/planner-layering-example4-acceptance.md new file mode 100644 index 000000000..1a0840013 --- /dev/null +++ b/docs/design_docs/proposals/planner-layering-example4-acceptance.md @@ -0,0 +1,148 @@ +# Planner layering, Example 4: acceptance spec + +Audience: planner designers and the implementers of Stage 2 materialization +and its pricing in Stage 3. +Source: [planner-layering.md](planner-layering.md), "Example 4: Materialization +of window summaries in physical planning" and the Materialization section. +Tests: `crates/integration-tests/tests/planner_layering_example4.rs` (helpers +and workloads in `tests/planner_layering_common/`). + +This spec was written by a test designer who does not implement the stages. +Tests run `plan_selection::plan_stages`. A test that needs a missing feature +is `#[ignore]`d with the feature's name. + +## Workloads + +Example 3's two patterns ([its spec](planner-layering-example3-acceptance.md)), +with Pattern A varied as the doc does: + +| Variant | Recurrence | Predictability | `arrival` | +|---|---|---|---| +| A, as given | batch, `invocations: 1` at T | `ad_hoc` | `mixed` | +| A, monthly | repeating every 30 days, window ending at each run | `Predictable { known_at: T }` | `mixed` | +| A, at rest | batch, `invocations: 1` at T | `ad_hoc` | `at_rest` (no ingestion rate) | +| B | repeating every 1 min | `Predictable` | `continuously_ingesting` | + +## Expected physical candidates + +Only the physical candidates of the shared logical candidate are specified; +every other logical candidate gets its own the same way (not counted). + +**Pattern A, shared EH of KLLs (read by q1–q5).** + +| Candidate | Materialization of the EH | Expected | +|---|---|---| +| A1 | query time, kept for the batch | generated; selected as given | +| A2 | ingestion time, old data backfilled once | generated unless `at_rest`; gains with monthly recurrence | +| A3 | not materialized: each query rebuilds it | generated; costs more than A1 | + +**Pattern B, 1-min tumbling KLLs.** + +| Candidate | Materialization of the tumbling KLLs | Expected | +|---|---|---| +| B1 | ingestion time, kept 5 min; merge 5 + p99 at query time | generated; usually selected | +| B2 | not materialized: rebuild all 5 from 5 min of raw samples | generated; costs at least B1 and B3 | +| B3 | query time, kept 5 min: build only the newest | generated; no ingestion-time node | + +**Pattern B, sliding window (L = 5 min, s = 1 min).** Ingestion time or query +time, both kept. Not materializing it is the no-window plan, so it is not a +separate option. + +**Today:** no window summaries and no materialization; every node runs at +query time, so none of these candidates exists. + +## Invariants and tests + +**Reading materialization.** The export records only an execution timing per +node. Tests read Stage 2's choice through +`planner_layering_common::materialization(p, node)` (`IngestionTime`, +`QueryTimeKept`, `NotMaterialized`; today it maps ingestion time to +`IngestionTime` and query time to `NotMaterialized`) and the kept event-time +span through `retention_ms(p, node)` (`None` today). The materialization +implementer makes both read the new IR. + +### Constraints over every candidate + +| Invariant | Test | Status | +|---|---|---| +| Every node upstream of an ingestion-time node also runs at ingestion time (all variants) | `stage2_ingestion_time_upstream_is_ingestion_time` | passes (vacuously today) | +| With data at rest, no node runs at ingestion time | `stage2_at_rest_runs_nothing_at_ingestion_time` | passes (vacuously today) | +| A materialized output covers its consumers: retention ≥ max over its readers of `lookback` + offset | `stage2_materialized_output_covers_its_consumers` | passes (vacuously today) | + +### Pattern A + +| Invariant | Test | Status | +|---|---|---| +| The shared EH yields exactly A1, A2, A3 | `stage2_a_shared_eh_has_three_materialization_options` | ignored: window composition and materialization | +| At rest, A2 is not generated; A1 and A3 remain | `stage2_a_at_rest_drops_the_ingestion_time_option` | ignored: same | +| A materialized EH is one build node read by all five queries, priced once | `stage2_a_materialized_eh_is_built_once_for_all_consumers` | ignored: same | +| A3 costs strictly more than A1 (five builds instead of one) | `stage3_a_rebuilding_per_query_costs_more_than_building_once` | ignored: same | +| As given, A1 is the cheapest of the three | `stage3_a_once_adhoc_prefers_the_query_time_eh` | ignored: same | +| Monthly, A2's cost relative to A1 drops: cost(A2)/cost(A1) monthly < as given | `stage3_a_monthly_amortizes_ingestion_time_maintenance` | ignored: same, and recurrence in `plan_stages` | + +### Pattern B + +| Invariant | Test | Status | +|---|---|---| +| The tumbling KLL candidate yields exactly B1, B2, B3 | `stage2_b_tumbling_kll_has_three_materialization_options` | ignored: window composition and materialization | +| B1 builds at ingestion time; its merge and estimate run at query time | `stage2_b_b1_builds_at_ingestion_and_merges_at_query_time` | ignored: same | +| B3 runs nothing at ingestion time | `stage2_b_b3_keeps_query_time_windows` | ignored: same | +| The sliding KLL yields exactly ingestion time and query time, kept | `stage2_b_sliding_kll_has_two_materialization_options` | ignored: same | +| B2 costs at least B1 and B3 | `stage3_b_rebuilding_every_window_costs_most` | ignored: same | +| With the built-in models, B1 is the cheapest of the three | `stage3_b_prefers_ingestion_time_tumbling_windows` | ignored: same | + +### Selection direction, relative to the cost model + +* A3 > A1 and B2 ≥ B1, B3 hold under any model that charges each build: A3 + and B2 do strictly more of the same work. +* A1 ≤ A2 as given, and B1 ≤ B3, depend on the model. The doc's reasons are + that A2 maintains years of history for one batch, and that B3 adds the + newest build to every evaluation and needs raw data at query time. The + tests assert the doc's direction under the built-in models; if the model + prices an ingestion-time update above a query-time one, B3 may win and the + doc allows it ("B3 can win when ingestion-time work is expensive"). +* Monthly recurrence is tested as a ratio, so it does not depend on whether A2 + wins outright ("A2 can win"). + +## Ambiguities and conflicts with the planner + +1. **Charging A3.** Stage 3 charges each node once, which is right for a + shared, materialized summary. A3 is one logical EH node that is not + materialized, so it is rebuilt by each of its five consumers. The tests + require A3 to cost more than A1. Either Stage 3 charges a non-materialized + node once per consuming query, or Stage 2 duplicates it per consumer + (which makes A3 look like five independent EHs). Decision needed. +2. **Recurrence input.** Materialization depends on `recurrence`, + `predictability` and `arrival`, but `plan_stages` takes only the queries, + accuracy targets and data workload. It needs the query workload (or the + per-entry recurrence) for A2's amortization and for B1. +3. **Monthly variant.** The doc does not say whether each monthly run still + ends at T or at its own run time, nor give `known_at`. Tests use a + repeating entry whose window ends at each run, known at T. +4. **Retention unit.** "Kept 5 min" is read as the event-time span the stored + state covers. For A1, which lives only for the batch, that is 5 y of + history. The test requires retention ≥ each reader's `lookback` + offset. +5. **Backfill.** A2 backfills old data once. Whether that one-time cost is + charged to the plan, and how it is amortized, is not stated; not tested. +6. **Pruning A2 at rest.** "A2 is not generated" is read as Stage 2 pruning + (no candidate), not a Stage 3 rejection. +7. **Upstream of an ingestion-time node.** The constraint covers the scan and + range too, so an ingestion-time build's whole input chain must be marked + ingestion time in the export. +8. **Other logical candidates.** "Every other logical candidate gets its own + physical candidates the same way" gives no counts; tests count only the + shared candidate's options. +9. **B3 retention.** The doc says B3 keeps 5 min but uses "4 kept" windows + plus the newest. The test asks for 5 min (the query window). + +## Devtool + +No `--example planner-layering-4`. Pattern A and B already run as +`planner-layering-3a` and `3b`, and until Stage 2 materializes something, +Example 4's variants plan identically. A useful id needs: + +* Pass 2 window composition and Stage 2 materialization; +* the workload's recurrence passed to `plan_stages`; +* the variant (as given, monthly, at rest) selectable in `stage_pipeline`; +* the viewer showing each node's materialization and retention (today only + `output_state.timing` is exported). diff --git a/docs/design_docs/proposals/planner-layering.md b/docs/design_docs/proposals/planner-layering.md index 5273bfb3f..d9f56a38c 100644 --- a/docs/design_docs/proposals/planner-layering.md +++ b/docs/design_docs/proposals/planner-layering.md @@ -56,7 +56,7 @@ In order to achieve the goals, ASAPPlanner needs to abstract the modeling the fo | Deployment inputs | Empirical cost model, empirical accuracy model and execution capabilities | | ASAP replacement strategies | Rules that replace a sub-DAG of the query expression with summary expressions, and the summary families each computation may use | -ASAPPlanner takes a [query workload](https://github.com/ProjectASAP/ASAPPlanner/blob/main/crates/types/src/workload.rs), a [data workload](https://github.com/ProjectASAP/ASAPPlanner/blob/main/crates/types/src/workload.rs#L531) and the deployment's +ASAPPlanner takes a [query workload](https://github.com/ProjectASAP/ASAPPlanner/blob/main/crates/types/src/workload/mod.rs), a [data workload](https://github.com/ProjectASAP/ASAPPlanner/blob/main/crates/types/src/workload/mod.rs#L534) and the deployment's inputs (TODO: define this data structure, [#525](https://github.com/ProjectASAP/ASAPPlanner/issues/525)), and returns one optimal physical plan. It decides what is computed, how it is computed, and which plan is best. The deployment only supplies inputs and executes the plan: it provides its empirical cost model, empirical accuracy @@ -254,7 +254,7 @@ A summary-based candidate uses three kinds of summary nodes: summaries into a coarser one. * A **summary estimation node** computes an answer from a summary, for example the p99 estimate from a KLL, or the entropy estimate from a UnivMon. -* **summary subtract node** and **summary delete node** design is TODO. +* **summary subtract node** and **summary delete node** design is TODO. One summary build node can feed several estimation nodes, which is what Pass 2 exploits. @@ -439,7 +439,7 @@ given: it does not choose among summaries or decide what to materialize. Each example's workload is shown as tables. Field names in code font are the fields of -[`workload.rs`](https://github.com/ProjectASAP/ASAPPlanner/blob/main/crates/types/src/workload.rs). +[`workload/mod.rs`](https://github.com/ProjectASAP/ASAPPlanner/blob/main/crates/types/src/workload/mod.rs). Each query reads the event-time window [`as_of` − `lookback`, `as_of`] (fields of `TimeSelection`). `as_of` is the window's end; `lookback` is its length. An `as_of` of "evaluation time" means `as_of: None`: the window ends whenever the diff --git a/docs/develop_docs/asap-aware-mapping-architecture.md b/docs/develop_docs/asap-aware-mapping-architecture.md index aee8fe53a..5e400f194 100644 --- a/docs/develop_docs/asap-aware-mapping-architecture.md +++ b/docs/develop_docs/asap-aware-mapping-architecture.md @@ -16,7 +16,8 @@ defined in [mapping contracts](asap-aware-mapping-contracts.md). Names such as `MyStrategy`, `MyCostModel`, and `PreferDDSketch` are illustrative; they do not ship with this crate. Samples that use real public -types and functions follow the APIs exported by `asap-aware-mapping`. +types and functions follow the APIs exported by `asap-logical-optimizer` +(Stage 1 candidate search) and `asap-plan-selection` (cost models and selection). If you only need to find the right extension point, start with the [extension map](extend-asap-aware-mapping.md#7-current-extension-map). If you are implementing a strategy, read this mental model, the [mapping contracts](asap-aware-mapping-contracts.md), and the [extension guide](extend-asap-aware-mapping.md). @@ -42,7 +43,9 @@ A strategy should answer: A cost model should answer: -> Given valid choices, which choices are preferable, and how should they be parameterized? +> Given valid choices, which choices are preferable? + +It is consulted only at selection time; sketch parameters come from the analytical estimators. Do not put cost-based pruning into a `ReplacementStrategy`. A strategy must enumerate every valid alternative, even when the default cost model clearly prefers one. See [Rule 2](extend-asap-aware-mapping.md#rule-2-enumerate-do-not-rank). @@ -55,18 +58,20 @@ The diagram below follows a workload of one or more query roots through target d Terminology used in the diagram: - A **workload** is the set of named queries planned together. A **query root** - is the top-level `QueryExpr` (the logical query-expression type) for one of - those queries. **Pre-ASAP** means this logical input form, before the planner - realizes an operation as a concrete ASAP realization; **post-ASAP** means - the resulting realization form. + is the top-level `Rc` (the unified operator IR) for one of + those queries. **Pre-ASAP** means a DAG that contains only ordinary + `NonASAPOp` operators, before the planner realizes an operation with ASAP + primitives; **post-ASAP** means the same IR after some nodes became `ASAPOp` + summary operators. - A **DAG** (directed acyclic graph) represents query operators whose sub-DAGs may be shared. See [sub-DAG sharing and ASAP-aware CSE](../design_docs/proposals/planner-layering.md#pass-2-asap-aware-common-subexpression-elimination) for the sharing rules. Rust's `Rc` (reference-counted pointer) records shared node identity. - A **target** is one replaceable site. A **candidate** is one valid alternative - for it. `Replacement::Summary` is a constructed post-ASAP summary—maintained state - such as an exact accumulator or an approximate sketch—while - `Replacement::Rewrite` is another pre-ASAP logical expression. + for it. `Replacement::SubDAG` is a replacement sub-DAG: either a constructed + post-ASAP summary (it contains an `ASAPOp`, e.g. an exact accumulator or an + approximate sketch) or a logical rewrite with no ASAP operator + (`is_logical_rewrite` tells them apart). `Replacement::ExactComposition` refers to a child target whose realization must remain undecided until compatible selection. A **sketch** is a compact data structure that trades exactness for bounded error. A @@ -86,18 +91,16 @@ flowchart TB classDef report fill:#f2eafe,stroke:#7950b3,color:#34204f subgraph DISCOVERY[1. Discover every replaceable site] - WL["Input workload
one or more named pre-ASAP QueryExpr roots"]:::input + WL["Input workload
one or more named pre-ASAP OperatorNode roots"]:::input SEARCH["search_workload_with
run CSE once, then visit every node in every root DAG"]:::generate - TARGET["TargetSubDAG
one candidate site plus the number of workload locations
that reference the same Rc<QueryExpr>"]:::generate + TARGET["TargetSubDAG
one candidate site plus the number of workload locations
that reference the same Rc<OperatorNode>"]:::generate WL -->|"roots"| SEARCH -->|"one target per distinct node"| TARGET end subgraph GENERATION[2. Generate all legal alternatives at each site] STRATEGY["ReplacementStrategy
when a target matches, enumerate every legal replacement;
implementations generate but do not choose"]:::generate - CAND["ReplacementSubDAG candidates
each contains a Summary, Rewrite or ExactComposition
plus typed provenance and rationale;
no alternative is removed solely on cost"]:::store + CAND["ReplacementSubDAG candidates
each contains a Subtree (summary or logical rewrite) or ExactComposition
plus typed provenance and rationale;
no alternative is removed solely on cost"]:::store TARGET -->|"try every registered strategy"| STRATEGY --> CAND - CM(["CostModel
orders candidates and supplies
deployment-specific parameters"]):::choose - CM -. "rank and parameterize; accuracy checks remain required" .-> STRATEGY end subgraph SEARCHSPACE[3. Store the workload-wide search space] @@ -106,7 +109,9 @@ flowchart TB end subgraph RANKING[Optional ranked view] - SORT["CandidateLogicalASAPDAGs::cost_sorted
use the CostModel to order each candidate set
and cost every candidate"]:::choose + CM(["CostModel
selection-time preferences and costs"]):::choose + SORT["candidate_selection::cost_sorted
use the CostModel to order each candidate set
and cost every candidate"]:::choose + CM -.-> SORT RANKED["RankedTargetSubDAGCandidates
the same candidates in preferred order,
with costs aligned by index"]:::choose SPACE --> SORT -->|"reorder only; preserve every candidate"| RANKED end @@ -157,7 +162,7 @@ flowchart LR classDef workload fill:#e7f7ef,stroke:#31835e,color:#173f2d classDef common fill:#fff6dd,stroke:#b78922,color:#513d0c - ROOTS["Input
one or more named QueryExpr roots"]:::workload + ROOTS["Input
one or more named OperatorNode roots"]:::workload ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable sub-DAGs"]:::workload CSE --> WALK["Discover sites
walk the complete DAG, including nodes below unshared parents"]:::workload WALK --> T["Build TargetSubDAG
retain the sub-DAG's Rc identity and measured consumer_count"]:::workload @@ -191,20 +196,19 @@ capability and accuracy checks; supported algorithm applicability alone is not a result certificate. The complete `replacements()` result is the candidate set produced by one -strategy for one target. A strategy may order or parameterize candidates with -help from a `CostModel`, but it must not remove a valid candidate because of -cost. +strategy for one target. Strategies take no `CostModel` and must not remove a +valid candidate because of cost; ranking happens at selection time. ### 3.3 Current concrete strategies The default context-free registry contains five `ReplacementStrategy` implementations: -- `SketchAlgorithmStrategy` matches supported aggregate and binary shapes. Its - `replacements(target)` method constructs every legal post-ASAP `SummaryNode`, +- `ASAPStrategies` matches supported aggregate and binary shapes. Its + `replacements(target)` method constructs every legal post-ASAP summary sub-DAG, including applicable sketch, exact-accumulator, and pass-through - realizations. Candidates are sized and ordered for the target's accuracy - requirement; candidates without a sufficient guarantee are rejected before - costing. + realizations. Candidates are sized analytically for the target's accuracy + requirement and listed in `summary_candidates` order; candidates without a + sufficient guarantee are rejected before costing. - `SharedSubDAGStrategy` uses `consumer_count` to identify shared targets. It emits both build-once-and-share and recompute-independently rewrites when a target has multiple consumers. @@ -213,11 +217,11 @@ The default context-free registry contains five `ReplacementStrategy` implementa - `ExactCompositionStrategy` preserves child-target references for compatible composition selection. -`default_strategies_with` uses `SemanticEquivalentRewriteStrategy` in its rewrite -slot. The evidence-aware registry supplies the accuracy evidence provider to +`default_strategies` uses `SemanticEquivalentRewriteStrategy` (via its +`AvgToSumOverCountStrategy` alias) in its rewrite slot. The evidence-aware registry supplies the accuracy evidence provider to summary and Hydra construction. Search derives `RollupStrategy` after CSE from the actual sibling set. See the -[registry definitions](../../crates/asap-aware-mapping/src/replacement.rs). +[registry definitions](../../crates/logical-optimizer/src/pass1/replacement.rs). The important rule is: @@ -235,7 +239,7 @@ rejection reasons. This compact representation preserves independent choices without enumerating a flat list of `2^N` complete plans for `N` replaceable targets. -`CandidateLogicalASAPDAGs::cost_sorted` ranks each target's existing candidates with the +`candidate_selection::cost_sorted` ranks each target's existing candidates with the supplied `CostModel`. It returns the same candidates in preferred order, with costs aligned by index; ranking does not select or remove a candidate. @@ -249,14 +253,15 @@ matter. A single-target inspection caller may take the first candidate with `.into_iter().next()` and handle the empty case according to its -execution policy. Constructing all candidates before taking the first costs -more than constructing only the preferred candidate, but it keeps the strategy +execution policy; the first candidate is in `summary_candidates` order, not cost +order. Constructing all candidates before taking the first costs more than +constructing only one, but it keeps the strategy contract consistent and preserves the full choice set for other callers. -`CandidateLogicalASAPDAGs::global_selection` optionally coordinates cross-target sharing and +`candidate_selection::global_selection` optionally coordinates cross-target sharing and composition choices. `GlobalSelection::assemble_selected_dag` constructs the selected -semantic DAG. These plain APIs do not establish lifecycle or physical deployment -feasibility. Recurrence and lifecycle-aware variants require the corresponding +semantic DAG. These APIs do not decide materialization or establish physical +deployment feasibility. Recurrence-aware variants require the corresponding workload and evidence inputs; downstream owns physical commitment and execution. See the [library workflow](library-api.md#optional-whole-plan-selection-and-dag-assembly). diff --git a/docs/develop_docs/asap-aware-mapping-contracts.md b/docs/develop_docs/asap-aware-mapping-contracts.md index 447cb3823..f519d573e 100644 --- a/docs/develop_docs/asap-aware-mapping-contracts.md +++ b/docs/develop_docs/asap-aware-mapping-contracts.md @@ -10,25 +10,25 @@ first; use the [extension guide](extend-asap-aware-mapping.md) when changing one ### `TargetSubDAG` -A pre-ASAP `QueryExpr` node that a strategy may replace. +A pre-ASAP `OperatorNode` that a strategy may replace. ```rust pub struct TargetSubDAG<'a> { - pub root: &'a Rc, + pub root: &'a Rc, pub consumer_count: usize, } ``` -`root` is the actual `Rc` from the workload. +`root` is the actual `Rc` from the workload. -`consumer_count` counts structural references, not runtime executions. It is the number of places in the workload DAG that point to this exact `Rc` node. +`consumer_count` counts structural references, not runtime executions. It is the number of places in the workload DAG that point to this exact `Rc` node. For example, consider two top-level queries: - `sum by (service) (rate(m[5m]))` - `avg by (service) (rate(m[5m]))` -After `share_common_sub_dags` merges their identical `rate(m[5m])` sub-DAGs, both query DAGs point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. +After `share_common_sub_dags` merges their identical `rate(m[5m])` sub-DAGs, both query trees point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. Use: @@ -54,33 +54,38 @@ when the caller already knows the real number of consumers. The actual object that substitutes the target. -There are currently three forms: +There are currently two forms: ```rust pub enum Replacement { - Summary(Rc), - Rewrite(Rc), + SubDAG(Rc), ExactComposition(ExactComposition), } ``` -Use `Replacement::Summary` when the alternative is a constructed post-ASAP summary plan. +Use `Replacement::SubDAG` for a replacement sub-DAG. It is one of: -Use `Replacement::Rewrite` when the alternative is still a logical pre-ASAP `QueryExpr`. +- a constructed post-ASAP summary plan: the sub-DAG contains an `ASAPOp` + (`SummaryAgg`, `SummaryEstimate`, ...); +- a logical rewrite: only `NonASAPOp` nodes and no guarantee yet. + +`is_logical_rewrite(&node)` tells the two apart. A kept pre-ASAP sub-DAG +(`retain_exact`) has no ASAP operator but carries an exact guarantee, so it +counts as a bound decision, not a rewrite. Use `Replacement::ExactComposition` when an exact operation refers to a child target whose realization must remain undecided. Selection coordinates the parent/child pair; DAG assembly constructs and validates the composed DAG. -See [exact_composition.rs](../../crates/asap-aware-mapping/src/exact_composition.rs). +See [exact_composition.rs](../../crates/logical-optimizer/src/pass1/exact_composition.rs). Examples: ```text Quantile(...) - -> KLL SummaryNode + -> SummaryEstimate(SummaryAgg(KLL)) ``` -is a `Summary`; KLL (Karnin–Lang–Liberty) is a quantile-sketch algorithm. +is a summary `Subtree`; KLL (Karnin–Lang–Liberty) is a quantile-sketch algorithm. ```text compute independently @@ -88,7 +93,7 @@ compute independently reuse an already shared logical sub-DAG ``` -is represented as a `Rewrite`. +is represented as two logical-rewrite `Subtree`s. --- @@ -142,9 +147,9 @@ Search calls `propose`, so rejected candidates remain available for explanation. > What are all semantically valid alternatives for this target? -`replacements` must be **exhaustive and not cost-filtered**. When its output has -a preferred order, that ordering must come from the supplied `CostModel`; the -strategy must still return every supported legal candidate. Required accuracy, +`replacements` must be **exhaustive and not cost-filtered**. Strategies take no +`CostModel`; output order carries no cost preference, and ranking happens at +selection time. The strategy must return every supported legal candidate. Required accuracy, schema and capability checks can reject an otherwise applicable algorithm; exhaustiveness is not a promise of all theoretically possible plans. @@ -157,9 +162,11 @@ aggregation must compute without committing to a physical summary algorithm. A realization may be an approximate sketch, an exact mergeable accumulator, or a pass-through that keeps the original operation instead of building a summary. `realizations_for_intent` enumerates these concrete -realizations; `SketchAlgorithmStrategy::replacements()` constructs each one as +realizations; `ASAPStrategies::replacements()` constructs each one as a `ReplacementSubDAG`. It returns all -candidates in preferred order without selecting a winner. At workload scale, +candidates in `summary_candidates` order, sized by the analytical estimators, +without selecting a winner; `AggIntent::Extension` intents stay +`Realization::PassThrough`. At workload scale, `search_workload`/`search_workload_with` preserve all supported legal alternatives across every `TargetSubDAG`. Optional planner APIs coordinate compatible semantic selections; physical commitment and placement remain downstream deployment decisions. @@ -170,8 +177,8 @@ This guide uses the Cascades/Volcano terminology: realization. For example, a quantile `AggIntent` may have KLL and DDSketch `Realization` values. - A **transformation rule** maps a logical operation to another logical - operation. In this crate, that kind of candidate is represented by - `Replacement::Rewrite`. + operation. In this crate, that kind of candidate is a logical-rewrite + `Replacement::SubDAG`. - A **replacement candidate** packages either kind of result as a `ReplacementSubDAG` for search. `CandidateLogicalASAPDAGs` stores and ranks these candidates. - **Physical commitment and placement** happen downstream. An `Realization` @@ -183,7 +190,7 @@ The concrete flow is: ```text AggIntent -> realizations_for_intent(): enumerate Realization values - -> SketchAlgorithmStrategy: construct ReplacementSubDAG candidates + -> ASAPStrategies: construct ReplacementSubDAG candidates -> CandidateLogicalASAPDAGs: store and rank candidates -> downstream deployment: select and place a final choice ``` @@ -196,17 +203,14 @@ see [code architecture §3](asap-aware-mapping-architecture.md#3-how-the-current ### `CostModel` -`CostModel` covers every deployment-specific numeric or configuration decision—not only which candidate is cheapest. For example, sketch sizing trades memory and update cost for accuracy, so it belongs here too. +`CostModel` covers deployment-specific preference and cost decisions. It is consulted only at selection time (`cost_sorted`, `global_selection` and their `_with_recurrence` variants), never during candidate generation: sketch parameters come from the analytical estimators (`accuracy::estimators::size_params`), and extension intents stay pass-through. -The crate cannot hardcode real deployment costs: `asap-aware-mapping` uses `asap-types` and pinned `asap_sketchlib` mapping +The crate cannot hardcode real deployment costs: `asap-plan-selection` uses `asap-types` and the `asap_sketchlib` mapping bounds, but does not execute workloads or own deployment measurements. Most hooks therefore provide the crate's built-in static behavior as a default. Override only the decisions your deployment needs to change. | Hook | Use it to | Default? | |---|---|---| | `rank_candidates` | Order valid sketch algorithms | No | -| `size_params` | Convert an accuracy target into sketch parameters | Yes | -| `realize_extension` | Map a custom intent to a realization | Yes | -| `readout_extension` | Query a custom extension summary | Panics until paired with a custom realization | | `cse_recompute_cost` | Estimate independent recomputation | Yes | | `cse_shared_maintenance_cost` | Estimate shared maintenance | Yes | | `cse_share_decision` | Choose sharing or recomputation | Yes | @@ -218,41 +222,6 @@ bounds, but does not execute workloads or own deployment measurements. Most hook fn rank_candidates(&self, intent: &AggIntent, candidates: &[SketchAlgorithm]) -> Vec; ``` -- **`size_params`** — choose parameters, such as sketch capacity, for an already-selected `SketchAlgorithm` and accuracy target `(eps, delta)`, where `eps` is the tolerated error and `delta` is the tolerated probability of exceeding that error. It is separate from ranking so a deployment can customize sizing without changing algorithm preference. The trait provides a default implementation. - - ```rust - fn size_params(&self, kind: SketchAlgorithm, intent: &AggIntent, eps: f64, delta: f64) -> SketchParams; - ``` - -- **`realize_extension`** — map a deployment-defined `AggIntent::Extension` to a post-ASAP `Realization`. The default is `Realization::PassThrough`. - - Use `AggIntent::Extension { ext_kind, payload }` for intent shapes that only your deployment needs. Core treats both fields as opaque. For example, a deployment can tag an approximate-frequency intent with `ext_kind: "frequency"` and recognize it in `realize_extension`: - - ```rust - fn realize_extension(&self, ext_kind: &str, _payload: &serde_json::Value) -> Realization { - if ext_kind == "frequency" { - Realization::Sketch(SketchKind::new( - SketchAlgorithm::CountSketch, - SketchParams::CountSketch { width: 1024, depth: 5 }, - )) - } else { - Realization::PassThrough // fall back to the default for anything else - } - } - ``` - - Return `Realization::PassThrough` for unrecognized extension kinds. Do not panic. - - ```rust - fn realize_extension(&self, ext_kind: &str, payload: &serde_json::Value) -> Realization; - ``` - -- **`readout_extension`** — define how queries read an extension summary that `realize_extension` mapped to a `Sketch`. The two hooks are a pair: realization defines what is maintained; readout defines how it is queried. Override both for the same `ext_kind`. The default readout panics to prevent a silent wrong answer. - - ```rust - fn readout_extension(&self, ext_kind: &str, payload: &serde_json::Value, col: &ColumnRef) -> SketchStatistic; - ``` - - **`cse_recompute_cost`** — estimate the one-time cost of recomputing a CSE candidate's sub-DAG independently at a single consumer. Default: `default_cse_recompute_cost`, a structural-size proxy. ```rust @@ -273,7 +242,7 @@ bounds, but does not execute workloads or own deployment measurements. Most hook fn cse_share_decision(&self, candidate: &CseCandidate) -> ShareDecision; ``` -- **`estimate_cost`** — attach a comparable numeric cost to an already-constructed replacement. `CandidateLogicalASAPDAGs::cost_sorted` calls it for every candidate and keeps the returned values aligned with the ranked candidates. The trait default returns `f64::NAN` deliberately; override it when a custom model's callers need displayable or otherwise consumable numeric costs. `DefaultCostModel` provides real values derived from its CSE cost hooks. +- **`estimate_cost`** — attach a comparable numeric cost to an already-constructed replacement. `candidate_selection::cost_sorted` calls it for every candidate and keeps the returned values aligned with the ranked candidates. The trait default returns `f64::NAN` deliberately; override it when a custom model's callers need displayable or otherwise consumable numeric costs. `DefaultCostModel` provides real values derived from its CSE cost hooks. ```rust fn estimate_cost( @@ -292,19 +261,20 @@ A custom cost model does not necessarily need to override every hook. The curren `ReplacementStrategy` answers "what are the candidates for this one target?" `CandidateLogicalASAPDAGs` answers the same question for every target in a whole workload at once, without enumerating `2^N` fully-copied plans for `N` independently-choosable sites. ```rust -// replacement.rs +// replacement.rs (TargetSubDAGCandidates) and +// plan_selection/candidate_selection.rs (RankedTargetSubDAGCandidates) // One TargetSubDAGCandidates per distinct TargetSubDAG in the whole workload — // never a flat list of fully assembled plans. pub struct TargetSubDAGCandidates { - pub target: Rc, + pub target: Rc, pub consumer_count: usize, pub candidates: Vec, // accepted alternatives, unranked pub rejected: Vec, // failed accuracy checks } pub struct RankedTargetSubDAGCandidates<'a> { - pub target: &'a Rc, + pub target: &'a Rc, pub consumer_count: usize, pub candidates: Vec<&'a ReplacementSubDAG>, // same candidates, ranked pub costs: Vec, // costs[i] <-> candidates[i] @@ -313,7 +283,7 @@ pub struct RankedTargetSubDAGCandidates<'a> { `search_workload(roots)` runs the shared-sub-DAG pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubDAGStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. -`CandidateLogicalASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubDAGStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks +`candidate_selection::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — the `SharedSubDAGStrategy` share/recompute pair (recognized by `ReplacementProvenance::CseShare`/`CseRecompute`) goes through `CostModel::cse_share_decision`; a set with a Hydra shared-grid alternative goes through `CostModel::grouping_state_cost`; a set whose candidates all realize sketches (a `ASAPStrategies` choice) goes through `CostModel::rank_candidates`; and any other mixed set is ordered by `CostModel::candidate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks may already have removed proposals before this boundary. In particular, `search_workload_with_targets` checks explicit per-root targets, while retaining direct DDSketch ratios with missing domain evidence and no root guarantee for @@ -329,7 +299,7 @@ Sketches separate their query category from the concrete algorithm and its param | Level | Type | Example | | --- | --- | --- | -| **family** | `SummaryFamilyType` | `Sketch`, `Sample`, `Wavelet`, `StatModel`, `ExactAggregate` | +| **family** | `FieldDataType` (non-`Plain` variants) | `Sketch`, `Sample`, `Wavelet`, `StatModel`, `ExactAggregate` | | **category** | `SketchCategory` | `Quantile`, `Cardinality`, `Frequency`, `TopK` | | **algorithm** | `SketchAlgorithm` | `Kll` / `DDSketch` (both quantile); `Hll` (HyperLogLog) / `Theta` / `Kmv` (K-Minimum Values), all cardinality | | **committed choice** | `SketchKind` | one validated category + algorithm + parameter combination | @@ -340,7 +310,7 @@ to the selected algorithm and classifies the pair into its category. The public `.category()`, `.algorithm()`, and `.params()` accessors expose the committed values without permitting an invalid combination. -Where this matters in practice: `CostModel::rank_candidates`, `CostModel::size_params`, and `SketchAlgorithmStrategy::replacements` operate at the **algorithm** level. `summary_candidates(intent)` returns a list of `SketchAlgorithm`s (`[Kll, DDSketch]` for a `Quantile` intent), never a bare `SketchKind` with nothing chosen underneath it. `SketchKind` appears after an algorithm has been selected and sized—on `Realization::Sketch(SketchKind)` and `SummaryFamilyType::Sketch(SketchKind)`. +Where this matters in practice: `CostModel::rank_candidates` and `ASAPStrategies::replacements` operate at the **algorithm** level. `summary_candidates(intent)` returns a list of `SketchAlgorithm`s (`[Kll, DDSketch]` for a `Quantile` intent), never a bare `SketchKind` with nothing chosen underneath it. `SketchKind` appears after an algorithm has been selected and sized—on `Realization::Sketch(SketchKind)` and `FieldDataType::Sketch(SketchKind, GroupingStrategy)`. `Sample`, `Wavelet`, and `StatModel` each use a flat `(Kind, Params)` pair. `Sketch` needs the additional algorithm level because multiple algorithms can serve the same purpose—for example, KLL and DDSketch both answer quantile queries. @@ -378,22 +348,22 @@ The crate provides no default `Matcher` implementation because the answer depend Concretely, `explanation.rs` reports three candidate kinds from each `TargetSubDAGCandidates`: -- `ExplanationKind::SketchApproximation` — the set contains a `Replacement::Summary` that realizes `SummaryFamilyType::Sketch(..)`, not just an exact/pass-through candidate. -- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubDAGStrategy`'s "build once and share" candidate (the `Replacement::Rewrite` whose `Rc` is the set's `target`). +- `ExplanationKind::SketchApproximation` — the set contains a summary `Replacement::SubDAG` that realizes `FieldDataType::Sketch(..)`, not just an exact/pass-through candidate. +- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubDAGStrategy`'s "build once and share" candidate (the `Replacement::SubDAG` whose `Rc` is the set's `target`). - `ExplanationKind::ExactComposition` — the candidate set contains an exact operation composed with a child target whose realization remains a coordinated choice. Each `ReplacementExplanation::reason` is copied verbatim from the matching candidate's own `ReplacementSubDAG::rationale`. Nothing in `explanation.rs` re-explains why a candidate is valid; that explanation already exists exactly once, on the candidate itself. -`ReplacementExplanation` carries both `node_hash` and `target`. A downstream consumer first compares `node_hash` with an exported `DAGNode::hash` to narrow the search, then compares the exact target expression with the node's in-process source expression. This preserves the hash's role as a fast filter while making the final association collision-safe; `location` remains human-readable presentation text rather than a machine identifier. +`ReplacementExplanation` carries both `node_hash` and `target`. A downstream consumer first compares `node_hash` with an exported `DAGNode::hash` to narrow the search, then compares the exact `target` node with the exported node's in-process `DAGNode::source_node`. This preserves the hash's role as a fast filter while making the final association collision-safe; `location` remains human-readable presentation text rather than a machine identifier. ### Why there is no `ExplanationRule` trait -Explanations are derived from candidates already present in `CandidateLogicalASAPDAGs`. A new candidate kind therefore requires an `impl ReplacementStrategy` wired into `default_strategies`/`default_strategies_with`; a second explanation-specific trait would duplicate registration and could drift from the actual search space. Custom callers supply strategies through `explain_replacements_with`, using the same extension point exposed by `search_workload_with`. +Explanations are derived from candidates already present in `CandidateLogicalASAPDAGs`. A new candidate kind therefore requires an `impl ReplacementStrategy` wired into `default_strategies`; a second explanation-specific trait would duplicate registration and could drift from the actual search space. Custom callers supply strategies through `explain_replacements_with`, using the same extension point exposed by `search_workload_with`. ### How it derives `location` text -`CandidateLogicalASAPDAGs`/`TargetSubDAGCandidates` track `Rc` pointer identity, not human-readable breadcrumbs. `ReplacementExplanation::location` provides prose such as `root "dash_a" > lhs` so reporting consumers can identify the relevant part of the query without interpreting pointer identity. Location derivation does not make replacement or costing decisions. +`CandidateLogicalASAPDAGs`/`TargetSubDAGCandidates` track `Rc` pointer identity, not human-readable breadcrumbs. `ReplacementExplanation::location` provides prose such as `root "dash_a" > lhs` so reporting consumers can identify the relevant part of the query without interpreting pointer identity. Location derivation does not make replacement or costing decisions. --- diff --git a/docs/develop_docs/end-to-end-accuracy-guarantees.md b/docs/develop_docs/end-to-end-accuracy-guarantees.md index 94c69d988..1baf0a60c 100644 --- a/docs/develop_docs/end-to-end-accuracy-guarantees.md +++ b/docs/develop_docs/end-to-end-accuracy-guarantees.md @@ -40,10 +40,10 @@ The main implementation locations are: | Concern | Location | | --- | --- | -| Guarantee and error vocabulary | `asap_types::post_asap::guarantee` | -| Accuracy model and built-in propagation | `asap_aware_mapping::accuracy` | -| Candidate construction and legality filtering | `asap_aware_mapping::replacement` | -| Parameter sizing hooks | `asap_aware_mapping::cost_model` | +| Guarantee and error vocabulary | `asap_types::ir::properties::guarantee` | +| Accuracy model and built-in propagation | `asap_logical_optimizer::accuracy` | +| Candidate construction and legality filtering | `asap_logical_optimizer::pass1::replacement` | +| Parameter sizing | `asap_logical_optimizer::accuracy::estimators` | | Guarantee and rejection export | `asap_types::dag_export` and the `dag_export` devtool | Read the sections below when changing one of those contracts. diff --git a/docs/develop_docs/extend-asap-aware-mapping.md b/docs/develop_docs/extend-asap-aware-mapping.md index 5a5e069a8..6432ad768 100644 --- a/docs/develop_docs/extend-asap-aware-mapping.md +++ b/docs/develop_docs/extend-asap-aware-mapping.md @@ -44,7 +44,7 @@ There are four decisions to make. `matches` should contain the minimum structural and semantic checks needed to determine whether the strategy applies. -For example, the aggregate path in `SketchAlgorithmStrategy` requires a +For example, the aggregate path in `ASAPStrategies` requires a supported shape: - the node is an `Aggregate`, @@ -112,23 +112,20 @@ and let costing decide later. --- -### Choose `Summary` vs. `Rewrite` +### Summary sub-DAG vs. logical rewrite -Return: +Both are returned as: ```rust -Replacement::Summary(...) +Replacement::SubDAG(node) ``` -when the candidate is a fully constructed post-ASAP summary. +- A fully constructed post-ASAP summary: `node` contains an `ASAPOp`. +- A logical pre-ASAP rewrite: `node` has only `NonASAPOp` nodes and no + guarantee. `is_logical_rewrite(&node)` checks this. -Return: - -```rust -Replacement::Rewrite(...) -``` - -when the candidate is a logical pre-ASAP rewrite. +Set `provenance` to say which one it is (`ReplacementProvenance::SummaryRealization`, +`LogicalRewrite`, ...); selection reads provenance, not the sub-DAG's shape. Use `Replacement::ExactComposition` when a candidate depends on a child target whose implementation must be selected compatibly later. Do not bind it to the @@ -204,7 +201,7 @@ how to realize it, wrap that logic. Do not create a second implementation of the same semantics inside the strategy. -The existing `SketchAlgorithmStrategy` is the model to follow: it reuses +The existing `ASAPStrategies` is the model to follow: it reuses `replacement.rs`'s existing candidate list and summary-construction path. --- @@ -229,24 +226,18 @@ If your transformation requires context not currently represented in `TargetSubD --- -### Example: current `SketchAlgorithmStrategy` +### Example: current `ASAPStrategies` -`SketchAlgorithmStrategy` is the reference implementation for a strategy that +`ASAPStrategies` is the reference implementation for a strategy that produces constructed post-ASAP summaries. Construction: ```rust -let strategy = - SketchAlgorithmStrategy::default_cost_model(); +let strategy = ASAPStrategies::default(); ``` -or with a custom cost model: - -```rust -let model = MyCostModel; // illustrative -let strategy = SketchAlgorithmStrategy::new(&model); -``` +It takes no cost model; a cost model is consumed only at selection time. The strategy matches supported aggregate nodes. @@ -254,15 +245,15 @@ At a high level: ```mermaid flowchart LR - A["Input TargetSubDAG
root is a supported Aggregate"] --> B["SketchAlgorithmStrategy::matches
check whether the target shape can produce summaries"] - B -->|"true"| C["SketchAlgorithmStrategy::replacements
use CostModel preferences and sizing while preserving
every semantically valid realization"] + A["Input TargetSubDAG
root is a supported Aggregate"] --> B["ASAPStrategies::matches
check whether the target shape can produce summaries"] + B -->|"true"| C["ASAPStrategies::replacements
size each summary_candidates entry analytically,
preserving every semantically valid realization"] B -->|"false"| NONE["Empty candidate list"] - C --> F["Output Vec<ReplacementSubDAG>
each entry contains a constructed SummaryNode and rationale;
all candidates retained in preferred order"] + C --> F["Output Vec<ReplacementSubDAG>
each entry contains a constructed summary sub-DAG and rationale;
all candidates retained in summary_candidates order"] ``` For an approximate quantile, both KLL and DDSketch remain candidates when their committed parameters and evidence satisfy the applicable accuracy checks, -even if the cost model prefers one. When only one realization is legal, such as an exact accumulator or pass-through, the strategy returns that single candidate. +regardless of which one a cost model later prefers. When only one realization is legal, such as an exact accumulator or pass-through, the strategy returns that single candidate. --- @@ -271,23 +262,24 @@ even if the cost model prefers one. When only one realization is legal, such as Call the public strategy interface and inspect every returned candidate: ```rust -let strategy = SketchAlgorithmStrategy::new(&cost_model); +let strategy = ASAPStrategies::default(); let candidates = strategy.replacements(&target); for candidate in candidates { match candidate.replacement { - Replacement::Summary(summary) => { - // Inspect or execute this constructed SummaryNode. + Replacement::SubDAG(node) => { + // A constructed summary sub-DAG (`node.is_asap()`), or a kept + // pre-ASAP sub-DAG with an exact guarantee for pass-through. } - Replacement::Rewrite(_) => unreachable!( - "SketchAlgorithmStrategy produces summary candidates" + Replacement::ExactComposition(_) => unreachable!( + "ASAPStrategies produces sub-DAG candidates" ), } } ``` The public contract is the behavior contributors should preserve: every legal -candidate is returned, ordering follows the supplied `CostModel`, each summary +candidate is returned in `summary_candidates` order, each summary is fully constructed, and each candidate carries a useful rationale. Nested aggregate choices remain independent. @@ -310,16 +302,16 @@ and returns two alternatives: 2. Build independently for each consumer. ``` -The shared candidate reuses the same `Rc`: +The shared candidate reuses the same `Rc`: ```rust -Replacement::Rewrite(Rc::clone(target.root)) +Replacement::SubDAG(Rc::clone(target.root)) ``` The independent candidate creates a structurally equal but separately allocated node: ```rust -Replacement::Rewrite( +Replacement::SubDAG( Rc::new((**target.root).clone()) ) ``` @@ -327,7 +319,7 @@ Replacement::Rewrite( This strategy does **not** decide whether sharing is cheaper. That preference belongs to the cost model. -`CandidateLogicalASAPDAGs::cost_sorted` calls `CostModel::cse_share_decision` when it ranks a +`candidate_selection::cost_sorted` calls `CostModel::cse_share_decision` when it ranks a share-versus-recompute candidate pair. The strategy still returns both alternatives because enumeration and ranking are separate steps: @@ -352,8 +344,7 @@ The basic calling pattern is: ```rust let target = TargetSubDAG::new(&root); -let strategy = - SketchAlgorithmStrategy::default_cost_model(); +let strategy = ASAPStrategies::default(); if strategy.matches(&target) { let candidates = @@ -490,15 +481,14 @@ Do not test only the rationale string; test the actual replacement semantics. #### Custom cost model behavior -If a strategy accepts a cost model, verify that a custom model changes the intended costing behavior without changing the exhaustive candidate set. - -The current sketch strategy does exactly this: +Strategies do not take a cost model. Verify instead that a custom model changes +the selection-time ranking without changing the exhaustive candidate set: ```mermaid flowchart LR - INPUT["Legal candidate set
KLL + DDSketch"] --> MODEL["Custom CostModel
prefers DDSketch for this AggIntent"] + INPUT["Strategy output
KLL + DDSketch"] --> MODEL["cost_sorted / global_selection
with a CostModel preferring DDSketch"] MODEL --> ORDER["rank_candidates output
DDSketch first, KLL second"] - ORDER --> RESULT["Strategy output
both candidates remain; only their order changes"] + ORDER --> RESULT["Ranked view
both candidates remain; only their order changes"] ``` That is the expected separation between enumeration and ranking. @@ -540,19 +530,17 @@ impl CostModel for PreferDDSketch { } ``` -Then inject it into code that accepts a `&dyn CostModel`: +Then pass it to the selection-time APIs that accept a `&dyn CostModel`: ```rust let model = PreferDDSketch; -let strategy = - SketchAlgorithmStrategy::new(&model); - -let replacements = - strategy.replacements(&target); +let space = search_workload_with(roots, &default_strategies()); +let ranked = cost_sorted(&space, &model); +let selection = global_selection(&space, &model); ``` -Important: changing `rank_candidates` changes the preferred ordering, but `SketchAlgorithmStrategy` still enumerates every valid sketch candidate. +Important: `rank_candidates` changes only the selection-time ordering; `ASAPStrategies` still enumerates every valid sketch candidate, in `summary_candidates` order, sized analytically. A custom cost model should not change which alternatives are semantically legal. @@ -589,72 +577,7 @@ It must return a permutation of the supplied candidates: every input candidate e --- -#### `size_params` - -Use when the sketch algorithm is already known and you want to choose its parameters from an accuracy target. - -Signature: - -```rust -fn size_params( - &self, - kind: SketchAlgorithm, - intent: &AggIntent, - eps: f64, - delta: f64, -) -> SketchParams; -``` - -Typical uses include: - -- choosing KLL capacity, -- choosing HLL precision, -- selecting sketch-specific error parameters. - -Conceptually: - -```mermaid -flowchart LR - ALG["Chosen SketchAlgorithm
for example, KLL or HLL"] --> SIZE["CostModel::size_params
translate a requested accuracy budget into
algorithm-specific storage parameters"] - INTENT["AggIntent
what the query is computing"] --> SIZE - ACC["Accuracy budget
epsilon and delta"] --> SIZE - SIZE --> PARAMS["SketchParams
for example, KLL capacity or HLL precision"] -``` - ---- - -#### `realize_extension` - -Use for extension-defined implementation kinds. - -```rust -fn realize_extension( - &self, - ext_kind: &str, - payload: &serde_json::Value, -) -> Realization; -``` - -This is the hook for turning an extension description into a concrete `Realization`. - -Use it for implementation families that are intentionally outside the built-in enum dispatch. - ---- - -#### `readout_extension` - -Use when an extension-defined summary also needs custom query/readout behavior. - -```rust -fn readout_extension( - &self, - ext_kind: &str, - payload: &serde_json::Value, - col: &ColumnRef, -) -> SketchStatistic; -``` - -This complements `realize_extension`: realization defines what gets maintained; readout defines how it is queried (see the [CostModel reference](asap-aware-mapping-contracts.md#costmodel)). +Sketch parameters are not a `CostModel` hook: candidates are sized by the analytical estimators (`accuracy::estimators::size_params`, also exposed as `replacement::default_size_params`), and `AggIntent::Extension` intents always stay `Realization::PassThrough`. --- @@ -711,7 +634,7 @@ fn estimate_cost( ) -> f64; ``` -The default returns `f64::NAN`, making the absence of a numeric model explicit. Override this hook when passing the model to `CandidateLogicalASAPDAGs::cost_sorted` if downstream code displays or otherwise consumes the `costs` values. Prefer to derive the result from the same inputs used by `rank_candidates` and the CSE cost hooks so numeric costs do not disagree with relative ordering. +The default returns `f64::NAN`, making the absence of a numeric model explicit. Override this hook when passing the model to `candidate_selection::cost_sorted` if downstream code displays or otherwise consumes the `costs` values. Prefer to derive the result from the same inputs used by `rank_candidates` and the CSE cost hooks so numeric costs do not disagree with relative ordering. --- @@ -740,22 +663,17 @@ Then test integration through a consumer of the cost model. For example: ```rust -let strategy = - SketchAlgorithmStrategy::new(&model); - -let replacements = - strategy.replacements(&target); +let ranked = + cost_sorted(&space, &model); ``` The important assertion is usually not that other valid candidates disappeared. They should not. Instead verify that: -- the model changes ordering or parameters as intended, +- the model changes ordering as intended, - all legal candidates remain available to the replacement layer. -For sizing, test representative accuracy targets and assert the resulting `SketchParams`. - For CSE costing, create a representative `CseCandidate` and test recompute cost, shared-maintenance cost, and the resulting `ShareDecision`. --- @@ -770,17 +688,17 @@ Declare built-in sketch applicability through the public candidate registry: summary_candidates(intent) ``` -`SketchAlgorithmStrategy` consumes this registry through its public `replacements` method. +`ASAPStrategies` consumes this registry through its public `replacements` method. Therefore, when adding a new built-in sketch algorithm, the intended flow is: ```mermaid flowchart LR MAP["1. Declare legality
add the algorithm to summary_candidates
for each AggIntent it can answer"] - MAP --> MODEL["2. Define costing
rank it, derive its SketchParams,
and provide a comparable numeric cost"] - MODEL --> BUILD["3. Define realization behavior
ensure the public strategy output contains a valid SummaryNode
with the correct maintained state and readout"] + MAP --> MODEL["2. Define sizing and costing
derive its SketchParams in the analytical estimators;
rank it and provide a comparable numeric cost"] + MODEL --> BUILD["3. Define realization behavior
ensure the public strategy output contains a valid summary sub-DAG
with the correct maintained state and evaluation"] BUILD --> ACC["4. Certify accuracy
derive from committed parameters;
propagate and check the final target"] - ACC --> ENUM["5. Verify integration
SketchAlgorithmStrategy includes it automatically;
tests confirm enumeration, ordering, sizing, and cost"] + ACC --> ENUM["5. Verify integration
ASAPStrategies includes it automatically;
tests confirm enumeration, ordering, sizing, and cost"] ``` This keeps one source of truth for sketch applicability. Applicability alone @@ -793,9 +711,9 @@ ranking; preserve exact fallback and structured rejection information. See the [accuracy implementation companion](end-to-end-accuracy-guarantees.md) for formulas and evidence requirements. For a new algorithm, also update its -parameter, readout, schema and serialization definitions in `asap-types`. +parameter, evaluation, schema and serialization definitions in `asap-types`. -Do not special-case the new sketch inside `SketchAlgorithmStrategy` unless the strategy itself needs fundamentally new behavior. +Do not special-case the new sketch inside `ASAPStrategies` unless the strategy itself needs fundamentally new behavior. ### Verifying a new sketch algorithm @@ -804,7 +722,7 @@ or malformed evidence, incompatible metrics and unsupported composition. Test root-target checking before cost ranking, exact fallback, and exported rejection or guarantee data. A cheaper estimate must never admit an accuracy-illegal plan. -After wiring the new algorithm into `summary_candidates` and giving the cost model a real `rank_candidates`/`size_params` opinion about it, check two things. First, that `SketchAlgorithmStrategy::replacements()` for a matching `TargetSubDAG` actually includes a candidate realizing the new algorithm — extend a test shaped like `replacement.rs`'s own test-module coverage-matrix tests (e.g. `agg_intent_to_summary_kind_coverage_matrix`) to cover the new algorithm's `AggIntent`. Second, that `cost_sorted`/`estimate_cost` produce sane, comparable numbers for the new candidate rather than a `NaN` placeholder or an outlier that swamps every other candidate. +After wiring the new algorithm into `summary_candidates` and giving the analytical estimators a sizing rule and the cost model a real `rank_candidates` opinion about it, check two things. First, that `ASAPStrategies::replacements()` for a matching `TargetSubDAG` actually includes a candidate realizing the new algorithm — extend a test shaped like `replacement.rs`'s own test-module coverage-matrix tests (e.g. `agg_intent_to_summary_kind_coverage_matrix`) to cover the new algorithm's `AggIntent`. Second, that `cost_sorted`/`estimate_cost` produce sane, comparable numbers for the new candidate rather than a `NaN` placeholder or an outlier that swamps every other candidate. --- @@ -895,10 +813,11 @@ silently disagree. ### Mistake: reimplementing summary construction inside a strategy -If the candidate should produce a normal `SummaryNode`, use the existing +If the candidate should produce a normal summary sub-DAG (`SummaryAgg` / +`SummaryEstimate`), use the existing summary-construction path. -A strategy should steer or wrap that path when necessary, not recreate schema derivation, column resolution, readout construction, or parameter sizing. +A strategy should steer or wrap that path when necessary, not recreate schema derivation, column resolution, evaluation construction, or parameter sizing. --- @@ -922,7 +841,7 @@ Workload-wide target discovery, deduplication, and consumer counting are separat For CSE-style decisions, pointer identity can encode actual sharing. -Two `Rc` values can be structurally equal but deliberately represent independent computation. +Two `Rc` values can be structurally equal but deliberately represent independent computation. Use the distinction intentionally. @@ -937,27 +856,25 @@ When adding a new strategy: - [ ] Implement `ReplacementStrategy::replacements`. - [ ] Return every semantically valid replacement. - [ ] Return an empty vector for non-matching targets. -- [ ] Use `Replacement::Summary` for constructed post-ASAP output. -- [ ] Use `Replacement::Rewrite` for logical pre-ASAP alternatives. +- [ ] Return `Replacement::SubDAG` for both constructed post-ASAP output and + logical pre-ASAP alternatives, with the matching `provenance`. - [ ] Add a useful rationale to every candidate. - [ ] Reuse existing legality and implementation logic instead of duplicating it. - [ ] Keep ranking and cost-based pruning out of the strategy. - [ ] Test positive and negative applicability. - [ ] Test exhaustive enumeration. - [ ] Test the actual structural semantics of each replacement. -- [ ] Test behavior with a custom cost model if the strategy uses one. +- [ ] Test that a custom cost model reorders, but does not remove, the strategy's candidates at selection. When adding a new cost model: - [ ] Override only the hooks whose behavior should change. - [ ] Keep semantic applicability outside the cost model. - [ ] Use `rank_candidates` for algorithm preference; return every input candidate exactly once. -- [ ] Use `size_params` for accuracy-to-parameter mapping. -- [ ] Use extension hooks for extension-defined implementations/readouts. - [ ] Use CSE hooks for recompute-vs.-sharing costs. - [ ] Override `estimate_cost` if consumers require numeric costs instead of `NaN`. - [ ] Test the hook directly. -- [ ] Test integration through a consumer such as `SketchAlgorithmStrategy`. +- [ ] Test integration through a selection-time consumer such as `cost_sorted` or `global_selection`. - [ ] Verify that changing cost preferences does not silently remove valid replacement candidates. --- @@ -973,25 +890,23 @@ Use this table to find the right place for a change. | Change when a strategy applies | `ReplacementStrategy::matches` | | Add a new built-in sketch candidate | `replacement.rs`'s summary-candidate mapping plus realization and accuracy contracts | | Prefer one sketch algorithm over another | `CostModel::rank_candidates` | -| Change sketch sizing for an accuracy target | `CostModel::size_params` | -| Add extension-defined implementation behavior | `CostModel::realize_extension` | -| Add extension-defined readout behavior | `CostModel::readout_extension` | +| Change sketch sizing for an accuracy target | `accuracy::estimators::size_params` (analytical; not a `CostModel` hook) | | Change CSE recomputation cost | `CostModel::cse_recompute_cost` | | Change shared-maintenance cost | `CostModel::cse_shared_maintenance_cost` | | Change current share/recompute choice | `CostModel::cse_share_decision` | | Decide whether an available implementation satisfies a required one | `impl Matcher` | -| Produce a normal (ranked-first) post-ASAP summary for one target | `SketchAlgorithmStrategy::replacements(...).into_iter().next()` | +| Produce the first-listed post-ASAP summary for one target (unranked, `summary_candidates` order) | `ASAPStrategies::replacements(...).into_iter().next()` | | Search a whole workload for supported legal candidates | `search_workload`/`search_workload_with` | | Enforce per-root result accuracy requirements | `search_workload_with_targets` | -| Coordinate compatible choices across groups | `CandidateLogicalASAPDAGs::global_selection` | +| Coordinate compatible choices across groups | `candidate_selection::global_selection` | | Assemble the selected logical DAG | `GlobalSelection::assemble_selected_dag` | -| Get every candidate ranked best-first, across a whole workload | `CandidateLogicalASAPDAGs::cost_sorted` | +| Get every candidate ranked best-first, across a whole workload | `candidate_selection::cost_sorted` | | Get a real numeric cost per candidate, not just a relative rank | `CostModel::estimate_cost` | | Enumerate valid sketch algorithms | `summary_candidates` | | Build a target with no workload context | `TargetSubDAG::new` | | Build a target with known sharing context | `TargetSubDAG::with_consumer_count` | | Explain why a replacement exists, where, and why | `explanation::explain_replacements`/`explain_replacements_with` | -| Add a new kind of replacement explanation | new `impl ReplacementStrategy`, wired into `default_strategies`/`default_strategies_with` — not a new explanation-specific trait, see §8 | +| Add a new kind of replacement explanation | new `impl ReplacementStrategy`, wired into `default_strategies` — not a new explanation-specific trait, see §8 | --- @@ -1000,7 +915,7 @@ Use this table to find the right place for a change. ### Using it ```rust -use asap_aware_mapping::{explain_replacements, ExplanationKind}; +use asap_logical_optimizer::{explain_replacements, ExplanationKind}; let explanations = explain_replacements(vec![("dashboard_p99", query)]); for explanation in &explanations { @@ -1012,8 +927,8 @@ for explanation in &explanations { } ``` -To plug in a deployment-specific strategy or `CostModel`, use `explain_replacements_with` with a strategy set built the same way `default_strategies_with` builds one — see [§2](#2-adding-or-customizing-a-costmodel) and [§4](#4-adding-both-a-strategy-and-a-cost-model). +To plug in a deployment-specific strategy, use `explain_replacements_with` with a strategy set built the same way `default_strategies` builds one — see [§1](#1-adding-a-new-replacementstrategy) and [§4](#4-adding-both-a-strategy-and-a-cost-model). Explanations do not consult a `CostModel`. ### Adding a new kind of replacement explanation -There is no separate checklist here: follow [§1](#1-adding-a-new-replacementstrategy) to add the new `ReplacementStrategy` and wire it into `default_strategies`/`default_strategies_with`, then add an `ExplanationKind` variant and ensure `explain_replacements` returns that kind for the new public candidate shape. Test the behavior through `explain_replacements` or `explain_replacements_with`; explanation reporting should not introduce a second discovery rule. +There is no separate checklist here: follow [§1](#1-adding-a-new-replacementstrategy) to add the new `ReplacementStrategy` and wire it into `default_strategies`, then add an `ExplanationKind` variant and ensure `explain_replacements` returns that kind for the new public candidate shape. Test the behavior through `explain_replacements` or `explain_replacements_with`; explanation reporting should not introduce a second discovery rule. diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index 719de6ba2..9dbd31f44 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -15,7 +15,6 @@ do not deploy a plan, and a serializable DAG is not evidence of runtime readines | Pre-ASAP IR | Frontend `lower_*` | [Lower a query](#lower-a-query-into-pre-asap-ir) | | All ranked candidates | `search_workload_with_targets` -> `cost_sorted` | [Generate and rank](#generate-and-rank-candidates) | | Custom optimization set | Construct `Vec>`, then search | [Strategies and models](#choose-strategies-and-models) | -| Summary-maintenance lifecycle comparison | Lifecycle-aware selection -> DAG assembly with maintenance decisions | [Lifecycle recipe](#lifecycle-and-capabilities) | | Selected semantic DAG / export | `global_selection` -> `assemble_selected_dag` -> export | [Selection example](#optional-whole-plan-selection-and-dag-assembly) | Each recipe ends at a different artifact. Use only the stages needed for that @@ -23,14 +22,17 @@ artifact, while preserving the checks required by its intended consumer. ## Dependencies -Inside this workspace, depend on the frontend you need, `asap-aware-mapping`, -and `asap-types`. External users can use Git dependencies pinned to a compatible -revision; use the same revision across these crates. For the example below: +Inside this workspace, depend on the frontend you need, +`asap-logical-optimizer` (Stage 1 candidate search), `asap-plan-selection` +(cost models and selection) and `asap-types`. External users can use Git +dependencies pinned to a compatible revision; use the same revision across +these crates. For the example below: ```toml [dependencies] asap-frontend-promql = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2492c42c9f5b34760706a5162aa586d3025" } -asap-aware-mapping = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2492c42c9f5b34760706a5162aa586d3025" } +asap-plan-selection = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2492c42c9f5b34760706a5162aa586d3025" } +asap-logical-optimizer = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2492c42c9f5b34760706a5162aa586d3025" } asap-types = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2492c42c9f5b34760706a5162aa586d3025" } ``` @@ -38,14 +40,16 @@ asap-types = { git = "https://github.com/ProjectASAP/ASAPPlanner", rev = "e7fdb2 | Public function | Required input | Output | | --- | --- | --- | -| `asap_frontend_promql::lower_promql_workload` | PromQL `PlanningWorkload` with a nonzero `data_ingestion_interval` | All-or-nothing `Result, PromqlError>` for normalized batch and repeating entries | -| `asap_frontend_metricsql::lower_metricsql` | Query string, `AccuracyTarget` | `Result` | -| `asap_frontend_sql::lower_sql` | Query string, `SqlCatalog`, accuracy | Async `Result`; default SQL dialect is DataFusionSQL | +| `asap_frontend_promql::lower_promql_workload` | PromQL `PlanningWorkload` with a nonzero `data_ingestion_interval` | All-or-nothing `Result>, PromqlError>` for normalized batch and repeating entries | +| `asap_frontend_metricsql::lower_metricsql` | Query string, `AccuracyTarget` | `Result, MetricsqlError>` | +| `asap_frontend_sql::lower_sql` | Query string, `SqlCatalog`, accuracy | Async `Result, SqlError>`; default SQL dialect is DataFusionSQL | | `asap_frontend_sql::lower_sql_dialect` | Same inputs plus `SqlDialect` | Async resolved Pre-ASAP query or error | | `asap_frontend_sql::lower_sql_batch` | `QueryWorkload` and catalog | Per-query results for `query_batch`; does not iterate `repeating_queries` | Lowering resolves the supported source language into the canonical query -representation. It does not enumerate Post-ASAP alternatives. A frontend may +representation: an `asap_types::ir::OperatorNode` DAG containing only +`NonASAPOp` operators, with no timing (see the +[Pre-ASAP IR reference](pre-asap-ir.md)). It does not enumerate Post-ASAP alternatives. A frontend may reject unsupported syntax or semantics; a declared language/dialect enum does not imply complete support. PromQL workload lowering uses normalized `PlanningWorkload::query_workload.entries()` order, preserving entry-to-root associations for later @@ -58,12 +62,12 @@ PromQL's public signature (types are imported from their respective crates): ```text lower_promql_workload(workload: &PlanningWorkload, now_ms: u64) - -> Result, PromqlError> + -> Result>, PromqlError> ``` `DataWorkload.data_ingestion_interval` must contain a nonzero `Evidence`. Pass the actual planning time as `now_ms` (Unix milliseconds), consistently with -downstream lifecycle planning. Expired or future cadence evidence is rejected, +downstream planning. Expired or future cadence evidence is rejected, as is expiring evidence without an observation timestamp. The histogram variant takes the same timestamp after its histogram catalog argument. The examples use `0` only because their explicitly supplied cadence is timeless. @@ -125,9 +129,9 @@ For SQL, the corresponding signatures are: ```text async lower_sql(query: &str, catalog: &SqlCatalog, accuracy: AccuracyTarget) - -> Result + -> Result, SqlError> async lower_sql_dialect(query: &str, catalog: &SqlCatalog, - dialect: SqlDialect, accuracy: AccuracyTarget) -> Result + dialect: SqlDialect, accuracy: AccuracyTarget) -> Result, SqlError> ``` | `SqlDialect` value | Current behavior | @@ -162,13 +166,13 @@ It keeps the alternatives available; it does not select an entire workload plan. ```text search_workload_with_targets<'s, Id>( - roots: Vec<(Id, Rc, Option)>, + roots: Vec<(Id, Rc, Option)>, strategies: &[Box], accuracy_model: &dyn AccuracyModel, ) -> CandidateLogicalASAPDAGs -CandidateLogicalASAPDAGs::cost_sorted(&self, cost_model: &dyn CostModel) - -> Vec> +candidate_selection::cost_sorted<'a, Id>(space: &'a CandidateLogicalASAPDAGs, cost_model: &dyn CostModel) + -> Vec> ``` | Argument | Choices / meaning | Required? | @@ -197,15 +201,15 @@ accuracy target, and prints every ranked candidate instead of selecting a winner The default cost model is suitable for inspection, not deployment calibration. ```rust -use std::rc::Rc; use asap_frontend_promql::lower_promql_workload; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, Query, PlanningWorkload, QueryLanguage, QueryRequirements, QueryWorkload, }; -use asap_aware_mapping::{ - default_strategies_with, search_workload_with_targets, - DefaultAccuracyModel, DefaultCostModel, +use asap_plan_selection::candidate_selection::cost_sorted; +use asap_plan_selection::DefaultCostModel; +use asap_logical_optimizer::{ + default_strategies, search_workload_with_targets, DefaultAccuracyModel, }; use asap_types::types::AccuracyTarget; @@ -235,15 +239,15 @@ fn main() -> Result<(), Box> { ..Default::default() }), }; - let root = Rc::new(lower_promql_workload(&workload, 0)?.remove(0)); + let root = lower_promql_workload(&workload, 0)?.remove(0); let cost_model = DefaultCostModel; - let strategies = default_strategies_with(&cost_model); + let strategies = default_strategies(); let space = search_workload_with_targets( vec![("q1", root, Some(accuracy))], &strategies, &DefaultAccuracyModel, ); - for group in space.cost_sorted(&cost_model) { + for group in cost_sorted(&space, &cost_model) { for (candidate, cost) in group.candidates.iter().zip(&group.costs) { println!("candidate={candidate:?}, reported_cost={cost:?}"); } @@ -252,14 +256,14 @@ fn main() -> Result<(), Box> { } ``` -| API (`asap_aware_mapping`, unless qualified) | Inputs | Output and limits | +| API (`asap_logical_optimizer`; `candidate_selection` is `asap_plan_selection::candidate_selection`) | Inputs | Output and limits | | --- | --- | --- | -| `search_workload` | `(query_id, Rc)` roots | `CandidateLogicalASAPDAGs` with built-in strategies/model; no explicit per-root target argument | +| `search_workload` | `(query_id, Rc)` roots | `CandidateLogicalASAPDAGs` with built-in strategies/model; no explicit per-root target argument | | `search_workload_with` | Roots, strategy slice | `CandidateLogicalASAPDAGs`; callers choose context-free replacement strategies | | `search_workload_with_targets` | Roots with optional end-to-end targets, strategies, accuracy model | Candidate space with supplied root-target checks; `None` does not supply a root-level requirement; uncertified direct DDSketch ratios remain available for backend selection | -| `CandidateLogicalASAPDAGs::cost_sorted` | Cost model | `Vec`; retains alternatives and pairs `candidates[i]` with `costs[i]` | -| `CandidateLogicalASAPDAGs::cost_sorted_with_recurrence` | Cost model, recurrence profiles, optional horizon | Ranked per-target candidate sets or `RecurrenceError`; uses recurrence for applicable share/recompute comparisons | -| `SketchAlgorithmStrategy::replacements` through `ReplacementStrategy` | One `TargetSubDAG` | Alternatives at that target; not whole-workload search | +| `candidate_selection::cost_sorted` | Cost model | `Vec`; retains alternatives and pairs `candidates[i]` with `costs[i]` | +| `candidate_selection::cost_sorted_with_recurrence` | Cost model, recurrence profiles, optional horizon | Ranked per-target candidate sets or `RecurrenceError`; uses recurrence for applicable share/recompute comparisons | +| `ASAPStrategies::replacements` through `ReplacementStrategy` | One `TargetSubDAG` | Alternatives at that target; not whole-workload search | `cost_sorted` is a ranking view, not a request to discard all but the first candidate. Display costs follow model hooks and may be unavailable/non-finite; @@ -279,15 +283,16 @@ choices are not multiplied in. Exceeding `expansion_limit` is an error, never a partial inventory. For PromQL roots that carry a target, `search_workload_with_targets` also asks -each strategy's `ReplacementStrategy::propose_for_root`. `SketchAlgorithmStrategy` +each strategy's `ReplacementStrategy::propose_for_root`. `ASAPStrategies` answers an instant-vector TopK with current-series heap realizations over rows carrying the complete series identity (`$promql_series_identity`). They are finalized, deduplicated, and marked `ReplacementProvenance::RootPhysicalRealization`. Callers do not apply `with_series_identity` themselves. Compile each with -`promql_rows::compile_current_series_readout`; other queries keep their previous +`promql_rows::compile_current_series_evaluation`; other queries keep their previous inventory. `global_selection` never commits these candidates; the backend compiles and prices them. CandidateLogicalASAPDAGs lists no placement variants: node timing -comes from the summary maintenance lifecycle. +comes from a `MaterializationAssignment` (all query time until Stage 2 +materialization, #509, decides otherwise). ## Choose strategies and models @@ -300,11 +305,11 @@ pass. An omitted strategy contributes no proposals of its own. | Value to put inside `Box::new(...)` | Meaning | In default factories? | | --- | --- | --- | -| `SketchAlgorithmStrategy::new(&model)` | Enumerates supported exact/sketch implementations and parameter choices for aggregate targets | Yes | -| `HydraGroupingStrategy::new(&model)` | Considers a shared multi-subpopulation structure for supported grouped sketch families, subject to accuracy evidence | Yes | +| `ASAPStrategies::default()` | Enumerates supported exact/sketch implementations and parameter choices for aggregate targets | Yes | +| `HydraGroupingStrategy::default()` | Considers a shared multi-subpopulation structure for supported grouped sketch families, subject to accuracy evidence | Yes | | `SharedSubDAGStrategy` | Proposes sharing versus independent recomputation at reused sub-DAGs | Yes | | `SemanticEquivalentRewriteStrategy` | Proposes supported equivalent aggregate rewrites, including decomposing average into sum/count | Yes | -| `ExactCompositionStrategy::new(&model)` | Proposes supported exact operations around summary readouts or in maintenance | Yes | +| `ExactCompositionStrategy` | Proposes exact operations around summary evaluations or in maintenance; not filtered by runtime support; `global_selection` commits one only with positive (`Some(true)`) cost-model support evidence | Yes | | Your `ReplacementStrategy` implementation | Adds domain-specific legal replacement proposals | No | `AvgToSumOverCountStrategy` is an alias for `SemanticEquivalentRewriteStrategy` @@ -328,32 +333,33 @@ whole-workload search. Selecting a strategy does not force its candidate to win. ```text default_strategies() -> Vec> -default_strategies_with<'a>(cost_model: &'a dyn CostModel) - -> Vec> replacement::default_strategies_with_evidence<'a>( - cost_model: &'a dyn CostModel, evidence: &'a dyn AccuracyEvidenceProvider, + evidence: &'a dyn AccuracyEvidenceProvider, ) -> Vec> ``` | Factory | Use when | Models used | | --- | --- | --- | -| `default_strategies()` | Exploring with built-in defaults | Built-in cost/accuracy/allocation defaults | -| `default_strategies_with(&model)` | Supplying deployment-specific costing/sizing | Supplied cost model; default accuracy/allocation | -| `default_strategies_with_evidence(&model, &evidence)` | Supplying planning-time accuracy evidence as well | Supplied cost and evidence; default accuracy/allocation | +| `default_strategies()` | Exploring with built-in defaults | Built-in accuracy/allocation defaults; no extra evidence | +| `default_strategies_with_evidence(&evidence)` | Supplying planning-time accuracy evidence | Supplied evidence; default accuracy/allocation | | Explicit vector | Controlling which context-free strategies are supplied | Models passed into each constructor | +No factory takes a cost model: candidate generation is cost-model independent. +Pass the deployment cost model to `cost_sorted`/`global_selection`. + ### Example: supply two strategies and run search ```rust -use std::rc::Rc; use asap_frontend_promql::lower_promql_workload; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, Query, PlanningWorkload, QueryLanguage, QueryRequirements, QueryWorkload, }; -use asap_aware_mapping::{ - search_workload_with_targets, DefaultAccuracyModel, DefaultCostModel, - ReplacementStrategy, SketchAlgorithmStrategy, SharedSubDAGStrategy, +use asap_plan_selection::candidate_selection::cost_sorted; +use asap_plan_selection::DefaultCostModel; +use asap_logical_optimizer::{ + search_workload_with_targets, DefaultAccuracyModel, ReplacementStrategy, + ASAPStrategies, SharedSubDAGStrategy, }; use asap_types::types::AccuracyTarget; @@ -383,16 +389,16 @@ fn main() -> Result<(), Box> { ..Default::default() }), }; - let root = Rc::new(lower_promql_workload(&workload, 0)?.remove(0)); + let root = lower_promql_workload(&workload, 0)?.remove(0); let model = DefaultCostModel; let strategies: Vec> = vec![ - Box::new(SketchAlgorithmStrategy::new(&model)), + Box::new(ASAPStrategies::default()), Box::new(SharedSubDAGStrategy), ]; let space = search_workload_with_targets( vec![("q1", root, Some(accuracy))], &strategies, &DefaultAccuracyModel, ); - println!("{:#?}", space.cost_sorted(&model)); + println!("{:#?}", cost_sorted(&space, &model)); Ok(()) } ``` @@ -404,12 +410,13 @@ not waive semantic or accuracy requirements. ### Model and evidence options Traits permit custom implementations; the following are concrete built-in options. -Module-qualified paths below are relative to `asap_aware_mapping`. +Cost models are in `asap_plan_selection` (module-qualified paths below are +relative to `asap_plan_selection::cost`); accuracy models and evidence are in `asap_logical_optimizer`. | Parameter | Available value / constructor | Meaning | | --- | --- | --- | -| `&dyn CostModel` | `DefaultCostModel` | Built-in ordering/sizing and structural estimates; no measured deployment guarantee | -| `&dyn CostModel` | `empirical_cost::EmpiricalCostModel::new(provider)` | Offline sketch-benchmark model: ranks algorithms using matching offline measurements and supplies partial lifecycle costs | +| `&dyn CostModel` | `DefaultCostModel` | Built-in ordering and structural estimates; no measured deployment guarantee | +| `&dyn CostModel` | `empirical_cost::EmpiricalCostModel::new(provider)` | Offline sketch-benchmark model: ranks algorithms using matching offline measurements | | `&dyn CostModel` | `physical_plan_cost_model::PhysicalPlanCostModel::new(&provider, calibration)?` | Deployment-specific physical-plan model: compares complete physical alternatives using provider evidence and resource calibration; evidence may be offline or online | | `&dyn AccuracyModel` | `DefaultAccuracyModel` | Built-in guarantee rules and satisfaction checks | | `&dyn AccuracyBudgetAllocator` | `EqualSplitAllocator` | Built-in allocation of composition accuracy budgets | @@ -423,7 +430,7 @@ These models differ in scope, not simply in whether they are offline or online. | Model | Evidence and comparison | Missing evidence / limits | | --- | --- | --- | -| `EmpiricalCostModel` | Offline sketch benchmarks matched to exact parameters, distribution, environment and validity interval; current algorithm ranking uses measured update CPU nanoseconds | If the measurements required for ranking are incomplete, preserves the incoming algorithm order. Supplies partial build/update lifecycle costs; `estimate_cost()` still uses `DefaultCostModel` structural scores | +| `EmpiricalCostModel` | Offline sketch benchmarks matched to exact parameters, distribution, environment and validity interval; current algorithm ranking uses measured update CPU nanoseconds | If the measurements required for ranking are incomplete, preserves the incoming algorithm order. `estimate_cost()` still uses `DefaultCostModel` structural scores | | `PhysicalPlanCostModel` | A downstream provider supplies a consistent evidence snapshot and complete physical alternatives; calibration converts modeled resource quantities into comparable costs | A candidate with incomplete evidence is unavailable, without structural-cost fallback. Current candidate admission also requires it to cost less than the raw alternative | `PhysicalPlanCostModel` does not collect online telemetry itself. Its provider @@ -440,19 +447,18 @@ accuracy guarantees. ### Example: configure all sketch-strategy providers ```rust -use asap_aware_mapping::{ - DefaultAccuracyModel, DefaultCostModel, EqualSplitAllocator, - NoAccuracyEvidence, ReplacementStrategy, SketchAlgorithmStrategy, +use asap_logical_optimizer::{ + DefaultAccuracyModel, EqualSplitAllocator, + NoAccuracyEvidence, ReplacementStrategy, ASAPStrategies, }; fn main() { - let cost = DefaultCostModel; let accuracy = DefaultAccuracyModel; let allocation = EqualSplitAllocator; let evidence = NoAccuracyEvidence; let strategies: Vec> = vec![Box::new( - SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - &cost, &accuracy, &allocation, &evidence, + ASAPStrategies::new_with_planning_inputs_and_evidence( + &accuracy, &allocation, &evidence, ), )]; // Use &strategies and &accuracy in search_workload_with_targets. @@ -463,32 +469,32 @@ fn main() { Constructor definition: ```text -SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( - cost_model: &dyn CostModel, +ASAPStrategies::new_with_planning_inputs_and_evidence( accuracy_model: &dyn AccuracyModel, allocator: &dyn AccuracyBudgetAllocator, evidence: &dyn AccuracyEvidenceProvider, -) -> SketchAlgorithmStrategy +) -> ASAPStrategies ``` All provider arguments are required for this constructor. They must outlive the -strategy vector. `SketchAlgorithmStrategy::new(&cost_model)` is the shorter -constructor using default accuracy/allocation and no extra evidence. +strategy vector. `ASAPStrategies::default()` uses default accuracy/allocation +and no extra evidence. | Extension point | What it controls | What it cannot establish alone | | --- | --- | --- | | `ReplacementStrategy` | Proposed semantic alternatives | Permission to violate query semantics or downstream support | -| `CostModel` | Candidate ordering/sizing hooks, recurrence/lifecycle and complete-cost evidence hooks | Correctness, measured costs without evidence, or installed runtime support | +| `CostModel` | Selection-time ranking, cost, support-evidence and recurrence cost hooks | Correctness, measured costs without evidence, or installed runtime support | | `AccuracyModel` | Derivation, propagation and satisfaction of guarantees | A meaningful guarantee without its required assumptions/evidence | | `AccuracyBudgetAllocator` | Local accuracy requirements proposed within composition | End-to-end correctness without subsequent validation | | `AccuracyEvidenceProvider` | Planning-time statistics used by supported strategies | Authority to change query requirements | -Models may be consumed during generation as well as ranking. Construct strategies -with the intended model/evidence; replacing only the final sorting model does not -regenerate parameter choices. For evidence-aware defaults, use -`asap_aware_mapping::replacement::default_strategies_with_evidence`. +Accuracy models, allocators and evidence are consumed during generation; the cost +model is consumed only at selection (`cost_sorted`, `global_selection` and their +`_with_recurrence` variants). Sketch parameters come from the analytical +estimators, not the cost model. For evidence-aware defaults, use +`asap_logical_optimizer::pass1::replacement::default_strategies_with_evidence`. For custom accuracy/allocation/evidence on sketches, -`SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence` exposes these providers. +`ASAPStrategies::new_with_planning_inputs_and_evidence` exposes these providers. Keep each provider's evidence scope and freshness valid for the query population. ## Workload inputs and defaults @@ -497,201 +503,21 @@ Keep each provider's evidence scope and freshness valid for the query population inputs. `QueryWorkload` contains the language and optional batch/repeating entries. Entries carry requirements, predictability, recurrence and time selection. These facts are separate: repeated queries can read data at rest. -`WorkloadDemand` associates a target with the relevant workload entry indices -and explicitly includes or omits the parallel data evidence. -Both recurrence and lifecycle planning validate this independent data evidence: -ingestion rates must be finite and nonnegative, and data at rest cannot have a -positive ingestion rate. `DataWorkload::validate()` shares these checks with -`PlanningWorkload::validate()`. +`DataWorkload::validate()` checks the independent data evidence: ingestion +rates must be finite and nonnegative, and data at rest cannot have a positive +ingestion rate. `PlanningWorkload::validate()` shares these checks. | Type/input | Current behavior | Caller responsibility | | --- | --- | --- | | `QueryRequirements::default()` | `ImplicitExact`, unspecified response latency | Pass approximation explicitly and thread per-root requirements into search | | `DataWorkload::default()` | Unknown arrival, unknown evidence | Supply facts needed for the requested comparisons | | `Evidence::default()` | No value, unknown source | Unknown/stale evidence is not zero; provide scoped valid observations | -| `DefaultCostModel` | Built-in ordering/sizing and structural cost hooks | Supply deployment evidence for calibrated comparisons | -| `SummaryMaintenanceLifecycleCostInputs::default()` | All primitive costs unknown | Implement the required lifecycle cost hooks; structural defaults are insufficient | -| `horizon: None` in lifecycle planning | Horizon-dependent alternatives are unselectable | Supply a positive horizon when comparing rates/amortized reuse | -| Lifecycle capabilities default | All four modes enabled | Override with the actual runtime support | -| Per-summary maintenance capabilities default | Incremental update, merge, delete all false | Advertise supported operations for the concrete state representation | +| `DefaultCostModel` | Built-in ordering and structural cost hooks | Supply deployment evidence for calibrated comparisons | `Default` is a Rust constructor contract, not a general serde omission rule. Several workload fields require explicit serialized values. A struct field being optional also does not guarantee every planning operation can succeed without it. -## Lifecycle and capabilities - -Use this workflow when Planner owns summary-maintenance lifecycle decisions; -otherwise the backend may make them from logical candidates. It includes both -selection and DAG assembly, so callers do not first run the ordinary workflow. -The first helper returns one `GlobalSelection`; the second is called per root -and returns a plan containing `root: Rc` plus maintenance decisions. -See the [workflow design](../design_docs/architecture/input-output-workflow.md#summary-maintenance-lifecycle-aware-helper). - -Two capabilities are distinct: the runtime can orchestrate a lifecycle, and the -chosen summary representation supports the required state operations. Both must -hold. Workload legality and known cost evidence can further restrict alternatives. - -### API definition and options - -```text -global_selection_with_summary_maintenance_lifecycles<'a, Id>( - space: &'a CandidateLogicalASAPDAGs, demand: WorkloadDemand<'_>, - now_ms: u64, horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, -) -> Result, SummaryMaintenanceLifecycleSelectionError> - -assemble_selected_dag_with_summary_maintenance_lifecycles( - selection: &GlobalSelection<'_>, target: &Rc, - demand: WorkloadDemand<'_>, now_ms: u64, horizon: Option, - capabilities: SummaryMaintenanceLifecycleCapabilities, cost_model: &dyn CostModel, -) -> Result, SummaryMaintenanceLifecycleAssemblyError> -``` - -| Argument | Values / requirements | -| --- | --- | -| `space`, `demand` | Actual candidate space plus query demand, optional data evidence, and one normalized workload entry index for each `space.roots` entry | -| `target` | A root from `space.roots`, after canonical sharing | -| `demand` | `WorkloadDemand::new_with_data(...)` when data evidence is available; use `new_without_data(...)` only when its absence is intentional | -| `now_ms` | Actual planning time in Unix milliseconds for evidence freshness | -| `horizon` | `Some(Horizon(seconds))` with positive finite seconds, or `None` when horizon-dependent comparisons are unavailable | -| `capabilities` | Explicit Boolean fields below; several may be true | -| `cost_model` | A model supplying required lifecycle and raw-comparison evidence; default structural estimates are not enough | - -| Capability field | `true` permits consideration of… | `false` means… | -| --- | --- | --- | -| `supports_ephemeral` | Fresh build per invocation, retired afterward | Exclude that lifecycle | -| `supports_prepared` | Build before a predictable execution and retain until it | Exclude that lifecycle | -| `supports_shared` | Retain state for multiple reads | Exclude that lifecycle | -| `supports_continuously_maintained` | Keep state current as updates arrive | Exclude that lifecycle | - -All flags default to true; integrations should pass real support. Enabling a -flag does not override workload, algorithm-operation or evidence checks. - -### Example: lifecycle-aware planning for a batch-only runtime - -This helper takes the real workload and cost provider from your application. -It supports one searched root mapped to one workload entry, and returns a typed -plan/error rather than making up costs. For a shared root consumed by several -entries, construct demand using all applicable indices. - -```rust -use asap_aware_mapping::{ - global_selection_with_summary_maintenance_lifecycles, - assemble_selected_dag_with_summary_maintenance_lifecycles, CostModel, Horizon, CandidateLogicalASAPDAGs, - SummaryMaintenanceLifecycleCapabilities, SummaryMaintenanceLifecyclePlan, - WorkloadDemand, -}; -use asap_types::workload::PlanningWorkload; - -fn plan_batch_root( - space: &CandidateLogicalASAPDAGs<&str>, - workload: &PlanningWorkload, - entry_index: usize, - now_ms: u64, - horizon: Option, - model: &dyn CostModel, -) -> Result, Box> { - if space.roots.len() != 1 { - return Err("this example requires exactly one root".into()); - } - let capabilities = SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: false, - }; - let indices = [entry_index]; - let demand = WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &indices, - }; - let selection = global_selection_with_summary_maintenance_lifecycles( - space, demand, now_ms, horizon, capabilities, model, - )?; - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, &space.roots[0].1, demand, - now_ms, horizon, capabilities, model, - )?; - if let Some(plan) = &plan { - println!("raw_recompute={}, deployments={:#?}", - plan.selected_raw_recompute, plan.deployments); - } - Ok(plan) -} -``` - -Use this helper with the `space` built by the search example and the corresponding -workload/provider. No incremental lifecycle is permitted, but unknown evidence -can still prevent choosing summary state. If only one legal alternative remains, -recording it is a complete lifecycle decision. Data-at-rest alone does not imply -that prepared or retained shared state is supported. - -| Function | Inputs | Output / promise | -| --- | --- | --- | -| `plan_summary_maintenance_lifecycles` | Assembled logical DAG root, `WorkloadDemand`, `now_ms`, optional horizon, runtime capabilities, cost model | `Result` for that fixed root; does not revisit all semantic candidates | -| `global_selection_with_summary_maintenance_lifecycles` | `CandidateLogicalASAPDAGs`, workload/root-entry associations, time, horizon, capabilities, cost model | Lifecycle-aware compatible selection/error, using eligible cost evidence | -| `assemble_selected_dag_with_summary_maintenance_lifecycles` | Selection, target root and lifecycle context | Optional lifecycle plan/error; attaches state deployment decisions | -| `enumerate_summary_maintenance_lifecycles` | Same inputs as `plan_summary_maintenance_lifecycles` | `SummaryMaintenanceLifecycleCandidates`: per unique retained state, every alternative with its cost or rejection; nothing selected. `guarantee(&lifecycle)` gives the mode/schedule that alternative would carry | -| `SummaryMaintenanceLifecycleCandidates::select(choices)` | One `(PostAsapNodeId, SummaryMaintenanceLifecycle)` per state, copied from `deployments()` | The same `SummaryMaintenanceLifecyclePlan` Planner selection would produce for that combination, or `SummaryMaintenanceLifecycleChoiceError` when a choice is unknown, missing, duplicated, rejected, schedule-incompatible, or not completely estimable | - -Inspect `deployments`, their selected lifecycle/alternatives/rejections, -`selected_raw_recompute`, and optional summary/raw costs. Success of a function -call alone is not a certificate that every desired summary was selected or fully -costed. A raw alternative remains a downstream execution obligation. - -Lifecycle feasibility and costs must affect final deployment comparison. Running -lifecycle analysis after structural selection can evaluate the selected root, -but does not make the earlier selection lifecycle-optimal. An application may -consume ranked candidates and perform this comparison downstream instead. - -A deployment that prices lifecycles itself calls -`enumerate_summary_maintenance_lifecycles`, prices the alternatives, and binds -its choice with `select`. A choice is accepted only if Planner could select it: -an alternative with `MissingCostEvidence` is accepted only when the cost model's -complete-candidate hook covers lifecycle costs. Window frameworks and totals come -from that hook, as in Planner selection. - -A lifecycle choice then fixes each physical placement through timing: a -continuously maintained state and its inputs run at ingestion time, while an -ephemeral one stays at query time. Compile each query's `PostAsapDAG` once and -cut every chosen assignment from that result: - -```rust -use asap_physical_operators::physical_planner::{ - compile, cut_candidate, frontier_from_timing, -}; - -let compiled = compile(&dag, inputs, &roots)?; // each node lowered once -for plan in lifecycle_plans { - let frontier = frontier_from_timing(&plan.execution_timed_dag()?)?; - // Precompute/query DAGs split at `frontier`; no logical lowering. - let candidate = cut_candidate(&compiled, &frontier)?; - // Check feasibility and price `candidate`; bind the selected one as is. -} -``` - -The frontier is the set of ingestion-time nodes read by query-time nodes (or an -ingestion-time root). `frontier_from_timing` rejects a query-time node feeding -an ingestion-time node. `cut_candidate` returns exactly what -`compile_candidate(&dag, inputs, &roots, &frontier)` returns and rejects the -same invalid frontiers. If the DAG has an ingestion-time `Binary`, compile with -the same timing for that node, because it lowers differently. Temporal pane -candidates are a different lowering and still use -`compile_temporal_pane_candidate`. - -Retained states are `SummaryAgg` nodes and `MaintainPopulation` nodes that do -not feed a `SummaryAgg`; a population that does feed one is part of that -state's input. The lifecycle cost hooks (`summary_maintenance_capabilities`, -`summary_maintenance_lifecycle_cost_inputs_for_horizon`) and the complete-candidate -hook therefore also receive `MaintainPopulation` nodes. A model that does not -recognize one should return unknown costs, which keep its alternatives -unselected; a model that prices every node uniformly now also prices -populations, so population candidates can win lifecycle-aware selection. `SummaryMaintenanceLifecyclePlan::execution_timed_dag` times a -population as it times a summary state: retained at ingestion, `Ephemeral` at -query time from the raw source. - ## Optional whole-plan selection and DAG assembly ### What does global selection mean? @@ -727,15 +553,15 @@ constructs the selected semantic DAG while preserving shared nodes. | `cost_sorted()` | How are the alternatives ranked for each subexpression? | Ranked alternatives per target | | `global_selection()` | Which compatible choices should be used together, accounting for sharing and dependencies? | A coordinated selection across targets under the supplied model | -Plain `global_selection()` does not automatically perform lifecycle planning or -establish physical deployment feasibility. Use the corresponding evidence-aware -workflow for those decisions. Downstream still owns physical commitment. +Plain `global_selection()` does not decide materialization or establish +physical deployment feasibility. Stage 2 materialization (#509) will own +materialization; downstream still owns physical commitment. -| Method on `CandidateLogicalASAPDAGs` / `GlobalSelection` | Behavior | +| Function or method | Behavior | | --- | --- | -| `CandidateLogicalASAPDAGs::global_selection(&model)` | Compatible structural selection across targets; no recurrence or lifecycle planning implied | -| `CandidateLogicalASAPDAGs::global_selection_with_recurrence(...)` | Compatible selection using supplied recurrence profiles/horizon; no lifecycle commitments implied | -| `GlobalSelection::assemble_selected_dag(&target)` | `Result>, RealizationError>`; constructs semantic IR, not stored summary data | +| `candidate_selection::global_selection(&space, &model)` | Compatible structural selection across targets; no recurrence or materialization planning implied | +| `candidate_selection::global_selection_with_recurrence(...)` | Compatible selection using supplied recurrence profiles/horizon; no materialization commitments implied | +| `GlobalSelection::assemble_selected_dag(&target)` | `Result>, RealizationError>`; constructs untimed semantic IR, not stored summary data | Use a target associated with the searched space; DAG assembly can return `None` when that target is absent. A downstream integration can use these convenience @@ -746,23 +572,25 @@ for checking complete physical alternatives and deployment constraints. ### API definition and example ```text -CandidateLogicalASAPDAGs::global_selection(&self, cost_model: &dyn CostModel) -> GlobalSelection<'_> -GlobalSelection::assemble_selected_dag(&self, target: &Rc) - -> Result>, RealizationError> +candidate_selection::global_selection<'a, Id>(space: &'a CandidateLogicalASAPDAGs, cost_model: &dyn CostModel) + -> CostedGlobalSelection<'a> // derefs to GlobalSelection +GlobalSelection::assemble_selected_dag(&self, target: &Rc) + -> Result>, RealizationError> ``` For structural inspection only, this complete example selects a semantic root -and exports its inspection DAG. It performs no lifecycle or deployment planning. -Use lifecycle-aware selection above when the comparison needs those decisions. +and exports its inspection DAG. It performs no materialization or deployment +planning. ```rust -use std::rc::Rc; use asap_frontend_promql::lower_promql_workload; use asap_types::workload::{ AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, Query, PlanningWorkload, QueryLanguage, QueryRequirements, QueryWorkload, }; -use asap_aware_mapping::{search_workload, DefaultCostModel}; +use asap_plan_selection::candidate_selection::global_selection; +use asap_plan_selection::DefaultCostModel; +use asap_logical_optimizer::search_workload; use asap_types::types::AccuracyTarget; fn main() -> Result<(), Box> { @@ -790,9 +618,9 @@ fn main() -> Result<(), Box> { ..Default::default() }), }; - let root = Rc::new(lower_promql_workload(&workload, 0)?.remove(0)); + let root = lower_promql_workload(&workload, 0)?.remove(0); let space = search_workload(vec![("q1", root)]); - let selection = space.global_selection(&DefaultCostModel); + let selection = global_selection(&space, &DefaultCostModel); // Search may canonicalize roots; use the root returned by CandidateLogicalASAPDAGs. if let Some(summary) = selection.assemble_selected_dag(&space.roots[0].1)? { let dag = asap_types::dag_export::export_summary(&summary); @@ -806,27 +634,26 @@ fn main() -> Result<(), Box> { | Function/type | Purpose | | --- | --- | -| `asap_types::dag_export::export(&query)` | Pre-ASAP inspection DAG | -| `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection DAG | -| `asap_types::post_asap::compile_post_asap_dag(&root)` | Compile a semantic DAG with execution-data-state validation; not a physical plan | +| `asap_types::dag_export::export(&query)` | Pre-ASAP inspection dag | +| `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection dag | +| `asap_types::ir::apply_materialization_timings(&root, &assignment, &mut TimingMemo::new())` | Write execution timing into every node from a `MaterializationAssignment` (default: all query time) and validate the data-state edges; `PlanOutput::execution_timed_dag()` applies the default to a planned workload | +| `asap_types::ir::export::compile_post_asap_dag(&timed_root)` | Export a timed DAG as a `PostAsapDAG` (wire version 7); rejects an untimed node; not a physical plan | | `PostAsapDAGDocument::new(dag)` and `.validate()` | Versioned semantic envelope and explicit validation; constructing it alone does not validate | -| `asap_aware_mapping::export_summary_maintenance_plan(&plan)` | DAG plus lifecycle deployments, alternatives and available cost/guarantee information | | `explain_replacements` / `explain_replacements_with` | Findings from default/custom-strategy search; not a complete physical feasibility report | Choose the export matching your intended handoff: an inspection DAG is not -interchangeable with a versioned execution contract. Preserve lifecycle and +interchangeable with a versioned execution contract. Preserve cost/guarantee evidence needed downstream instead of exporting only a bare DAG. For public symbol details, build local API documentation with: ```sh -cargo doc -p asap-aware-mapping -p asap-types --no-deps +cargo doc -p asap-logical-optimizer -p asap-plan-selection -p asap-types --no-deps ``` ## Source references - [Frontend PromQL](../../crates/frontend-promql/src/lib.rs), [SQL](../../crates/frontend-sql/src/lib.rs), [MetricsQL](../../crates/frontend-metricsql/src/lib.rs) -- [Search, ranking and selection](../../crates/asap-aware-mapping/src/replacement.rs) -- [Cost models](../../crates/asap-aware-mapping/src/cost_model.rs) -- [Lifecycle APIs](../../crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs) -- [Workload types](../../crates/types/src/workload.rs) +- [Search, ranking and selection](../../crates/logical-optimizer/src/pass1/replacement.rs) +- [Cost models](../../crates/plan-selection/src/cost/cost_model.rs) +- [Workload types](../../crates/types/src/workload/mod.rs) - [Planner-runtime contract](../design_docs/architecture/planner-runtime-contract.md) diff --git a/docs/develop_docs/local-logical-candidates.md b/docs/develop_docs/local-logical-candidates.md new file mode 100644 index 000000000..4673647bd --- /dev/null +++ b/docs/develop_docs/local-logical-candidates.md @@ -0,0 +1,39 @@ +# Local logical alternatives (Pass 1) + +`asap_logical_optimizer::pass1::logical_candidates` enumerates local realization choices over +unified `OperatorNode` and `QueryRoot` inputs. It is the first part of logical +ASAP optimization in [planner layering](../design_docs/proposals/planner-layering.md). + +`enumerate_local_logical_candidates(roots)` returns a `LocalLogicalCandidates` +inventory containing the original named roots and one `LocalLogicalTarget` per +reachable single-measure aggregate. Discovery includes operator producers read by +scalar roots and expressions. Pointer identity prevents repeated discovery of one +shared producer. Inputs with assigned execution timing are rejected. + +Each target retains its original operator, including grouping, filter and input +context, and has an unranked list of existing `Realization` descriptors: + +- Exact execution of the original sub-DAG is always retained as `PassThrough`. +- Mergeable exact intents also offer their exact accumulator kind and parameters. +- Approximate-capable intents offer all declared specialized/universal sketch + algorithms with nominal dimensions from the built-in sizing contracts. +- Exact accuracy requests do not acquire approximate alternatives. Distinct-tuple + counts do not acquire single-value UnivMon alternatives. + +For example, an approximate single-column distinct count offers exact execution, +HLL, Theta, KMV and UnivMon. These are candidate choices, not assessed accuracy +certificates. Catalog order is stable and has no cost/preference meaning. + +The API accepts no empirical cost or accuracy model, runtime capabilities, storage +policy or materialization assignment. It does not rank, select, construct runtime +state or claim physical feasibility. Pass 2 must compose and structurally validate +replacement sub-DAGs and retain independent/shared alternatives before physical +planning and complete workload selection. The descriptors are not executable +plans, and callers must not execute the first choice as a selection policy. + +Multi-measure aggregates remain intact in the roots until an explicit semantic +split is supported. Opaque deployment extensions retain exact execution here; +additional local alternatives require an explicit logical rule rather than a cost +model making a generation decision. The legacy ranked search remains available +for the existing pipeline until its later cutover; this module supplies the new +logical-only entry point without changing production selection prematurely. diff --git a/docs/develop_docs/logical-asap-dag.md b/docs/develop_docs/logical-asap-dag.md new file mode 100644 index 000000000..99696b175 --- /dev/null +++ b/docs/develop_docs/logical-asap-dag.md @@ -0,0 +1,50 @@ +# Logical ASAP DAG transport + +This interface exports the unified operator/scalar IR at the logical stage of +[planner layering](../design_docs/proposals/planner-layering.md). It describes +what to compute, including committed summary families, before physical planning +chooses implementations and materialization. + +## Interface + +`asap_types::ir::export` exposes: + +- `compile_logical_asap_dag(&Rc)` for a flat `LogicalASAPDAG`. +- `compile_logical_asap_query(&QueryRoot)` for operator or standalone scalar roots. +- `compile_logical_asap_dag_with_node_ids(...)` for that DAG and a compiler-local + `LogicalASAPNodeIdentityMap` with `node_id` and `operator_node` lookups. +- `LogicalASAPDAGDocument::new(dag)` and `validate()` for the versioned transport + envelope. Logical wire version 1 is distinct from the older phase-assigned + post-ASAP format. + +The compiler first checks the in-memory DAG's structural contracts. Untimed +ordinary plans and summary plans are valid inputs. Export does not assess +accuracy against request requirements or select a physical plan. + +Each `LogicalASAPDAGNode` contains an ID, operator payload, result kind, output +schema and optional accuracy guarantee. Each `LogicalASAPDAGEdge` contains +producer/consumer IDs, input role, intermediate schema and grouping compatibility. +The DAG has one semantic operator or scalar root. A standalone scalar constant +needs no synthetic operator node. IDs are local to one export. + +Scalar expressions remain owned by their operators. Their wire representations +replace explicit operator references with IDs. `ScalarRef` edges record those +producer dependencies. A shared operator is exported once even when ordinary +inputs and scalar expressions both reference it. + +## Physical boundary + +Logical nodes and edges contain no execution state, assigned timing, storage tier, +retention or pane-alignment assertion. Physical planning chooses, for each eligible +sub-DAG, no materialization, query-time materialization, or ingestion-time +materialization. Execution timing follows that choice and its dependencies. + +The optional `OperatorNode.timing` field belongs to the common IR and may later +record a physical assignment; it is not part of logical transport. There is no +intermediate timed-DAG stage. Physical planning owns phase validation and any +splitting needed when shared consumers require incompatible execution contexts. + +Transport validation checks graph identity, connectivity, acyclicity, edge schemas +and declared summary family/grouping metadata. Full scalar/operator typing remains +an in-memory structural validation responsibility. Neither check proves runtime +capability, cost, response latency or accuracy feasibility. diff --git a/docs/develop_docs/metrics-observability-corpora.md b/docs/develop_docs/metrics-observability-corpora.md index d127e92ce..b5472ef89 100644 --- a/docs/develop_docs/metrics-observability-corpora.md +++ b/docs/develop_docs/metrics-observability-corpora.md @@ -49,19 +49,19 @@ o11y-bench, and awesome-prometheus-alerts. They are not duplicated here. The test prints totals, parse errors, lowering errors, pre-ASAP successes, post-ASAP candidates, unchanged queries, and post-ASAP errors. `Pre-ASAP` means -that parsing and lowering produced a `QueryExpr`. `Post-ASAP candidate` means -the isolated `SketchAlgorithmStrategy` produced a non-`KeepPreAsap` summary -candidate. `Unchanged` is a successful pre-ASAP query for which that strategy -returned only the pre-ASAP fallback. +that parsing and lowering produced an `OperatorNode` DAG. `Post-ASAP candidate` +means the isolated `ASAPStrategies` produced a candidate that contains +an ASAP operator (`contains_asap()`). `Unchanged` is a successful pre-ASAP query +for which that strategy returned only the kept pre-ASAP sub-DAG (`retain_exact`). ## Strategies The corpus measurement deliberately uses only -`SketchAlgorithmStrategy::default_cost_model().replacements(...)` on each +`ASAPStrategies::default().replacements(...)` on each query root. It does not measure workload-wide search or the other default strategies. -The default workload search currently registers `SketchAlgorithmStrategy`, +The default workload search currently registers `ASAPStrategies`, `HydraGroupingStrategy`, `SharedSubDAGStrategy`, and `AvgToSumOverCountStrategy`. Workload context can additionally contribute `RollupStrategy` and `AccuracyReconciliationStrategy`. This baseline is diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md index d263b7825..d2c10f535 100644 --- a/docs/develop_docs/native-promql-inputs.md +++ b/docs/develop_docs/native-promql-inputs.md @@ -23,7 +23,7 @@ operator's metric-name/result-label rules. Source selection, complete window coverage and revision admission remain deployment responsibilities. Planner's maintained-population candidate recognizes this explicit identity -representation. Its TopK readout compiles automatically to `CurrentSeries`, +representation. Its TopK evaluation compiles automatically to `CurrentSeries`, `Sort`, and `Limit`; deployment supplies the raw boundary or an already maintained population boundary. Compilation does not open either source. diff --git a/docs/develop_docs/offline-sketch-evidence.md b/docs/develop_docs/offline-sketch-evidence.md index cd331e57d..fd0e5c416 100644 --- a/docs/develop_docs/offline-sketch-evidence.md +++ b/docs/develop_docs/offline-sketch-evidence.md @@ -1,18 +1,18 @@ # Consuming offline sketch measurements This document is for developers integrating sketch-bench with the planner. The -Rust schema is `asap_aware_mapping::empirical_cost::EvidenceArtifact`; its JSON +Rust schema is `asap_plan_selection::cost::empirical_cost::EvidenceArtifact`; its JSON schema version is `1`. Required artifact-level `benchmark_version` and `model_version` identify the producer and cost interpretation independently of the serialization schema. Producers export offline benchmark measurements using this contract; benchmark tooling is delivered separately from the core provider. The checked-in [JSON Schema](offline-sketch-evidence.schema.json) describes the -wire format. `crates/asap-aware-mapping/tests/data/offline-evidence-synthetic.json` +wire format. `crates/plan-selection/tests/data/offline-evidence-synthetic.json` is an explicitly fabricated format fixture, never benchmark evidence. Runtime validation additionally checks cross-field constraints and matching context. -Resource values share the internal `asap_types::resources::PhysicalResources` container. Its CPU payload preserves units: `ModeledCpu { cpu_ops: f64 }` represents modeled operations, while `MeasuredCpu` contains optional measured `build_cpu_ns`, `update_cpu_ns`, `merge_cpu_ns`, `prepare_cpu_ns`, and `read_cpu_ns` @@ -60,8 +60,7 @@ distribution or machine; the provider does not interpolate between datasets. Each measured resource is an optional `Measurement` with `value`, optional `stddev`, `samples`, and optional `method`. CPU fields are process CPU nanoseconds per operation; `build_cpu_ns` measures empty construction. Building an ingested -snapshot additionally requires `sample_count × update_cpu_ns`; the lifecycle -helper returns that sum only when both measurements exist. Memory and disk +snapshot additionally requires `sample_count × update_cpu_ns`. Memory and disk fields are bytes; `scan_bytes` records bytes read by scans, not storage occupancy. Producer methods must state what was measured and how normalization was performed. `retained_bytes` is distinct @@ -87,12 +86,10 @@ scores as CPU or measured savings. Deployment cost models can own the provider and call `lookup` with their own parameter sizing. This preserves the deployment's other cost and capability -hooks. The provider's lifecycle helper returns available build/update CPU costs -for a single independently instantiated state. It deliberately leaves retention, -retirement and read costs unknown. In particular, a point-frequency benchmark -read does not price a total-count read, even when both use CMS. A deployment must -match readout semantics and supply the missing lifecycle and raw-query evidence -before selecting and pricing a complete physical plan. Never combine these +hooks. A point-frequency benchmark read does not price a total-count read, even +when both use CMS. A deployment must match evaluation semantics and supply +retention, retirement, read and raw-query evidence before selecting and pricing +a complete physical plan. Never combine these nanosecond costs with CPU operation counts without explicit calibration. `error` contains offline observed statistics and a query descriptor. Its metric @@ -105,9 +102,11 @@ parameters solely because one dataset had low observed error. ## Query-matched offline recommendations -`empirical_comparison::recommend_offline` consumes the companion -[`OfflineComparisonEvidence` JSON format](offline-comparison-evidence.schema.json). -This combines the sketch artifact with explicit query bindings and a separately +The companion +[`OfflineComparisonEvidence` JSON format](offline-comparison-evidence.schema.json) +has no planner consumer: its former consumer, +`empirical_comparison::recommend_offline`, had no callers and was deleted under +#572. The contract below describes the format only. It combines the sketch artifact with explicit query bindings and a separately identified exact implementation measured on the same machine, OS, runtime and input distribution. Its `disjoint_live_state_v1` timing contract requires construction, ingestion, exact preparation and read CPU to be timed separately @@ -116,7 +115,7 @@ not be passed as these disjoint phase measurements. `MeasurementQueryBinding` is the producer's explicit assertion identifying the read/error probe population. The consumer checks that binding and the error -record's readout kind/value type; it cannot recover or certify the original +record's evaluation kind/value type; it cannot recover or certify the original probe set from an aggregate error number alone. The supported workload is an immutable i64 point-frequency snapshot, fully @@ -131,7 +130,7 @@ post-merge error and an exact merge baseline exist. The caller supplies an `EmpiricalAccuracyRequirement`: the exact observed error metric, maximum accepted mean, and minimum number of offline trials. This is -separate from `AccuracyTarget`. Every candidate must match the readout descriptor, +separate from `AccuracyTarget`. Every candidate must match the evaluation descriptor, error metric, trial count and all ordinary distribution/configuration/environment checks. A zero observed error is neither proof of exactness nor a per-key bound. diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 5d048d7f7..56ebf0e42 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -1,14 +1,15 @@ # Physical compile coverage for deployment computation Audience: developers moving computation from ASAPQuery-backend into -`asap_physical_operators::physical_planner`. +`asap_executor::physical_planner`. ## Contract -Logical selection decides what to compute. The maintenance lifecycle sets node -timing. `physical_planner::compile` turns a timed `PostAsapDAG` into physical +Logical selection decides what to compute. A `MaterializationAssignment` sets +node timing (all query time until Stage 2 materialization, #509, decides +otherwise). `physical_planner::compile` turns a timed `PostAsapDAG` into physical operator DAGs. The backend owns ingestion, panes, storage, stored-state -readout, external exact engines, pricing/selection, and execution scheduling. +evaluation, external exact engines, pricing/selection, and execution scheduling. A backend lowering is *covered* when `compile` accepts the corresponding `PostAsapDAG` node and produces operators with the same result. The backend @@ -29,7 +30,7 @@ Status values: | # | Backend site | Computation | Planner node | Status at #475 | Notes | |---|---|---|---|---|---| -| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` DAG for a native query | `Fallback { QueryExpr }` sub-DAGs plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | +| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` dag for a native query | `Fallback { QueryExpr }` sub-DAGs plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | | 2 | `QueryTimeOperator::Aggregate` (sum/min/max/avg/count) | Grouped value aggregation | `Value::Exact(Aggregate)`; `SummaryAgg{ExactAggregate, Reduce}` over finalized values | Supported | Also `promql_values::compile_aggregate`. | | 3 | `QueryTimeOperator::Sort`, `Limit` (topk, sort, sort_desc) | Ordering and per-group limits | `Value::Sort`, `Value::Limit` | Supported | | | 4 | `QueryTimeOperator::Binary`, `QueryPlanNode::Binary` (vector ⊗ scalar) | Arithmetic with a scalar operand | `Binary` whose operand is `Fallback{PromqlScalarBridge(Literal)}` | Missing | Query-time `Binary` accepts only label-map vector schemas. The literal node has no native binding. | @@ -43,15 +44,15 @@ Status values: | 12 | `logical_dag.rs` `Subquery`, `subquery_grid`, `expanded_inputs` | Re-evaluate the child on a step grid and assemble a matrix | `Fallback{PromqlSubquery}` | Missing | No Planner operator. | | 13 | `QueryPlanNode::Scalar`, `DAGCompiler::lower` scalar literal | Scalar constant | `Fallback{PromqlScalarBridge(Literal)}` | Missing | Only `promql_values::compile_scalar`. | | 14 | `DAGCompiler::lower` `ReduceSum`; `physical_values.rs` PerEntity projection | Sum over finalized values; per-entity identity | `SummaryAgg{ExactAggregate(Sum)}` | Supported | The backend builds an identity `Operator::project` itself for PerEntity. | -| 15 | `DAGCompiler::lower` `ExactReadout`; `post_asap_readout.rs` ExactReadout | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | -| 16 | `post_asap_readout.rs` SummaryEstimate (`readout_bound`, `expand_item_rows`) | Sketch estimate per group; TopK item expansion | `SummaryEstimate` | Partial | The backend's label-map state layout and MetricsQL `__name__` rules have no Planner equivalent. `compile_exact_readout` has no sketch counterpart. | -| 17 | `post_asap_readout.rs` SummaryMerge (`merge_bound_states`) | Merge states by group | `SummaryMerge` | Supported | Union plus `summary_merge`. | -| 18 | `post_asap_readout.rs` counter range parameters | Counter lookback for rate/increase | `TimeRange` ancestor of finalization | Supported | Applied through `with_counter_lookback`. | -| 19 | `post_asap_readout.rs` `execute_value_fragment` | Per-timestamp binding of a value fragment | n/a | Backend | Evaluation scheduling. | +| 15 | `DAGCompiler::lower` `ExactEvaluation`; `post_asap_evaluation.rs` ExactEvaluation | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | +| 16 | `post_asap_evaluation.rs` SummaryEstimate (`evaluation_bound`, `expand_item_rows`) | Sketch estimate per group; TopK item expansion | `SummaryEstimate` | Partial | The backend's label-map state layout and MetricsQL `__name__` rules have no Planner equivalent. `compile_exact_evaluation` has no sketch counterpart. | +| 17 | `post_asap_evaluation.rs` SummaryMerge (`merge_bound_states`) | Merge states by group | `SummaryMerge` | Supported | Union plus `summary_merge`. | +| 18 | `post_asap_evaluation.rs` counter range parameters | Counter lookback for rate/increase | `TimeRange` ancestor of finalization | Supported | Applied through `with_counter_lookback`. | +| 19 | `post_asap_evaluation.rs` `execute_value_fragment` | Per-timestamp binding of a value fragment | n/a | Backend | Evaluation scheduling. | | 20 | `DAGCompiler::lower` SummaryJoin / Subtract / Delete | Summary algebra | `SummaryJoin`, `SummarySubtract`, `SummaryDelete` | Missing | The backend also rejects these (`ExactFallback`). | -| 21 | `current_series.rs` Snapshot + TopK | Current-series ranking | `ReadPopulation{TopK}` | Supported | | -| 22 | `current_series.rs` Sum / Count / Average | Current-series aggregates | `ReadPopulation{Sum,Count,Average}` | Missing | `compile` accepts only TopK. | -| 23 | `current_series.rs` Quantile | Current-series quantile | `ReadPopulation{Quantile}` | Missing | No exact quantile reduction. | +| 21 | `current_series.rs` Snapshot + TopK | Current-series ranking | `EvaluatePopulation{TopK}` | Supported | | +| 22 | `current_series.rs` Sum / Count / Average | Current-series aggregates | `EvaluatePopulation{Sum,Count,Average}` | Missing | `compile` accepts only TopK. | +| 23 | `current_series.rs` Quantile | Current-series quantile | `EvaluatePopulation{Quantile}` | Missing | No exact quantile reduction. | | 24 | `raw_dag.rs` weight `Column` | Summary update from a sample/projected value | `SummaryAgg` | Supported | | | 25 | `raw_dag.rs` weight `Constant` | Unit/constant-weight update | `SummaryAgg` | Missing | `compile_node` requires a column weight. | | 26 | `raw_dag.rs` item `Column` / `Tuple` | Keyed update item | `SummaryAgg{item}` | Supported | `keyed_summary_build`. | @@ -70,7 +71,7 @@ Totals at #475: 11 Supported, 4 Partial, 14 Missing, 2 Backend. | 4, 8, 13 | Query-time `Binary` folds a scalar-literal operand into a projection over grouped value rows. | | 5 | Query-time `Binary` over grouped value rows performs an inner equi-join on equal label columns, then applies the operator. Per-series rows remain Partial. | | 15 | Count finalization converts exactly to the declared Float64 value. | -| 22, 23 | `ReadPopulation` Sum/Count/Average/Quantile compile to grouped aggregation. `Reduction::Quantile` implements PromQL interpolation. | +| 22, 23 | `EvaluatePopulation` Sum/Count/Average/Quantile compile to grouped aggregation. `Reduction::Quantile` implements PromQL interpolation. | Totals after this change: 17 Supported, 4 Partial, 8 Missing, 2 Backend. @@ -124,10 +125,10 @@ Totals are unchanged: 19 Supported, 5 Partial, 5 Missing, 2 Backend. | Row | Change | |---|---| -| 5 | Query-time `Binary` over rows with a series identity, such as per-series readouts of stored state, uses the Fallback's `series_labels` and `series_binary`. Examples: `avg_over_time` as stored sum/count, and `rate(a) / rate(b)`. Matching drops `__name__` and honors `on`/`ignoring` when the payload carries them. Only one-to-one arithmetic is covered; `group_left`/`group_right` stay rejected and comparisons are row 7. Now Supported. | +| 5 | Query-time `Binary` over rows with a series identity, such as per-series evaluations of stored state, uses the Fallback's `series_labels` and `series_binary`. Examples: `avg_over_time` as stored sum/count, and `rate(a) / rate(b)`. Matching drops `__name__` and honors `on`/`ignoring` when the payload carries them. Only one-to-one arithmetic is covered; `group_left`/`group_right` stay rejected and comparisons are row 7. Now Supported. | | 4, 8 | A literal operand also applies to per-series rows and drops `__name__`, in the Fallback too. Series whose label sets become equal are an error, as in Prometheus. | -Grouped `sum`/`avg`, current-series `Sum`/`Average` readouts, and +Grouped `sum`/`avg`, current-series `Sum`/`Average` evaluations, and `sum_over_time`/`avg_over_time` use Prometheus' Kahan-Neumaier summation. An average switches to an incremental mean once the running sum would overflow. The grouped path also serves SQL `SUM`/`AVG` over Float64, which are now @@ -175,7 +176,7 @@ and `group_left`, including a right-side series identity when needed. Thus |---|---| | 1 | Comparisons, `bool`, set operators, `group_left`/`group_right`, `scalar()` operands, and literals over aggregates whose value has another name, such as `sum by (job) (a) * 2`. Still Partial. | | 5 | Grouped `Binary` rows use the same operator instead of a relational join. A duplicate match group is now an error instead of a cross product. | -| 7 | Fallback, grouped `Binary`, and per-series comparisons and sets. Temporal stored readouts drop `__name__` before matching, including exact Count conversion, and reject duplicate output identities. | +| 7 | Fallback, grouped `Binary`, and per-series comparisons and sets. Temporal stored evaluations drop `__name__` before matching, including exact Count conversion, and reject duplicate output identities. | Totals after this change: 20 Supported, 5 Partial, 4 Missing, 2 Backend. @@ -188,7 +189,7 @@ Totals after this change: 20 Supported, 5 Partial, 4 Missing, 2 Backend. An argument whose output provably lacks `le`, such as `sum by (job) (rate(x_bucket[5m]))`, is rejected at lowering. Prometheus returns an empty vector for it. Candidate search keeps the classic form as one -exact `KeepPreAsap` sub-DAG for every accuracy target; it has no sketch +retained exact ordinary sub-DAG for every accuracy target; it has no sketch candidate. `histogram_quantiles` lowers each branch the same way; the Fallback compiler accepts its `Concat` of relabeled branches and rejects duplicate output label sets. Nested aggregation, such as @@ -208,8 +209,8 @@ In order of backend usage: After these shapes are covered, the backend can delete rows 28 and 30. 2. Rows 25 and 27: constant weights and `EntityIdentity` items for precompute `SummaryAgg`. -3. Row 16: a label-map sketch-state readout, the counterpart of - `compile_exact_readout`, and MetricsQL `__name__` retention rules. +3. Row 16: a label-map sketch-state evaluation, the counterpart of + `compile_exact_evaluation`, and MetricsQL `__name__` retention rules. 4. Row 20: summary join, subtract, and delete. `fill`, `fill_left`, and `fill_right` matching modifiers are rejected by the diff --git a/docs/develop_docs/physical-handoff-costs.md b/docs/develop_docs/physical-handoff-costs.md index 29525f2c5..1ccf3a2a8 100644 --- a/docs/develop_docs/physical-handoff-costs.md +++ b/docs/develop_docs/physical-handoff-costs.md @@ -1,6 +1,6 @@ # Physical handoff byte estimates -`asap_types::resources` owns the canonical `PhysicalHandoffBytes` and +`asap_types::workload::resources` owns the canonical `PhysicalHandoffBytes` and `PhysicalHandoffKind` definitions in `resources/physical_handoff.rs`. The mapping crate re-exports those same types from `physical_handoff_cost` for import compatibility; all estimator and export consumers therefore use shared definitions, not copies. @@ -88,13 +88,13 @@ traffic is inferred from logical edges, operator buffers, or scan bytes. Unknown endpoints, mismatched payloads, absent node evidence, duplicate IDs, stale evidence, invalid coefficients, and integer overflow return typed errors; ranking/export report the comparison as unavailable. This extends the physical -plan adapter; lifecycle-specific summary-maintenance costing and caching are -separate follow-up integration points. +plan adapter; summary-maintenance costing for Stage 2 materialization (#509) is +a separate follow-up integration point. Verification: ```sh -cargo test -p asap-aware-mapping --test physical_handoff_cost +cargo test -p asap-plan-selection --test physical_handoff_cost cargo test -p asap-devtools --bin dag_export handoff_bytes_export_and_change_plan_selection python3 -m unittest discover -s tools/dag-viewer -p 'test_render.py' ``` diff --git a/docs/develop_docs/planner-vocabulary-migration.md b/docs/develop_docs/planner-vocabulary-migration.md index ea46e992c..d4b29fecd 100644 --- a/docs/develop_docs/planner-vocabulary-migration.md +++ b/docs/develop_docs/planner-vocabulary-migration.md @@ -32,8 +32,8 @@ names. | Physical evidence/comparison `boundaries` fields | `handoffs` | | `BoundaryEstimate::per_boundary` | `PhysicalHandoffEstimate::per_handoff` | | Internal `Models` | `CandidatePlanningInputs` | -| `SketchAlgorithmStrategy::with_models` | `SketchAlgorithmStrategy::new_with_planning_inputs` | -| `SketchAlgorithmStrategy::with_models_and_evidence` | `SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence` | +| `ASAPStrategies::with_models` | `ASAPStrategies::new_with_planning_inputs` | +| `ASAPStrategies::with_models_and_evidence` | `ASAPStrategies::new_with_planning_inputs_and_evidence` | | `HydraGroupingStrategy::with_models_and_evidence` | `HydraGroupingStrategy::new_with_planning_inputs_and_evidence` | For example, `Binder::new().bind(&dag)` becomes diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index abf5dc50b..0f169eca4 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -2,7 +2,15 @@ This is the detailed node reference. Start with the [Pre-ASAP IR concept](../design_docs/concepts/pre-asap-ir.md) for purpose and the compact catalog. -The goal of the pre-ASAP IR is represent operations from different query languages in a single representation, and make it easier to analyze how/where ASAP primitives can be used. +ASAPPlanner has **one operator IR before and after ASAP optimization**, defined in +`crates/types/src/ir/`. "Pre-ASAP" is not a separate type: it is this IR as a front end +emits it, before any ASAP operator has been introduced. This document covers what every +plan shares — the node, the schema, scalar expressions, how front ends produce the DAG, and +the catalog of ordinary (`NonASAPOp`) operators. The ASAP operators, execution timing and +the exported wire form are described in the [Post-ASAP IR](../design_docs/concepts/post-asap-ir.md) +document; the two do not repeat each other. + +The goal of the pre-ASAP form is to represent operations from different query languages in a single representation, and make it easier to analyze how/where ASAP primitives can be used. Only operations that are semantically relevant to answering the query and selecting an ASAP primitive need to become first-class nodes here. ## Design principles @@ -13,7 +21,135 @@ Only operations that are semantically relevant to answering the query and select > Notes: **SQL and PromQL use different schema models**. SQL typically uses a closed schema, where tables, columns, and types are predefined, while PromQL uses an open (schemaless) schema, where metrics and labels can evolve without a fixed table schema. Closed schemas provide stronger structure and validation; open schemas provide greater flexibility and makes it easier to evolve or ingest diverse data, but can require more care around naming conventions, label cardinality, and query consistency. -The pre-ASAP IR is defined using the `QueryExpr` enum. We discuss some of important enum types below. +## The node + +A plan is a DAG of `Rc` (`crates/types/src/ir/operator/node.rs`). Nodes are immutable +and shared through `Rc`: a structurally identical sub-DAG referenced from several parents is +one node, and that pointer identity is what CSE, target discovery and plan assembly key on. + +```rust +pub struct OperatorNode { + pub operator: Operator, // NonASAP(NonASAPOp) | ASAP(ASAPOp) + pub result_kind: OperatorResultKind, // Relation | InstantVector | RangeVector | State | Scalar + pub schema: Schema, // output schema, derived at construction + pub guarantee: Option, // None until accuracy assessment establishes one + pub timing: Option, // None until a materialization assignment is applied +} +``` + +- `operator` is the operation. A front-end DAG contains only `Operator::NonASAP` nodes; + `OperatorNode::expect_non_asap()` relies on that. +- `result_kind` is the output category, derived from the operator and its inputs. Matching + column schemas do not make categories interchangeable (a range vector is not an instant + vector). +- `schema` is derived by `OperatorNode::new(operator)`; it fails when the schema cannot be + derived (a column reference out of range, a reserved ASAP operator). ASAP planning may + retain a more specific schema through `OperatorNode::with_schema`. +- `guarantee` is `None` until accuracy assessment establishes one; `None` never means exact. +- `timing` is `None` in every front-end DAG and every candidate. It is written by + `ir::properties::timing::apply_materialization_timings` (see the Post-ASAP IR document); export rejects an + untimed node. + +`OperatorNode::children()` returns the operator's inputs in field order followed by the +operator nodes its scalar expressions read (see "Scalar expressions"). Every DAG traversal — +`map_children`, `reachable`, `contains_asap`, CSE, export — follows that same list. +`OperatorNode::validate_structure()` checks every operator's input contract, scalar typing +against the owning operator's input schema, and that each retained schema agrees with the +derived one. + +## Schema + +One `Schema` type (`crates/types/src/ir/schema/mod.rs`) describes every edge, whether it +carries rows or summary state: + +```rust +pub struct Schema { + pub fields: Vec, // positional; every ColumnId indexes into this + pub time_index: Option, // the time axis, if any (PromQL leaves always have one) + pub unique_keys: Vec>, + pub closed: bool, // true: these are all the columns; false: open (schemaless) superset +} + +pub struct Field { + pub name: String, + pub dtype: FieldDataType, // Plain(DataType) | ExactAggregate(..) | Sketch(..) | Sample(..) | Wavelet(..) | StatModel(..) + pub nullable: bool, + pub table: Option, // SQL table/alias qualifier; None for PromQL labels +} +``` + +A pre-ASAP field is always `FieldDataType::Plain(DataType)`. The other variants carry summary +state and only appear below an ASAP operator; a scalar expression that reads such a field is a +typing error (`ScalarExpr::scalar_type`), because state must be read out before a value can use +it. Column references are positional `ColumnId`s (indexes into the input schema), never names. + +`Schema::has_unique_key()` is the legality gate CSE uses: a non-ASAP producer is only shared +across consumers when its row identity is provable. + +## Scalar expressions + +Value computation lives in `ScalarExpr` (`crates/types/src/ir/scalar/mod.rs`), owned **by value** +by an operator field: `Scan.predicates`, `Filter.pred`, `Join.pred`, `Project.cols[i].expr`, +`Aggregate.having`, `Sort.keys[i].expr`, `SQLWindowFunc.args`/`order_by`, `PromqlRelabel.value`, +`Values.rows`, and `QueryRoot::Scalar` and `PromqlVectorFromScalar`. A scalar expression never +produces a table and is never a node of the DAG; it is evaluated against the input schema of +the operator that owns it. + +Variants: `Column(ColumnId)`, `Literal(ScalarValue)`, `Negative` (unary minus), `Compare`, +`BoolAnd` / `BoolOr` (flat conjunction/disjunction), `Not`, `IsNull` / `IsNotNull`, `Cast` +(with `try_cast`), `InList`, `FunctionCall { name, args }`, `Arithmetic`, `Case`, +`CurrentTimestamp` (SQL `NOW()`), `EvalTimestamp` (PromQL `time()`), and four +**plan-reading** variants that reference an operator node: + +| Variant | Meaning | +|---|---| +| `PromqlScalarFromVector(Rc)` | PromQL `scalar(v)`: the single sample of an instant vector, NaN otherwise | +| `ScalarSubquery(Rc)` | Uncorrelated SQL scalar subquery: one column; zero rows is NULL, more than one row is an error | +| `Exists { subquery, negated }` | SQL `[NOT] EXISTS (subquery)` | +| `InSubquery { expr, subquery, negated }` | SQL `expr [NOT] IN (subquery)` over a one-column relation | + +These are the **only** operator references inside a scalar tree. `ScalarExpr::operator_refs()` +lists them, `NonASAPOp::children()` appends them after the operator's own inputs, and +canonicalization lowers the three SQL subquery forms to joins (see below), so a canonical SQL +DAG contains none of them. `PromqlScalarFromVector` survives canonicalization: its referenced +vector is a real plan dependency, exported as a `ScalarRef` edge. + +`Compare`, `Arithmetic` and `Negative` carry an `ExprSemantics` (`Sql` or `Promql`): both +languages use `Float64`, so the result type alone does not preserve NaN, ordering or error +rules, and the executing engine needs to know which language's rules apply. + +Wrapper types: `Predicate(ScalarExpr)`, `ProjectItem { alias, expr }`, +`SortKey { expr, ascending, nulls_first }`. + +## How a front end produces the DAG + +A front end never constructs `OperatorNode`s directly. It builds a name-based tree in +`crates/frontend-common` — `UnresolvedOp` / `UnresolvedScalar`, a mirror of `NonASAPOp` / +`ScalarExpr` in which every column reference is a `ColumnRef` and a PromQL `Scan` has no schema +yet — and calls `asap_frontend_common::resolve_root`, which does three things in order: + +1. **Resolution** — a bottom-up walk that binds every `ColumnRef` to a positional `ColumnId` + against the derived schema of the already-resolved child. A schemaless (PromQL) leaf gets + its binding schema from `SchemaResolver`, built from the names the query references. + `Join` / `SetOp` sides and the operators referenced from scalar positions are each bound as + a root in their own scope; a `BinaryOp` side additionally inherits the label names its + enclosing scope references. +2. **Schema derivation** — each `OperatorNode::new` derives the node's output schema and + result kind from the operator and its children. +3. **Canonicalization** — `asap_types::ir::canonicalize::canonicalize` erases structural + differences between semantically identical queries: it promotes an additive + `Limit { Sort { Aggregate } }` ranking to the `AggIntent::TopK` heavy-hitter shape, and + lowers `EXISTS` / `NOT EXISTS` / `IN (subquery)` predicates to `Join { Semi | Anti }` and a + scalar subquery to a `Join { Cross }` plus column reference. The pass is idempotent and + keeps the pointer identity of every untouched sub-DAG. + +The result is `Rc`. `lower_promql_workload`, `lower_sql` / `lower_sql_dialect` / +`lower_sql_batch` and `lower_metricsql` all return it. + +Workload search then runs structural CSE (`asap_types::ir::cse::share_common_sub_dags`) once +across every root: bottom-up hash-consing where the structural hash is only a filter and the +typed `PartialEq` decides sharing, following scalar references like any other input, and +gated by `Schema::has_unique_key()` for non-ASAP producers. ## Fields and column references @@ -47,27 +183,28 @@ to one source language. - [`Aggregate`](#aggregate) — collapses input rows into fewer output rows via a reduction and aggregate intents. **[Time-related nodes](#time-related-nodes)** -- [`TimeRange`](#timerange) — a range-vector lookback over the time axis (PromQL `[5m]`). +- [`TimeRange`](#timerange) — temporal selection over a time-series input (PromQL instant lookback or `[5m]` range selector). - [`TimeShift`](#timeshift) — shifts *when* a selector is evaluated (PromQL `offset`/`@`). - [`PromqlSubquery`](#promqlsubquery) — re-evaluates an instant-vector expression over a range at a given step. **[Relational nodes](#relational-nodes)** — common to both SQL and PromQL - [`Scan`](#scan) — identifies the logical data source. +- [`Values`](#values) — SQL `VALUES` rows, or the one empty row of a `SELECT` without `FROM`. - [`Filter`](#filter) — restricts rows using a predicate. - [`Project`](#project) — column projection (SQL `SELECT` list). -- [`BinaryOp`](#binaryop) — arithmetic / comparison / boolean composition of two inputs. +- [`BinaryOp`](#binaryop) — arithmetic / comparison / set composition of two inputs. - [`Sort`](#sort) — generic (non-heavy-hitter) order-by, optionally per-group. -- [`Limit`](#limit) — caps the row count, with an offset. +- [`Limit`](#limit) — caps the row count, with an offset, optionally per-group. - [`Dedup`](#dedup) — row-level deduplication. - [`Join`](#join) — logical join of two inputs. - [`SetOp`](#setop) — SQL's typed set operations (`UNION`/`INTERSECT`/`EXCEPT`). - [`Concat`](#concat) — exact, untyped `UNION ALL` of union-compatible branches. -**[PromQL-specific nodes](#promql-specific-nodes)** -- [`PromqlScalarBridge`](#promqlscalarbridge) — a scalar sub-expression at an operator-DAG position. -- [`EvalTimestamp`](#evaltimestamp) — the query evaluation time as a scalar (PromQL `time()`). +**[Scalar-position nodes](#scalar-position-nodes)** +- `QueryRoot::Scalar` — a standalone scalar expression, without an operator node. - [`PromqlVectorFromScalar`](#promqlvectorfromscalar) — promotes a scalar to a label-less instant vector. -- [`PromqlScalarFromVector`](#promqlscalarfromvector) — collapses a single-series vector to a scalar. + +**[PromQL-specific nodes](#promql-specific-nodes)** - [`PromqlRelabel`](#promqlrelabel) — per-series label rewrite (PromQL `label_replace`/`label_join`). - [`PromqlInfoEnrich`](#promqlinfoenrich) — left-join label enrichment from an info metric. - [`PromqlSeriesSample`](#promqlseriessample) — keeps a subset of whole series, not a reduction. @@ -75,6 +212,9 @@ to one source language. **[SQL-specific nodes](#sql-specific-nodes)** - [`SQLWindowFunc`](#sqlwindowfunc) — SQL analytic window function (`OVER (...)`). +PromQL `time()` and `scalar(v)` are scalar expressions (`ScalarExpr::EvalTimestamp`, +`ScalarExpr::PromqlScalarFromVector`), not nodes. + ## Aggregation-related nodes ### Aggregate @@ -109,7 +249,7 @@ list of aggregate intents (`measures`). value is still recomputed by the agg intent, e.g. `Rate`), for a computation with no `by(...)` clause to attach to. `PerEntity` is different from `by` for all columns, because in PromQL, it is schemaless and you don't know all columns beforehand. E.g. PromQL `rate(http_requests_total[5m])`, which has one rate value - per input series: + per input series: ```text Aggregate( @@ -117,7 +257,7 @@ list of aggregate intents (`measures`). measures = [Rate], output_names = [], having = None, - child = TimeRange(range = 5m, child = Scan("http_requests_total")) + child = TimeRange(range = 5m, kind = Range, child = Scan("http_requests_total")) ) ``` @@ -224,7 +364,7 @@ Example for `filters`: `count(CASE WHEN p THEN x END)` (`p`, plus `x IS NOT NULL` when `x` is nullable), and from `count(expr)` over any other nullable `expr` (`expr IS NOT NULL`), because canonical `Count` counts rows and never consults its argument. A filtered measure has no summary binding yet: - `asap-aware-mapping` keeps such an `Aggregate` as `KeepPreAsap`, and canonicalization does + `asap-logical-optimizer` retains such an `Aggregate` as an ordinary exact sub-DAG, and canonicalization does not promote a filtered count ranking to a heavy-hitter `TopK`. Example for `having`: @@ -248,8 +388,8 @@ Example for `having`: **Rules/Invariants**: A filtering predicate will be passed to at the lowest node (closer to the leaves) in the AST/DAG that can express it — `Scan.predicates`, then `Aggregate.having`, then `Filter` as the fallback — so its constraint is visible at - the node it actually applies to, not behind an opaque wrapper, once pre-ASAP IR translates - to post-ASAP IR with summary binding. The upper nodes (closer to the root) in the AST/DAG can still have a `Filter` node with the same condition. This intentional duplication is for Summary related translation and optimizations. + the node it actually applies to, not behind an opaque wrapper, once summary binding reads it. + The upper nodes (closer to the root) in the AST/DAG can still have a `Filter` node with the same condition. This intentional duplication is for Summary related translation and optimizations. For example, `SELECT srcip, COUNT(*) AS cnt FROM packets GROUP BY srcip HAVING COUNT(*) > 10` pins `cnt > 10` to the lowest node that can express it, `Aggregate.having`: @@ -282,7 +422,7 @@ Example for `having`: Both are valid at once, and neither is derived from the other: `having` is the canonical spot a summary-aware pass reads to decide whether `Aggregate` can bind to a summary, while the outer `Filter` is what a plain logical evaluator runs without knowing `having` exists. The duplication is forward-looking groundwork for - once HAVING-aware summary binding (pre-ASAP-IR to post-ASAP-IR translation) lands. + once HAVING-aware summary binding lands. Neither direction of that push-down is enforced yet: the SQL front end doesn't populate `having` from a real `HAVING` clause (#201), and canonicalization doesn't fold an existing @@ -294,14 +434,21 @@ Example for `having`: ### TimeRange -Represents a range of time. Kept different from `Filter` to treat time as an explicit concern. +Temporal selection over a time-series input. Kept different from `Filter` to treat time as an +explicit concern. `kind` records which samples a PromQL selector reads: + +- `TimeRangeKind::Instant` — an instant selector: `range` is the lookback horizon and the + latest eligible sample per series is selected (the planner injects the declared + `data_ingestion_interval` around a bare selector). +- `TimeRangeKind::Range` — a range selector (`m[5m]`): every sample in the window. ```promql rate(http_requests_total[5m]) ``` **Fields:** -- `range` — how far back to look (the PromQL `[5m]` duration). +- `range` — how far back to look (the PromQL `[5m]` duration, or the instant lookback). +- `kind` — `Instant` or `Range`. - `child` — the input the range applies to. ### TimeShift @@ -350,9 +497,19 @@ the same logical data domain. **Fields:** - `source` — the logical data source (a table name or PromQL metric selector). - `predicates` — row-level filters pushed all the way down to this scan (Rules/Invariants - rule 1); enforced structurally at lowering time — a `Filter` directly over a `Scan` never - survives. -- `schema` — the binding schema every positional column reference in the DAG resolves against. + rule 1): PromQL label matchers and pushed-down `WHERE` conjuncts. +- `schema` — the binding schema every positional column reference in the tree resolves against. + A catalog-backed SQL leaf carries its catalog schema; a PromQL leaf carries the usage-derived + schema `SchemaResolver` built from the labels the query references. + +### Values + +SQL `VALUES` rows, or the one empty row of a `SELECT` without `FROM` +(`SELECT 1 + 1`). Row expressions have no input-column scope. + +**Fields:** +- `rows` — one `Vec` per row. +- `schema` — the output schema of the rows. ### Filter @@ -386,6 +543,10 @@ that's neither a base scan column nor an aggregate output: SELECT * FROM (SELECT srcip, bytes_in + bytes_out AS total FROM packets) t WHERE total > 500 ``` +A `Filter` whose predicate contains `EXISTS` / `NOT EXISTS` / `IN (subquery)` does not +survive canonicalization: the conjunct becomes a `Join { Semi | Anti }` under the remaining +predicate. + **Fields:** - `pred` — the row-level predicate to apply. - `child` — the input being filtered. @@ -406,18 +567,21 @@ SELECT srcip, dstip FROM packets ### BinaryOp -Arithmetic / comparison / boolean composition. PromQL binary operators between two vectors, -a vector and a scalar, or two scalars. +Arithmetic / comparison / set composition of two operands. PromQL binary operators between two vectors, +two vectors. Mixed vector/scalar arithmetic uses `Project`; non-bool comparison uses `Filter`. Standalone scalar expressions are `QueryRoot::Scalar`. ```promql up > 1 ``` **Fields:** -- `op` — the arithmetic/comparison/boolean operator. +- `operator` — a `BinaryOperator { kind, vector_match, checked_relative_division, checked_finite_division }`: + - `kind` — `BinaryOpKind::Arithmetic(..)`, `Compare(..)` or `Set(..)` (PromQL `and`/`or`/`unless`). + - `vector_match` — PromQL vector-matching modifiers (`on`/`ignoring`, `group_left`/`group_right`); `None` outside PromQL and the only supported value today. + - `checked_relative_division` / `checked_finite_division` — typed division guards set by summary planning, never by a front end (see [physical-plan integration](../design_docs/architecture/physical-plan-integration.md#conditional-temporal-average-lowering)). +- `return_bool` — the PromQL `bool` modifier: a comparison returns `0`/`1` instead of filtering. Valid only for comparison operators. - `lhs` — the left operand. - `rhs` — the right operand. -- `vector_match` — PromQL vector-matching modifiers (`on`/`ignoring`, `group_left`/`group_right`); `None` outside PromQL. ### Sort @@ -429,7 +593,7 @@ sort_desc(up) ``` **Fields:** -- `keys` — the ordering columns/expressions and direction. +- `keys` — the ordering expressions and direction (`SortKey`). - `partition_by` — grouping keys that make the ordering per-group instead of global; empty = a single global order. - `child` — the input being ordered. @@ -443,8 +607,9 @@ topk(3, up) ``` **Fields:** -- `n` — the maximum number of rows to keep. +- `n` — the maximum number of rows to keep; `None` is offset-only. - `offset` — how many leading rows to skip first. +- `partition_by` — applies the limit per group (PromQL `topk by (..)`); empty = global. - `child` — the input being capped. ### Dedup @@ -463,14 +628,16 @@ SELECT DISTINCT srcip, dstip FROM packets ### Join -Logical join; the physical strategy (hash/merge/broadcast) is picked in the post-ASAP IR. SQL `JOIN`. +Logical join; the physical strategy (hash/merge/broadcast) is picked downstream of the planner. SQL `JOIN`, +and the shape canonicalization lowers subqueries to. ```sql SELECT u.prefix FROM bgp_updates u JOIN bgp_rib_state r ON u.prefix = r.prefix ``` **Fields:** -- `kind` — the join type (inner/left/right/full/semi/anti). +- `kind` — the join type (`Inner`/`Left`/`Right`/`Full`/`Cross`/`Semi`/`Anti`). A semi/anti join + outputs the left input's columns alone, but its predicate resolves against `left ++ right`. - `pred` — the join condition. - `left` — the left input. - `right` — the right input. @@ -496,7 +663,7 @@ SELECT srcip FROM packets UNION ALL SELECT dstip FROM packets never dedup. Used when a single `Aggregate` can't express the shape — the canonical case is PromQL `histogram_quantiles` (one branch per φ, each its own `HistogramQuantile` reduction relabeled with its `le` value) — and SQL `ROLLUP`/`CUBE`/`GROUPING SETS` (one branch per -grouping level). +grouping level). The output schema is the first child's. ```promql histogram_quantiles(rate(http_request_duration_seconds_bucket[5m]), "le", 0.5, 0.9) @@ -504,39 +671,32 @@ histogram_quantiles(rate(http_request_duration_seconds_bucket[5m]), "le", 0.5, 0 **Fields:** - `children` — the union-compatible branches to concatenate; must be non-empty. +- `discriminator_unique_key` — an optional caller-proven compound unique key + `(discriminator, inner_key)` over the output; nothing verifies the claim. -## PromQL-specific nodes - -### PromqlScalarBridge +## Scalar-position nodes -A scalar sub-expression (issue #220: in practice always `Literal(ScalarValue::Float64(_))` — -a PromQL number literal, or a folded constant scalar expression) sitting at an **operator-DAG -position** — a `BinaryOp` operand for ` op ` thresholds and unit conversions, -a `PromqlVectorFromScalar` child, or a whole query's root. This wrapper is what marks the -position; it no longer duplicates `Literal`'s value the way the old `PromqlScalar(f64)` variant -did. - -```promql -up > 1 -``` +### Scalar query roots -**Fields:** a single unnamed child `QueryExpr` — the wrapped scalar sub-expression. +`QueryRoot` distinguishes an operator result from an owned `ScalarExpr`. It is +an API root discriminator, not an operator. `2`, `time()`, and +`scalar(sum(up)) + 1` therefore introduce no constant-wrapper nodes. -### EvalTimestamp +Use `lower_promql_query_workload` for mixed scalar/vector workloads. The +operator-only convenience API rejects standalone scalar roots. `ParsedWorkload` +retains each scalar's workload index; `PlanOutput::roots()` returns all results +in workload order. Scalar plan reads remain exact and retain their operator +references; summary selection currently operates on operator roots. -The query **evaluation timestamp** as Unix seconds — PromQL `time()` — and the implicit -input of the no-argument calendar functions (`hour()`, `day_of_week()`, ...). It is the -instant or range-step at which the expression is evaluated, not inherently the current -wall-clock time. The Prometheus instant-query HTTP API separately defaults an omitted -`time` request parameter to the server's current time. - -```promql -time() -``` +`up * 2` projects the sample expression while retaining time and full series +identity, removing the metric name. `up > 0` and `0 < up` filter the vector and +retain its sample and name. `up > bool 0` projects a zero-or-one `Case`. +Open label schemas acquire a full runtime series-identity field before this +lowering. The runtime must populate that field with all labels. ### PromqlVectorFromScalar -The scalar→instant-vector bridge — PromQL `vector(s)`. Promotes a scalar-typed child to a +The scalar→instant-vector bridge — PromQL `vector(s)`. Promotes a scalar expression to a single label-less series carrying that value at every step, e.g. for dead-man's-switch patterns (`up or vector(0)`). @@ -544,18 +704,9 @@ patterns (`up or vector(0)`). vector(1) ``` -**Fields:** a single unnamed child `QueryExpr` — the scalar-typed expression being promoted to a vector. - -### PromqlScalarFromVector +**Fields:** a single unnamed `ScalarExpr` — the scalar being promoted to a vector. -The instant-vector→scalar bridge — PromQL `scalar(v)`. Collapses a single-element vector to -its value (NaN at runtime if the input isn't exactly one series). - -```promql -scalar(up) -``` - -**Fields:** a single unnamed child `QueryExpr` — the single-series vector being collapsed to a scalar. +## PromQL-specific nodes ### PromqlRelabel @@ -616,5 +767,6 @@ SELECT srcip, LAG(time) OVER (PARTITION BY srcip ORDER BY time) FROM packets - `args` — the function's operand expressions; empty for rank-only functions. - `partition_by` — grouping keys the window is computed within. - `order_by` — the ordering the window function reads. +- `frame` — the optional window frame. - `output_name` — the name of the new output column. - `child` — the input the window function is computed over. diff --git a/docs/develop_docs/replacement-explanations.md b/docs/develop_docs/replacement-explanations.md index 4cd97300a..caefd00f3 100644 --- a/docs/develop_docs/replacement-explanations.md +++ b/docs/develop_docs/replacement-explanations.md @@ -24,4 +24,4 @@ Additional opportunities: - reuse finer-grained aggregation through roll-up ``` -Implemented as `asap-aware-mapping`'s `explanation` module (`explain_replacements`/`explain_replacements_with`, issue #257) +Implemented as `asap-logical-optimizer`'s `pass1::explanation` module (`explain_replacements`/`explain_replacements_with`, issue #257) diff --git a/docs/develop_docs/storage-operation-costs.md b/docs/develop_docs/storage-operation-costs.md index 1a8dacdbd..e9f1f0e3e 100644 --- a/docs/develop_docs/storage-operation-costs.md +++ b/docs/develop_docs/storage-operation-costs.md @@ -43,9 +43,9 @@ profile and coefficients before ranking. Scan read extents must add up to the scan's authoritative `source_read_bytes`; explicit additional storage actions can be bound to other physical nodes. -Their sole data type is `asap_types::resources::StorageResources`, defined in +Their sole data type is `asap_types::workload::resources::StorageResources`, defined in the shared resources module alongside CPU and byte dimensions. The mapping -crate re-exports it at `asap_aware_mapping::storage_io::StorageResources` for +crate re-exports it at `asap_plan_selection::cost::storage_io::StorageResources` for source compatibility; the four integer JSON fields are unchanged. Pure term enumeration and checked addition live with the shared type. Access profiles, request-count estimation, calibration, and ranking remain in the mapping @@ -71,8 +71,7 @@ estimate and storage request estimate remain independently inspectable. Missing entries, expired/future evidence, incompatible node snapshots, zero request sizes, invalid calibration, and overflow return typed analytical errors. When used by plan ranking/export they make that comparison unavailable. -This profile extends the physical-plan adapter; the separate summary-maintenance -lifecycle estimator retains its existing dimensions. Combined physical-plan +This profile extends the physical-plan adapter. Combined physical-plan ranking currently supports storage profiles only with an explicit `NoCache` profile. `CacheProfile::Evidence` together with storage evidence makes the comparison unavailable: aggregate cache hit ratios cannot identify which @@ -83,7 +82,7 @@ cache-adjusted bytes and CPU with uncached operation counts. Verification: ```sh -cargo test -p asap-aware-mapping --test storage_io -cargo test -p asap-aware-mapping --lib storage_io +cargo test -p asap-plan-selection --test storage_io +cargo test -p asap-plan-selection --lib storage_io cargo test -p asap-devtools --bin dag_export storage_requests_export_and_change_plan_selection ``` diff --git a/docs/develop_docs/target-candidate-api-migration.md b/docs/develop_docs/target-candidate-api-migration.md index a5b081a61..4619be2a8 100644 --- a/docs/develop_docs/target-candidate-api-migration.md +++ b/docs/develop_docs/target-candidate-api-migration.md @@ -13,7 +13,7 @@ are unchanged. #453 separately defines the integration API surface. | `MaterializeSummaryMaintenanceLifecycleError` | `SummaryMaintenanceLifecycleAssemblyError` | Failure assembling a DAG or deriving maintenance decisions | | Error variant `Materialize` | `AssembleDAG` | Wrap an underlying `RealizationError` from DAG assembly | | Internal `materialize_inner` / `materialize_residual` | `assemble_target` / `assemble_residual` | Assemble selected nodes, not runtime materialized views | -| Internal assembly cache `materialized` | `assembled_nodes` | Preserve shared `Rc` identity | +| Internal assembly cache `materialized` | `assembled_nodes` | Preserve shared node identity (now `Rc`, see below) | Update imports and calls together; old public names are not retained as aliases. Downstream Rust integrations using these symbols must migrate. No serialized @@ -24,5 +24,28 @@ The earlier #445 renames (`TargetSubDAGCandidates`, counterpart) are prerequisites, not additional changes here. The workflow remains one selection call per workload followed by one assembly -call per query root. `SummaryMaintenanceLifecyclePlan` contains the assembled -Post-ASAP DAG root plus maintenance decisions; it is not an executable plan. +call per query root. The summary-maintenance lifecycle API named above was later +removed; Stage 2 materialization (#509) will own maintenance decisions. + +## Later: unified operator IR (operator flattening) + +The pre-ASAP and post-ASAP trees became one IR in `asap_types::ir`. Every +node is an `Rc` whose `operator` is `Operator::NonASAP(NonASAPOp)` +or `Operator::ASAP(ASAPOp)`. Old public names are not kept as aliases. + +| Old | New | +|---|---| +| `Rc` (pre-ASAP) | `Rc` holding `Operator::NonASAP(NonASAPOp)` | +| `Rc` / `SummaryExpr` (post-ASAP) | The same `Rc`; summary steps are `Operator::ASAP(ASAPOp)` | +| `SummaryExpr::KeepPreAsap(q)` | The non-ASAP sub-DAG itself; `retain_exact` only adds an exact `guarantee` | +| `SummaryExpr::ValueOperation { .. }` over a evaluation | An ordinary `NonASAPOp` (`Project`, `Filter`, `Sort`, `Limit`, `Aggregate`) reading an ASAP node; `FinalizeExactAccumulator`, `MaintainPopulation`, `EvaluatePopulation` are `ASAPOp` variants | +| `Replacement::Summary(..)` / `Replacement::Rewrite(..)` | `Replacement::SubDAG(Rc)`; `is_logical_rewrite` tells them apart | +| `SummaryFamilyType` | `FieldDataType` (its non-`Plain` variants) | +| Timing stored on post-ASAP nodes | `OperatorNode::timing`, `None` until `ir::timing::apply_materialization_timings` writes it from a `MaterializationAssignment` (default: all query time) | +| `UnresolvedQueryExpr` + `asap_types::pre_asap::resolve_root` | `UnresolvedOp` / `UnresolvedScalar` + `asap_frontend_common::resolve_root` | +| `pre_asap::canonicalize`, `pre_asap::cse::share_common_sub_dags` | `ir::canonicalize::canonicalize`, `ir::cse::share_common_sub_dags` | +| `asap_types::post_asap::compile_post_asap_dag` (wire version 5, `Fallback`/`Binary`/`Value` payloads) | `asap_types::ir::export::compile_post_asap_dag` (wire version 7: one node per operator, `Relational` payloads, `ScalarRef` edges); input must be timed | +| Exported schema JSON `columns` | `fields` | + +Field and schema details: [Pre-ASAP IR](pre-asap-ir.md) and +[Post-ASAP IR](../design_docs/concepts/post-asap-ir.md). diff --git a/docs/user_guide_docs/run-a-query.md b/docs/user_guide_docs/run-a-query.md index 3f5825c29..c6f52aa61 100644 --- a/docs/user_guide_docs/run-a-query.md +++ b/docs/user_guide_docs/run-a-query.md @@ -6,7 +6,7 @@ corpus coverage. These commands do not deploy or execute a physical plan. To develop an application using the Rust library, start with [Library API: definitions, options, and examples](../develop_docs/library-api.md). That guide explains how to choose strategies and models, rank candidates, and -work with lifecycle capabilities. +assemble selected DAGs. ## Choose a command @@ -96,7 +96,7 @@ cargo run -p asap-devtools --bin show_post_asap_ir -- --data-ingestion-interval- available binding from the sketch strategy for each query, numbered in cost-model order. If no candidate is available, it prints the pre-ASAP fallback as candidate 1. It does not show the complete ranked workload candidate set or choose a -deployment lifecycle. Its SQL examples use a fixed demonstration catalog, not +deployment. Its SQL examples use a fixed demonstration catalog, not your database schema. Use the [library workflow](../develop_docs/library-api.md) to retain workload alternatives and provide your own models. @@ -109,8 +109,10 @@ whether to select them using its own evidence. Planner's automatic `global_selection` skips them; their presence alone does not show that they meet the requested target. -Each input line is followed by its debug IR or an `ERR:` message. Post-ASAP -output may contain summary state, readouts or exact `KeepPreAsap` work. An +Each input line is followed by its debug IR or an `ERR:` message. Pre-ASAP and +Post-ASAP output use the same node format: Post-ASAP output adds summary nodes +(state, readouts) and keeps the original exact operators wherever no summary +replaces them. An approximate target permits approximation; it does not guarantee a legal or certified sketch. The tool prints plans, not query results. diff --git a/tools/dag-viewer/README.md b/tools/dag-viewer/README.md index b90aae272..182c0f94c 100644 --- a/tools/dag-viewer/README.md +++ b/tools/dag-viewer/README.md @@ -69,7 +69,8 @@ cargo run -p asap-devtools --bin dag_export -- \ Load the JSON with the page's file picker. `--planner-cost-json` is a complete physical-evidence document: an immutable `evidence_version`, calibration, and -target records containing the exact target `QueryExpr` and comparison scope. +target records containing the exact target node (a serialized pre-ASAP +`OperatorNode`) and comparison scope. Each exact replacement candidate owns its complete logical-node `PhysicalNodeEvidence`; summary candidates additionally own their bound `PhysicalDAG`. Candidate-local evidence prevents statistics for one physical @@ -101,15 +102,6 @@ to calibrate against, and `--planner-cost-json` once there is. Without either flag, `--post-asap` exports the raw DAG only. -The viewer also accepts the JSON produced by -`export_summary_maintenance_plan`. It renders the materialized summary DAG as -a single lifecycle-plan lane. Selecting a `SummaryAgg` shows the chosen -lifecycle and maintenance mode together with every alternative's cost, -assumptions, and rejection reason. The selected-node panel also shows the -plan-level summary-versus-raw decision, costs, horizon, expected reads, and -evaluation/update rates. Raw-recomputation plans retain that decision summary -even though they have no deployed `SummaryAgg` to annotate. - ## Standalone HTML ```sh @@ -129,7 +121,7 @@ a selected replacement directly contains: { "decision": { "id": 7, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -147,8 +139,13 @@ The exporter assigns `workload_node_id`; union rendering reads that mapping directly. Node boxes use concrete IR fields: aggregate measures/grouping, sort keys, -filter predicates, projections, sources, summary families, and readout -queries. Category icons are deliberately omitted so they cannot be confused +filter predicates, projections, sources, summary families, and evaluation +queries. A node's `kind` is the operator variant name (`Operator::kind_name`): +a `NonASAPOp` such as `Aggregate` or `Values`, or an `ASAPOp` such as +`SummaryAgg` or `EvaluatePopulation`. `node-style.js` maps each kind to a color +category. Scalar expressions are not nodes; an operator a scalar expression +reads (`scalar(v)`, `EXISTS (subquery)`) is a child node, shown in `detail` +as `{"scalar_ref": }`. Schemas list their entries under `fields`. Category icons are deliberately omitted so they cannot be confused with IR text. ### Cost/benefit annotations (issue #286) diff --git a/tools/dag-viewer/dag.example.json b/tools/dag-viewer/dag.example.json index 85c0e97b6..5249bb412 100644 --- a/tools/dag-viewer/dag.example.json +++ b/tools/dag-viewer/dag.example.json @@ -210,7 +210,7 @@ { "decision_id": 0, "target_pre_id": 1, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -363,108 +363,7 @@ "kind": "Summary", "dag": { "nodes": [ - { - "id": 0, - "kind": "KeepPreAsap", - "label": "KeepPreAsap(Scan)", - "detail": { - "pre_asap_sub_dag": { - "nodes": [ - { - "children": [], - "detail": { - "predicates": [], - "schema": { - "closed": true, - "columns": [ - { - "dtype": "timestamp", - "name": "ts", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "service", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "region", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "float64", - "name": "latency", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "int64", - "name": "bytes", - "nullable": false, - "table": "metrics" - } - ], - "time_index": 0, - "unique_keys": [] - }, - "source": { - "Table": { - "table_ref": "metrics" - } - } - }, - "hash": 2606922452740434172, - "id": 0, - "kind": "Scan", - "label": "Scan(metrics)", - "schema": { - "closed": true, - "columns": [ - { - "dtype": "timestamp", - "name": "ts", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "service", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "utf8", - "name": "region", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "float64", - "name": "latency", - "nullable": false, - "table": "metrics" - }, - { - "dtype": "int64", - "name": "bytes", - "nullable": false, - "table": "metrics" - } - ], - "time_index": 0, - "unique_keys": [] - } - } - ], - "root": 0 - } - }, - "children": [] - }, + {"id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {"predicates": [], "schema": {"closed": true, "columns": [{"dtype": "timestamp", "name": "ts", "nullable": false, "table": "metrics"}, {"dtype": "utf8", "name": "service", "nullable": false, "table": "metrics"}, {"dtype": "utf8", "name": "region", "nullable": false, "table": "metrics"}, {"dtype": "float64", "name": "latency", "nullable": false, "table": "metrics"}, {"dtype": "int64", "name": "bytes", "nullable": false, "table": "metrics"}], "time_index": 0, "unique_keys": []}, "source": {"Table": {"table_ref": "metrics"}}}, "children": []}, { "id": 1, "kind": "SummaryAgg", @@ -657,7 +556,7 @@ "hash": 2606922452740434172, "decision": { "id": 0, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -763,7 +662,7 @@ "workload_node_id": 1, "decision": { "id": 0, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, @@ -862,7 +761,7 @@ "workload_node_id": 2, "decision": { "id": 0, - "strategy": "SketchAlgorithmStrategy", + "strategy": "ASAPStrategies", "rationale": "count realizes as a Cms sketch", "rank": 0, "cost": 1.14001088, diff --git a/tools/dag-viewer/examples/planner-layering-example1.expected.json b/tools/dag-viewer/examples/planner-layering-example1.expected.json new file mode 100644 index 000000000..c655fc33a --- /dev/null +++ b/tools/dag-viewer/examples/planner-layering-example1.expected.json @@ -0,0 +1,1603 @@ +{ + "format": "asap-stage-pipeline/v1", + "expected_only": true, + "_notes": [ + "Shape reference for planner-layering Example 1 (MVP): see docs/design_docs/proposals/planner-layering-example1-acceptance.md.", + "Nodes carry only payload kinds and a label, not full schemas. Real documents have the full LogicalASAPDAG/PhysicalASAPDAG nodes.", + "query_roots stands in for root: one workload DAG serves both queries, and LogicalASAPDAG has a single root.", + "Costs are null: the cost model decides them. The selection is illustrative (the doc's 'Raw with a shared input' winner); tests only require the cheapest valid candidate." + ], + "workload": { + "queries": [ + { + "id": "q1", + "language": "promql", + "text": "sum by (job) (rate(http_requests_total[1m]))", + "repeat_every_s": 10, + "lookback_s": 60, + "accuracy": "exact" + }, + { + "id": "q2", + "language": "promql", + "text": "topk by (job) (10, sum_over_time(http_requests_total[1m]))", + "repeat_every_s": 10, + "lookback_s": 60, + "accuracy": { + "epsilon": 0.01, + "delta": 0.001 + }, + "max_latency_ms": 100 + } + ] + }, + "stage0_logical": { + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)" + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)" + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 6, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum_over_time (per series)" + }, + { + "id": 7, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "topk by (job) (10)" + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + }, + { + "producer": 6, + "consumer": 7 + } + ], + "query_roots": { + "q1": 3, + "q2": 7 + } + } + }, + "stage1_logical_asap": { + "candidates": [ + { + "id": "L1", + "label": "Q1 exact · Q2 exact · separate input", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)" + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)" + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 6, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum_over_time (per series)" + }, + { + "id": 7, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "topk by (job) (10)" + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + }, + { + "producer": 6, + "consumer": 7 + } + ], + "query_roots": { + "q1": 3, + "q2": 7 + } + } + }, + { + "id": "L2", + "label": "Q1 exact · Q2 exact · shared input", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)" + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)" + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum_over_time (per series)" + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "topk by (job) (10)" + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 1, + "consumer": 4 + }, + { + "producer": 4, + "consumer": 5 + } + ], + "query_roots": { + "q1": 3, + "q2": 5 + } + } + }, + { + "id": "L3", + "label": "Q1 exact · Q2 Count-Min + heap · separate input", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)" + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)" + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 6, + "payload": { + "kind": "summary_agg" + }, + "label": "Count-Min + top-k heap, one per job" + }, + { + "id": 7, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job" + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + }, + { + "producer": 6, + "consumer": 7 + } + ], + "query_roots": { + "q1": 3, + "q2": 7 + } + } + }, + { + "id": "L4", + "label": "Q1 exact · Q2 Count-Min + heap · shared input", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)" + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)" + }, + { + "id": 4, + "payload": { + "kind": "summary_agg" + }, + "label": "Count-Min + top-k heap, one per job" + }, + { + "id": 5, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job" + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 1, + "consumer": 4 + }, + { + "producer": 4, + "consumer": 5 + } + ], + "query_roots": { + "q1": 3, + "q2": 5 + } + } + }, + { + "id": "L5", + "label": "Q1 exact · Q2 Hydra · separate input", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)" + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)" + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 6, + "payload": { + "kind": "summary_agg" + }, + "label": "Hydra (CMS) over all (job, series) keys + heap per job" + }, + { + "id": 7, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job" + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + }, + { + "producer": 6, + "consumer": 7 + } + ], + "query_roots": { + "q1": 3, + "q2": 7 + } + } + }, + { + "id": "L6", + "label": "Q1 exact · Q2 Hydra · shared input", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total" + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m" + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)" + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)" + }, + { + "id": 4, + "payload": { + "kind": "summary_agg" + }, + "label": "Hydra (CMS) over all (job, series) keys + heap per job" + }, + { + "id": 5, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job" + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 1, + "consumer": 4 + }, + { + "producer": 4, + "consumer": 5 + } + ], + "query_roots": { + "q1": 3, + "q2": 5 + } + } + } + ] + }, + "stage2_physical_asap": { + "candidates": [ + { + "id": "P1", + "from_logical": "L1", + "label": "Q1 exact · Q2 exact · separate input · sort + limit · query time", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 6, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum_over_time (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 7, + "payload": { + "kind": "relational", + "operator": { + "kind": "sort" + } + }, + "label": "sort by value, partition by job", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 8, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit" + } + }, + "label": "limit 10", + "output_state": { + "timing": "QueryTime" + } + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + }, + { + "producer": 6, + "consumer": 7 + }, + { + "producer": 7, + "consumer": 8 + } + ], + "query_roots": { + "q1": 3, + "q2": 8 + } + }, + "cost": { + "total": null, + "unit": "cpu_ms_per_s", + "per_node": {} + } + }, + { + "id": "P2", + "from_logical": "L2", + "label": "Q1 exact · Q2 exact · shared input · sort + limit · query time", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum_over_time (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "sort" + } + }, + "label": "sort by value, partition by job", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 6, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit" + } + }, + "label": "limit 10", + "output_state": { + "timing": "QueryTime" + } + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 1, + "consumer": 4 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + } + ], + "query_roots": { + "q1": 3, + "q2": 6 + } + }, + "cost": { + "total": null, + "unit": "cpu_ms_per_s", + "per_node": {} + } + }, + { + "id": "P3", + "from_logical": "L3", + "label": "Q1 exact · Q2 Count-Min + heap · separate input · build + estimate · query time", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 6, + "payload": { + "kind": "summary_agg" + }, + "label": "Count-Min + top-k heap, one per job", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 7, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job", + "output_state": { + "timing": "QueryTime" + } + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + }, + { + "producer": 6, + "consumer": 7 + } + ], + "query_roots": { + "q1": 3, + "q2": 7 + } + }, + "cost": { + "total": null, + "unit": "cpu_ms_per_s", + "per_node": {} + } + }, + { + "id": "P4", + "from_logical": "L4", + "label": "Q1 exact · Q2 Count-Min + heap · shared input · build + estimate · query time", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 4, + "payload": { + "kind": "summary_agg" + }, + "label": "Count-Min + top-k heap, one per job", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 5, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job", + "output_state": { + "timing": "QueryTime" + } + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 1, + "consumer": 4 + }, + { + "producer": 4, + "consumer": 5 + } + ], + "query_roots": { + "q1": 3, + "q2": 5 + } + }, + "cost": { + "total": null, + "unit": "cpu_ms_per_s", + "per_node": {} + } + }, + { + "id": "P5", + "from_logical": "L5", + "label": "Q1 exact · Q2 Hydra · separate input · build + estimate · query time", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 4, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 5, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 6, + "payload": { + "kind": "summary_agg" + }, + "label": "Hydra (CMS) over all (job, series) keys + heap per job", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 7, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job", + "output_state": { + "timing": "QueryTime" + } + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 4, + "consumer": 5 + }, + { + "producer": 5, + "consumer": 6 + }, + { + "producer": 6, + "consumer": 7 + } + ], + "query_roots": { + "q1": 3, + "q2": 7 + } + }, + "cost": { + "total": null, + "unit": "cpu_ms_per_s", + "per_node": {} + } + }, + { + "id": "P6", + "from_logical": "L6", + "label": "Q1 exact · Q2 Hydra · shared input · build + estimate · query time", + "dag": { + "nodes": [ + { + "id": 0, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan" + } + }, + "label": "http_requests_total", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 1, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range" + } + }, + "label": "range 1m", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 2, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "rate (per series)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 3, + "payload": { + "kind": "relational", + "operator": { + "kind": "aggregate" + } + }, + "label": "sum by (job)", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 4, + "payload": { + "kind": "summary_agg" + }, + "label": "Hydra (CMS) over all (job, series) keys + heap per job", + "output_state": { + "timing": "QueryTime" + } + }, + { + "id": 5, + "payload": { + "kind": "summary_estimate" + }, + "label": "top 10 per job", + "output_state": { + "timing": "QueryTime" + } + } + ], + "edges": [ + { + "producer": 0, + "consumer": 1 + }, + { + "producer": 1, + "consumer": 2 + }, + { + "producer": 2, + "consumer": 3 + }, + { + "producer": 1, + "consumer": 4 + }, + { + "producer": 4, + "consumer": 5 + } + ], + "query_roots": { + "q1": 3, + "q2": 5 + } + }, + "cost": { + "total": null, + "unit": "cpu_ms_per_s", + "per_node": {} + } + } + ] + }, + "stage3_selection": { + "selected": "P6", + "rejected": [ + { + "id": "P1", + "reason": "valid; costlier than P6 (illustrative)" + }, + { + "id": "P2", + "reason": "valid; costlier than P6 (illustrative)" + }, + { + "id": "P3", + "reason": "valid; costlier than P6 (illustrative)" + }, + { + "id": "P4", + "reason": "valid; costlier than P6 (illustrative)" + }, + { + "id": "P5", + "reason": "valid; costlier than P6 (illustrative)" + } + ] + } +} diff --git a/tools/dag-viewer/examples/planner-layering-example1.json b/tools/dag-viewer/examples/planner-layering-example1.json new file mode 100644 index 000000000..94d8bcea5 --- /dev/null +++ b/tools/dag-viewer/examples/planner-layering-example1.json @@ -0,0 +1,139325 @@ +{ + "format": "asap-stage-pipeline/v1", + "stage0_logical": { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 7 + } + ] + } + }, + "stage1_logical_asap": { + "candidates": [ + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 7 + } + ] + }, + "id": "L1", + "label": "Q1 exact · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 8 + } + ] + }, + "id": "L2", + "label": "Q1 exact · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 8 + } + ] + }, + "id": "L3", + "label": "Q1 exact · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 9 + } + ] + }, + "id": "L4", + "label": "Q1 exact · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 8 + } + ] + }, + "id": "L5", + "label": "Q1 exact · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 9 + } + ] + }, + "id": "L6", + "label": "Q1 exact · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 7 + } + ] + }, + "id": "L7", + "label": "Q1 exact · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 7 + } + ] + }, + "id": "L8", + "label": "Q1 exact · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L9", + "label": "Q1 exact (Rate acc) · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 9 + } + ] + }, + "id": "L10", + "label": "Q1 exact (Rate acc) · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 9 + } + ] + }, + "id": "L11", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 10 + } + ] + }, + "id": "L12", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 9 + } + ] + }, + "id": "L13", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 10 + } + ] + }, + "id": "L14", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L15", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L16", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L17", + "label": "Q1 exact (Sum acc) · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 9 + } + ] + }, + "id": "L18", + "label": "Q1 exact (Sum acc) · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 9 + } + ] + }, + "id": "L19", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 10 + } + ] + }, + "id": "L20", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 9 + } + ] + }, + "id": "L21", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 10 + } + ] + }, + "id": "L22", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L23", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L24", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 9 + } + ] + }, + "id": "L25", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 10 + } + ] + }, + "id": "L26", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 10 + } + ] + }, + "id": "L27", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + }, + { + "consumer": 11, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 10, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 11, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 11 + } + ] + }, + "id": "L28", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 10 + } + ] + }, + "id": "L29", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + }, + { + "consumer": 10, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input" + }, + { + "consumer": 11, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 10, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 11, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 11 + } + ] + }, + "id": "L30", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 9 + } + ] + }, + "id": "L31", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 9 + } + ] + }, + "id": "L32", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 5 + } + ] + }, + "id": "L33", + "label": "Q1 exact · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 6 + } + ] + }, + "id": "L34", + "label": "Q1 exact · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 6 + } + ] + }, + "id": "L35", + "label": "Q1 exact · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 7 + } + ] + }, + "id": "L36", + "label": "Q1 exact · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 6 + } + ] + }, + "id": "L37", + "label": "Q1 exact · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 7 + } + ] + }, + "id": "L38", + "label": "Q1 exact · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 5 + } + ] + }, + "id": "L39", + "label": "Q1 exact · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 3 + }, + { + "Operator": 5 + } + ] + }, + "id": "L40", + "label": "Q1 exact · Q2 whole-expression CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 6 + } + ] + }, + "id": "L41", + "label": "Q1 exact (Rate acc) · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 7 + } + ] + }, + "id": "L42", + "label": "Q1 exact (Rate acc) · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 7 + } + ] + }, + "id": "L43", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L44", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 7 + } + ] + }, + "id": "L45", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L46", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 6 + } + ] + }, + "id": "L47", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 6 + } + ] + }, + "id": "L48", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 6 + } + ] + }, + "id": "L49", + "label": "Q1 exact (Sum acc) · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 7 + } + ] + }, + "id": "L50", + "label": "Q1 exact (Sum acc) · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 7 + } + ] + }, + "id": "L51", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L52", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 7 + } + ] + }, + "id": "L53", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 8 + } + ] + }, + "id": "L54", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 6 + } + ] + }, + "id": "L55", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 4 + }, + { + "Operator": 6 + } + ] + }, + "id": "L56", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 7 + } + ] + }, + "id": "L57", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "accuracy": { + "EpsilonDelta": { + "delta": 0.001, + "epsilon": 0.01 + } + }, + "k": 10, + "kind": "top_k" + } + ], + "output_names": [], + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 8 + } + ] + }, + "id": "L58", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 8 + } + ] + }, + "id": "L59", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 9 + } + ] + }, + "id": "L60", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 8 + } + ] + }, + "id": "L61", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + }, + { + "consumer": 8, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input" + }, + { + "consumer": 9, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 9 + } + ] + }, + "id": "L62", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 7 + } + ] + }, + "id": "L63", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input" + }, + { + "consumer": 2, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 3, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input" + }, + { + "consumer": 4, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input" + }, + { + "consumer": 5, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input" + }, + { + "consumer": 6, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input" + }, + { + "consumer": 7, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + }, + "result_kind": "instant_vector" + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + }, + "result_kind": "range_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "kind": "finalize_exact_accumulator" + }, + "result_kind": "instant_vector" + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + }, + "result_kind": "state" + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + }, + "result_kind": "instant_vector" + } + ], + "roots": [ + { + "Operator": 5 + }, + { + "Operator": 7 + } + ] + }, + "id": "L64", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap · shared input" + } + ], + "capped": false, + "combinations": 64 + }, + "stage2_physical_asap": { + "candidates": [ + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 3, + 8 + ] + }, + "from_logical": "L1", + "id": "P1", + "label": "Q1 exact · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 3, + 9 + ] + }, + "from_logical": "L2", + "id": "P2", + "label": "Q1 exact · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 8 + ] + }, + "from_logical": "L3", + "id": "P3", + "label": "Q1 exact · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 9 + ] + }, + "from_logical": "L4", + "id": "P4", + "label": "Q1 exact · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 8 + ] + }, + "from_logical": "L5", + "id": "P5", + "label": "Q1 exact · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 9 + ] + }, + "from_logical": "L6", + "id": "P6", + "label": "Q1 exact · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 7 + ] + }, + "from_logical": "L7", + "id": "P7", + "label": "Q1 exact · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 7 + ] + }, + "from_logical": "L8", + "id": "P8", + "label": "Q1 exact · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 9 + ] + }, + "from_logical": "L9", + "id": "P9", + "label": "Q1 exact (Rate acc) · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 10 + ] + }, + "from_logical": "L10", + "id": "P10", + "label": "Q1 exact (Rate acc) · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 9 + ] + }, + "from_logical": "L11", + "id": "P11", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 10 + ] + }, + "from_logical": "L12", + "id": "P12", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 9 + ] + }, + "from_logical": "L13", + "id": "P13", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 10 + ] + }, + "from_logical": "L14", + "id": "P14", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L15", + "id": "P15", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L16", + "id": "P16", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 9 + ] + }, + "from_logical": "L17", + "id": "P17", + "label": "Q1 exact (Sum acc) · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 10 + ] + }, + "from_logical": "L18", + "id": "P18", + "label": "Q1 exact (Sum acc) · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 9 + ] + }, + "from_logical": "L19", + "id": "P19", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 10 + ] + }, + "from_logical": "L20", + "id": "P20", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 9 + ] + }, + "from_logical": "L21", + "id": "P21", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 10 + ] + }, + "from_logical": "L22", + "id": "P22", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L23", + "id": "P23", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L24", + "id": "P24", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 5, + 10 + ] + }, + "from_logical": "L25", + "id": "P25", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 11, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 10, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 11, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 5, + 11 + ] + }, + "from_logical": "L26", + "id": "P26", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 10 + ] + }, + "from_logical": "L27", + "id": "P27", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 11, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 10, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 11, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 11 + ] + }, + "from_logical": "L28", + "id": "P28", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 10 + ] + }, + "from_logical": "L29", + "id": "P29", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 10, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 9, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 11, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 10, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 10, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 11, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 11 + ] + }, + "from_logical": "L30", + "id": "P30", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc)" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 9 + ] + }, + "from_logical": "L31", + "id": "P31", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 9 + ] + }, + "from_logical": "L32", + "id": "P32", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 3, + 6 + ] + }, + "from_logical": "L33", + "id": "P33", + "label": "Q1 exact · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 3, + 7 + ] + }, + "from_logical": "L34", + "id": "P34", + "label": "Q1 exact · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 6 + ] + }, + "from_logical": "L35", + "id": "P35", + "label": "Q1 exact · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 7 + ] + }, + "from_logical": "L36", + "id": "P36", + "label": "Q1 exact · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 6 + ] + }, + "from_logical": "L37", + "id": "P37", + "label": "Q1 exact · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 7 + ] + }, + "from_logical": "L38", + "id": "P38", + "label": "Q1 exact · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 5 + ] + }, + "from_logical": "L39", + "id": "P39", + "label": "Q1 exact · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 3, + 5 + ] + }, + "from_logical": "L40", + "id": "P40", + "label": "Q1 exact · Q2 whole-expression CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 7 + ] + }, + "from_logical": "L41", + "id": "P41", + "label": "Q1 exact (Rate acc) · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L42", + "id": "P42", + "label": "Q1 exact (Rate acc) · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 7 + ] + }, + "from_logical": "L43", + "id": "P43", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L44", + "id": "P44", + "label": "Q1 exact (Rate acc) · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 7 + ] + }, + "from_logical": "L45", + "id": "P45", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L46", + "id": "P46", + "label": "Q1 exact (Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 6 + ] + }, + "from_logical": "L47", + "id": "P47", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": { + "Reduce": [ + 2 + ] + } + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 6 + ] + }, + "from_logical": "L48", + "id": "P48", + "label": "Q1 exact (Rate acc) · Q2 whole-expression CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 7 + ] + }, + "from_logical": "L49", + "id": "P49", + "label": "Q1 exact (Sum acc) · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L50", + "id": "P50", + "label": "Q1 exact (Sum acc) · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 7 + ] + }, + "from_logical": "L51", + "id": "P51", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L52", + "id": "P52", + "label": "Q1 exact (Sum acc) · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 7 + ] + }, + "from_logical": "L53", + "id": "P53", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 8 + ] + }, + "from_logical": "L54", + "id": "P54", + "label": "Q1 exact (Sum acc) · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 6 + ] + }, + "from_logical": "L55", + "id": "P55", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 5, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "kind": "rate" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 4, + 6 + ] + }, + "from_logical": "L56", + "id": "P56", + "label": "Q1 exact (Sum acc) · Q2 whole-expression CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 5, + 8 + ] + }, + "from_logical": "L57", + "id": "P57", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "keys": [ + { + "ascending": false, + "expr": { + "Column": 1 + }, + "nulls_first": false + } + ], + "kind": "sort", + "partition_by": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "limit", + "n": 10, + "offset": 0, + "partition_by": [ + 2 + ] + } + } + } + ], + "roots": [ + 5, + 9 + ] + }, + "from_logical": "L58", + "id": "P58", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 exact (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 8 + ] + }, + "from_logical": "L59", + "id": "P59", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 9 + ] + }, + "from_logical": "L60", + "id": "P60", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CMS+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "filters": [], + "having": null, + "kind": "aggregate", + "measures": [ + { + "col": null, + "kind": "sum" + } + ], + "output_names": [ + "" + ], + "reduction": "PerEntity" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 8 + ] + }, + "from_logical": "L61", + "id": "P61", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 8, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 7, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 9, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 8, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 8, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 9, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 9 + ] + }, + "from_logical": "L62", + "id": "P62", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 CountSketch+heap (Sum acc) · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CmsWithHeap", + "category": "TopK", + "params": { + "CmsWithHeap": { + "depth": 7, + "heap_size": 100, + "width": 272 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 7 + ] + }, + "from_logical": "L63", + "id": "P63", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CMS+heap · shared input" + }, + { + "dag": { + "edges": [ + { + "consumer": 1, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 0, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 2, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 3, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 2, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 4, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 3, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 5, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 4, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 6, + "data_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "producer": 1, + "role": "Input", + "window": "NotApplicable" + }, + { + "consumer": 7, + "data_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "grouping": "NotApplicable", + "intermediate_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "producer": 6, + "role": "Input", + "window": "NotApplicable" + } + ], + "nodes": [ + { + "coverage": null, + "guarantee": null, + "id": 0, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "scan", + "predicates": [], + "schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 1, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "relational", + "operator": { + "kind": "time_range", + "range": { + "nanos": 0, + "secs": 60 + }, + "range_kind": "range" + } + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 2, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "name": "state", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Rate", + "Rate" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": "PerEntity" + } + }, + { + "coverage": null, + "guarantee": null, + "id": 3, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "timestamp" + }, + "name": "ts", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + } + ], + "time_index": 0, + "unique_keys": [ + [ + 3, + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 4, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "ExactAggregate": [ + "Sum", + "Sum" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": null, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 5, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "sum", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "finalize_exact_accumulator" + } + }, + { + "coverage": { + "regions": [ + { + "population": {}, + "time_ms": null + } + ], + "source": { + "TimeSeries": { + "metric": "http_requests_total" + } + } + }, + "guarantee": null, + "id": 6, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "name": "state", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0 + ] + ] + }, + "output_state": { + "primitive": "SummaryState", + "timing": "query_time" + }, + "payload": { + "family": { + "Sketch": [ + { + "algorithm": "CountSketchWithHeap", + "category": "TopK", + "params": { + "CountSketchWithHeap": { + "depth": 125, + "heap_size": 100, + "width": 30000 + } + } + }, + "PerSubpopulationInstance" + ] + }, + "filter": null, + "grouping": "PerSubpopulationInstance", + "input": { + "item": { + "Column": { + "Named": "$promql_series_identity" + } + }, + "weight": { + "Column": "SampleValue" + }, + "weight_domain": { + "kind": "unknown_or_signed" + } + }, + "kind": "summary_agg", + "reduction": { + "Reduce": [ + 2 + ] + } + } + }, + { + "coverage": null, + "guarantee": null, + "id": 7, + "output_schema": { + "closed": true, + "fields": [ + { + "dtype": { + "Plain": "utf8" + }, + "name": "job", + "nullable": true, + "table": null + }, + { + "dtype": { + "Plain": "utf8" + }, + "name": "$promql_series_identity", + "nullable": false, + "table": null + }, + { + "dtype": { + "Plain": "float64" + }, + "name": "value", + "nullable": false, + "table": null + } + ], + "time_index": null, + "unique_keys": [ + [ + 0, + 1 + ] + ] + }, + "output_state": { + "primitive": "Raw", + "timing": "query_time" + }, + "payload": { + "kind": "summary_estimate", + "query": { + "TopK": { + "k": 10 + } + } + } + } + ], + "roots": [ + 5, + 7 + ] + }, + "from_logical": "L64", + "id": "P64", + "label": "Q1 exact (Sum acc, Rate acc) · Q2 whole-expression CountSketch+heap · shared input" + } + ] + }, + "stage3_selection": { + "costs": { + "P1": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "5": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "6": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "7": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "8": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 86.40100000000001, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P10": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 0.001, + "detail": "limit to 1000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "8": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "9": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 80.40100000000001, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P13": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "8": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "9": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 195.401, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P14": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "8": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "9": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 192.401, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P16": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "8": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 565.401, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P17": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "8": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "9": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 85.40110000000001, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P18": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 0.001, + "detail": "limit to 1000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "8": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "9": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 82.40110000000001, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P2": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "5": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "6": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "7": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "8": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "9": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 83.40100000000001, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P21": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "8": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "9": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 197.4011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P22": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "8": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "9": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 194.4011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P24": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "8": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 567.4011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P25": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 0.001, + "detail": "limit to 1000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "7": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "8": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "9": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 82.40110000000001, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P26": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "11": { + "cost": 0.001, + "detail": "limit to 1000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "7": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "8": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "9": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 79.40110000000001, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P29": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "7": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "8": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "9": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 194.4011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P30": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "10": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "11": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "7": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "8": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "9": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 191.4011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P32": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "7": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "8": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "9": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 564.4011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P33": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "5": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "6": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 59.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P34": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "5": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "6": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "7": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 56.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P37": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "5": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "6": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 171.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P38": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "5": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "6": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "7": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 168.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P40": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "5": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 541.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P41": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "6": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "7": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 56.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P42": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "6": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "7": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "8": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 53.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P45": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "6": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "7": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 168.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P46": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "6": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "7": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "8": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 165.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P48": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "6": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 538.201, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P49": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "6": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "7": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 58.201100000000004, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P5": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "5": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "6": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "7": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "8": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 198.401, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P50": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "6": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "7": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "8": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 55.201100000000004, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P53": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "6": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "7": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 170.20110000000003, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P54": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "6": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "7": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "8": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 167.20110000000003, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P56": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "4": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "5": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "6": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 540.2011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P57": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "7": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "8": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 55.201100000000004, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P58": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "7": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "8": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "9": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 52.201100000000004, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P6": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "5": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "6": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "7": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "8": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "9": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 195.401, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P61": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "7": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "8": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 167.20110000000003, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P62": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 4.0, + "detail": "build exact Sum accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "7": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "8": { + "cost": 126.0, + "detail": "build CountSketchWithHeap into 100 states: 1000000 rows x depth 126" + }, + "9": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 164.20110000000003, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P64": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 1.0, + "detail": "build exact Sum accumulator into 100 states: 1000000 rows x depth 1" + }, + "5": { + "cost": 0.00009999999999999999, + "detail": "finalize 100 accumulators" + }, + "6": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "7": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 537.2011, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P8": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "3": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "4": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "5": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "6": { + "cost": 504.0, + "detail": "build CountSketchWithHeap into 100 states: 4000000 rows x depth 126" + }, + "7": { + "cost": 0.001, + "detail": "estimate 1000 rows from 100 states" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 568.401, + "unit": "cpu_ms_per_workload_evaluation" + }, + "P9": { + "per_node": { + "0": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "1": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "2": { + "cost": 4.0, + "detail": "build exact Rate accumulator into 1000000 states: 4000000 rows x depth 1" + }, + "3": { + "cost": 1.0, + "detail": "finalize 1000000 accumulators" + }, + "4": { + "cost": 2.0, + "detail": "hash aggregate 1000000 rows into 100 groups" + }, + "5": { + "cost": 23.2, + "detail": "scan 4000000 samples" + }, + "6": { + "cost": 4.0, + "detail": "time range 60s: pass 4000000 rows" + }, + "7": { + "cost": 8.0, + "detail": "hash aggregate 4000000 rows into 1000000 groups" + }, + "8": { + "cost": 14.0, + "detail": "sort 1000000 rows in 100 partitions" + }, + "9": { + "cost": 0.001, + "detail": "limit to 1000 rows" + } + }, + "source": "analytical-cost-v1 (illustrative statistics)", + "total": 83.40100000000001, + "unit": "cpu_ms_per_workload_evaluation" + } + }, + "rejected": [ + { + "id": "P3", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P4", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P7", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P11", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P12", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P15", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P19", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P20", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P23", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P27", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P28", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P31", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P35", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P36", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P39", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P43", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P44", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P47", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P51", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P52", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P55", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P59", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P60", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P63", + "reason": "q2: CmsWithHeap needs non-negative update weights, and these are not proven non-negative", + "valid": false + }, + { + "id": "P1", + "reason": "costlier: 86.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P2", + "reason": "costlier: 83.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P5", + "reason": "costlier: 198.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P6", + "reason": "costlier: 195.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P8", + "reason": "costlier: 568.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P9", + "reason": "costlier: 83.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P10", + "reason": "costlier: 80.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P13", + "reason": "costlier: 195.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P14", + "reason": "costlier: 192.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P16", + "reason": "costlier: 565.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P17", + "reason": "costlier: 85.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P18", + "reason": "costlier: 82.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P21", + "reason": "costlier: 197.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P22", + "reason": "costlier: 194.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P24", + "reason": "costlier: 567.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P25", + "reason": "costlier: 82.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P26", + "reason": "costlier: 79.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P29", + "reason": "costlier: 194.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P30", + "reason": "costlier: 191.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P32", + "reason": "costlier: 564.401 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P33", + "reason": "costlier: 59.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P34", + "reason": "costlier: 56.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P37", + "reason": "costlier: 171.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P38", + "reason": "costlier: 168.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P40", + "reason": "costlier: 541.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P41", + "reason": "costlier: 56.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P42", + "reason": "costlier: 53.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P45", + "reason": "costlier: 168.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P46", + "reason": "costlier: 165.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P48", + "reason": "costlier: 538.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P49", + "reason": "costlier: 58.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P50", + "reason": "costlier: 55.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P53", + "reason": "costlier: 170.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P54", + "reason": "costlier: 167.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P56", + "reason": "costlier: 540.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P57", + "reason": "costlier: 55.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P61", + "reason": "costlier: 167.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P62", + "reason": "costlier: 164.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + }, + { + "id": "P64", + "reason": "costlier: 537.201 vs 52.201 cpu_ms_per_workload_evaluation", + "valid": true + } + ], + "selected": "P58" + }, + "workload": { + "queries": [ + { + "id": "q1", + "language": "promql", + "requirements": { + "accuracy": "exact", + "repeat_interval_ms": 10000 + }, + "text": "sum by (job) (rate(http_requests_total[1m]))" + }, + { + "id": "q2", + "language": "promql", + "requirements": { + "accuracy": { + "delta": 0.001, + "epsilon": 0.01 + }, + "latency_ms": 100.0, + "repeat_interval_ms": 10000 + }, + "text": "topk by (job) (10, sum_over_time(http_requests_total[1m]))" + } + ] + } +} diff --git a/tools/dag-viewer/generate-sample.sh b/tools/dag-viewer/generate-sample.sh index e916cba5f..66921d8c7 100755 --- a/tools/dag-viewer/generate-sample.sh +++ b/tools/dag-viewer/generate-sample.sh @@ -6,7 +6,7 @@ set -euo pipefail cd "$(dirname "${BASH_SOURCE[0]}")/../.." # --epsilon asks for an approximate accuracy target instead of the default -# Exact, so SketchAlgorithmStrategy actually has a sketch alternative to +# Exact, so ASAPStrategies actually has a sketch alternative to # report — without it, no query below would ever pick up a `notes` badge # (see crates/devtools/src/bin/dag_export.rs's own `--epsilon` doc comment). cargo run -p asap-devtools --bin dag_export -- \ diff --git a/tools/dag-viewer/lifecycle-summary-maintenance.png b/tools/dag-viewer/lifecycle-summary-maintenance.png deleted file mode 100644 index e873ffd9f..000000000 Binary files a/tools/dag-viewer/lifecycle-summary-maintenance.png and /dev/null differ diff --git a/tools/dag-viewer/node-style.js b/tools/dag-viewer/node-style.js index a957817c2..fee49586e 100644 --- a/tools/dag-viewer/node-style.js +++ b/tools/dag-viewer/node-style.js @@ -1,20 +1,17 @@ -// Logical QueryExpr/SummaryExpr kinds exported by +// Operator kinds (`Operator::kind_name`) exported by // crates/types/src/dag_export.rs. Categories describe the visible logical DAG // shape. They do not model hidden physical inputs: for example, // PromqlInfoEnrich is a one-child enrichment here even if physical costing // later accounts for an auxiliary source scan. const KIND_CATEGORY_JSON = `{ "Scan": "data", - "PromqlScalarBridge": "data", - "EvalTimestamp": "data", - "CurrentTimestamp": "data", + "Values": "data", "Filter": "filter", "PromqlSeriesSample": "sample", "Project": "derive", "PromqlRelabel": "derive", "PromqlInfoEnrich": "derive", "PromqlVectorFromScalar": "derive", - "PromqlScalarFromVector": "derive", "BinaryOp": "derive", "Aggregate": "aggregate", "TimeRange": "window", @@ -22,21 +19,21 @@ const KIND_CATEGORY_JSON = `{ "TimeShift": "window", "SQLWindowFunc": "window", "Join": "join", - "RelationalJoin": "join", "Dedup": "set", "SetOp": "set", "Concat": "combine", "Sort": "sort", "Limit": "sort", - "KeepPreAsap": "summary", "SummaryAgg": "summary", "SummaryJoin": "summary", "SummarySubtract": "summary", - "SummaryBinaryOp": "summary", - "ValueOperation": "summary", "SummaryDelete": "summary", "SummaryEstimate": "summary", - "SummaryMerge": "summary" + "SummaryMerge": "summary", + "FinalizeExactAccumulator": "summary", + "MaintainPopulation": "summary", + "EvaluatePopulation": "summary", + "Extension": "summary" }`; const KIND_CATEGORY = Object.freeze(JSON.parse(KIND_CATEGORY_JSON)); @@ -44,7 +41,7 @@ const KIND_CATEGORY = Object.freeze(JSON.parse(KIND_CATEGORY_JSON)); const CATEGORIES = { data: { label: 'Data', - description: 'Scan, PromqlScalarBridge, EvalTimestamp, CurrentTimestamp — leaves that introduce a value', + description: 'Scan and Values — data sources', light: { bg: '#eef5fd', border: '#0369a1' }, dark: { bg: '#0c2438', border: '#38bdf8' }, }, @@ -62,7 +59,7 @@ const CATEGORIES = { }, derive: { label: 'Derive', - description: 'Project, PromqlRelabel, PromqlInfoEnrich, PromqlVectorFromScalar, PromqlScalarFromVector, BinaryOp — transforms or enriches columns on otherwise-unchanged rows', + description: 'Project, PromqlRelabel, PromqlInfoEnrich, PromqlVectorFromScalar, BinaryOp — transforms or enriches columns on otherwise-unchanged rows', light: { bg: '#f5f0fd', border: '#6d28d9' }, dark: { bg: '#241a3d', border: '#a78bfa' }, }, @@ -102,10 +99,10 @@ const CATEGORIES = { light: { bg: '#eef4fd', border: '#1d4ed8' }, dark: { bg: '#12233d', border: '#60a5fa' }, }, - // Post-ASAP nodes use a neutral palette; KeepPreAsap has a muted override. + // ASAP operators use a neutral palette. summary: { label: 'Summary', - description: 'KeepPreAsap, SummaryBinaryOp, ValueOperation, SummaryAgg, SummaryJoin, SummarySubtract, SummaryDelete, SummaryEstimate, SummaryMerge — post-ASAP materialized structures', + description: 'SummaryAgg, SummaryEstimate, FinalizeExactAccumulator, MaintainPopulation, EvaluatePopulation, SummaryJoin, SummarySubtract, SummaryDelete, SummaryMerge, Extension — summary state and its evaluations', light: { bg: '#f1f2f4', border: '#4b5563' }, dark: { bg: '#20242b', border: '#9ca3af' }, }, diff --git a/tools/dag-viewer/post_asap_fixture.json b/tools/dag-viewer/post_asap_fixture.json index ea2de185f..e616e80e2 100644 --- a/tools/dag-viewer/post_asap_fixture.json +++ b/tools/dag-viewer/post_asap_fixture.json @@ -14,9 +14,9 @@ }, "post_dag": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}], "root": 0}}, "children": [] }, + {"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 }, - { "id": 2, "kind": "KeepPreAsap", "label": "KeepPreAsap(Project)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": []}], "root": 0}}, "children": [1] } + {"id": 2, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": [1]} ], "root": 2 }, @@ -39,7 +39,7 @@ "kind": "Summary", "dag": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": [], "hash": 111}], "root": 0}}, "children": [] }, + {"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 } ], "root": 1 @@ -64,7 +64,7 @@ "kind": "Summary", "dag": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {}, "children": [] }, + {"id": 0, "kind": "Scan", "label": "Scan", "detail": {}, "children": []}, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(HydraKll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "grouping": {"SharedMultiSubpopulation": {"params": {}}}}, "children": [0], "origin_pre_id": 1 } ], "root": 1 diff --git a/tools/dag-viewer/render.py b/tools/dag-viewer/render.py index fbdaca97a..7b16796da 100755 --- a/tools/dag-viewer/render.py +++ b/tools/dag-viewer/render.py @@ -17,7 +17,7 @@ This does not add anything index.html doesn't already do — it shares viewer.js and node-style.js with it verbatim (see viewer.js's header comment) and only differs in packaging: one query's worth of exported -`QueryExpr` detail *is* its plan (see the side panel on node click), and +`OperatorNode` detail *is* its plan (see the side panel on node click), and shared-hash highlighting *is* what this repo has for CSE today — both a hash-based proxy, not real CSE output; see README.md's "Shared-sub-DAG highlighting is a proxy" section. Structured cost/benefit annotations @@ -72,7 +72,7 @@ def _compact(value: object) -> str: if not isinstance(value, dict): return str(value) - # Common serde enum/newtype shapes in QueryExpr detail. + # Common serde enum/newtype shapes in OperatorNode detail. if set(value) == {"Column"}: return f"col[{_compact(value['Column'])}]" if set(value) == {"Table"} and isinstance(value["Table"], dict): @@ -269,23 +269,6 @@ def load_workload(paths: list[Path]) -> dict: for path in paths: data = json.loads(path.read_text()) incoming = data.get("queries", []) - if not incoming and isinstance(data.get("dag"), dict) and isinstance(data.get("deployments"), list): - incoming = [{ - "name": path.stem or "Summary maintenance plan", - "dag": data["dag"], - "post_dag": data["dag"], - "lifecycle_plan": True, - "lifecycle_summary": { - "selected_raw_recompute": data.get("selected_raw_recompute", False), - "summary_total_cost": data.get("summary_total_cost"), - "raw_recompute_total_cost": data.get("raw_recompute_total_cost"), - "horizon_seconds": data.get("horizon_seconds"), - "evaluation_rate_per_second": data.get("evaluation_rate_per_second"), - "update_rate_per_second": data.get("update_rate_per_second"), - "expected_reads": data.get("expected_reads"), - "deployment_count": len(data["deployments"]), - }, - }] for q in incoming: name = q["name"] if name in seen_names: diff --git a/tools/dag-viewer/test_render.py b/tools/dag-viewer/test_render.py index 20ccb64cd..e78e978f0 100644 --- a/tools/dag-viewer/test_render.py +++ b/tools/dag-viewer/test_render.py @@ -79,43 +79,6 @@ def test_boundary_terms_and_provenance_survive_standalone_export(self): self.assertIn(annotation["model_version"], html) self.assertIn(annotation["evidence_version"], html) - def test_loads_summary_maintenance_export_as_a_lifecycle_plan(self): - dag = named_dag("unused")["dag"] - summary = { - "selected_raw_recompute": True, - "summary_total_cost": None, - "raw_recompute_total_cost": 7.5, - "horizon_seconds": 60.0, - "evaluation_rate_per_second": 2.0, - "update_rate_per_second": 3.0, - "expected_reads": 120.0, - } - with tempfile.TemporaryDirectory() as d: - path = Path(d) / "lifecycle.json" - path.write_text(json.dumps({"dag": dag, "deployments": [], **summary})) - workload = load_workload([path]) - - query = workload["queries"][0] - self.assertEqual(query["name"], "lifecycle") - self.assertTrue(query["lifecycle_plan"]) - self.assertEqual(query["post_dag"], dag) - self.assertEqual( - query["lifecycle_summary"], - {**summary, "deployment_count": 0}, - ) - - def test_preserves_summary_plan_deployment_count(self): - dag = named_dag("unused")["dag"] - with tempfile.TemporaryDirectory() as d: - path = Path(d) / "lifecycle.json" - path.write_text(json.dumps({"dag": dag, "deployments": [{}, {}]})) - workload = load_workload([path]) - - self.assertEqual( - workload["queries"][0]["lifecycle_summary"]["deployment_count"], - 2, - ) - def test_merges_queries_across_files_in_order(self): with tempfile.TemporaryDirectory() as d: f1 = Path(d) / "a.json" diff --git a/tools/dag-viewer/viewer.js b/tools/dag-viewer/viewer.js index 6e4c696ee..d8191dc1a 100644 --- a/tools/dag-viewer/viewer.js +++ b/tools/dag-viewer/viewer.js @@ -18,7 +18,7 @@ cytoscape.use(window.cytoscapeDagre); // --post-asap whole-query merged post-ASAP DAG (same flattened // `{nodes, root}` shape as `dag`, but nodes may be post-ASAP-only kinds // like "SummaryAgg" mixed in, and any such node has no `hash` — there's no -// corresponding QueryExpr to hash) — left `undefined` when absent (omitted +// corresponding OperatorNode to hash) — left `undefined` when absent (omitted // whenever --post-asap wasn't set, or this query had zero replacements), // unlike `replacements` which always defaults to an array. `workload_cost` // is the optional per-query `NamedDAG.workload_cost` (issue #286), also @@ -122,13 +122,7 @@ function loadFiles(fileList) { reader.onload = () => { try { const parsed = JSON.parse(reader.result); - const incoming = parsed.queries || (parsed.dag && parsed.deployments ? [{ - name: file.name.replace(/\.json$/i, '') || 'Summary maintenance plan', - dag: parsed.dag, - post_dag: parsed.dag, - lifecycle_plan: true, - lifecycle_summary: lifecyclePlanSummary(parsed), - }] : []); + const incoming = parsed.queries || []; const existingNames = new Set(queries.map((q) => q.name)); // One batch id per *file* — every query this one dag_export // invocation produced shares its decision.id numbering. @@ -137,7 +131,7 @@ function loadFiles(fileList) { let name = q.name; if (existingNames.has(name)) name = `${q.name} (${file.name})`; existingNames.add(name); - queries.push({ name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch }); + queries.push({ name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, sourceBatch }); }); } catch (err) { alert(`Failed to parse ${file.name}: ${err.message}`); @@ -154,19 +148,6 @@ function loadFiles(fileList) { fileInput.value = ''; } -function lifecyclePlanSummary(plan) { - return { - selected_raw_recompute: Boolean(plan.selected_raw_recompute), - summary_total_cost: plan.summary_total_cost ?? null, - raw_recompute_total_cost: plan.raw_recompute_total_cost ?? null, - horizon_seconds: plan.horizon_seconds ?? null, - evaluation_rate_per_second: plan.evaluation_rate_per_second ?? null, - update_rate_per_second: plan.update_rate_per_second ?? null, - expected_reads: plan.expected_reads ?? null, - deployment_count: Array.isArray(plan.deployments) ? plan.deployments.length : 0, - }; -} - function getParticipants() { return Array.from(participants) .filter((i) => i >= 0 && i < queries.length) @@ -290,22 +271,6 @@ function buildCyStyle() { selector: 'node[category = "unknown"]', style: { 'border-style': 'dashed', 'border-width': 3 }, }, - { - // KeepPreAsap (post-ASAP lane only) is post-ASAP-only - // as a *kind*, but represents literally unchanged pre-ASAP content — - // override the 'summary' category's color/icon with the same neutral - // panel/muted/dashed treatment the rest of the chrome uses for "nothing - // to see here", so a glance at the After lane separates "the planner - // did something" (solid, colored) from "left alone" (dashed, muted). - // See node-style.js's CATEGORIES.summary comment for the category-level - // color choice this overrides. - selector: 'node[kind = "KeepPreAsap"]', - style: { - 'background-color': panelColor, - 'border-color': borderColor, - 'border-style': 'dashed', - }, - }, { selector: 'node.root', style: { 'border-width': 2.5 }, @@ -497,28 +462,6 @@ function renderPrePostAsap() { return; } hideModeHint(); - if (selected.length === 1 && selected[0].lifecycle_plan) { - viewTitleEl.textContent = `Summary maintenance: ${selected[0].name}`; - const elements = laneElements( - 'summary-maintenance', - `${selected[0].name} · lifecycle plan`, - selected[0].post_dag, - selected[0], - 'post', - ); - buildCy(elements); - finalizeDAGInteractions(); - applyHighlighting(); - const initial = cy.nodes().filter((node) => !node.data('isLane') && node.data('root')).first(); - if (initial && initial.length) { - initial.select(); - showPrePostDetail(initial.data()); - } else { - clearDetail(); - } - fitAndSyncZoom(); - return; - } viewTitleEl.textContent = selected.length === 1 ? `Pre/Post-ASAP: ${selected[0].name}` : `Pre/Post-ASAP workload union: ${selected.length} queries`; @@ -663,10 +606,7 @@ function laneElements(laneId, laneLabel, dag, query, stage, laneCost) { // exactly the plain IR label. label: node.label + nodeCostBadgeSuffix(node), node, - // Flat (not nested under `node`) so buildCyStyle's - // `node[kind = "KeepPreAsap"]` selector can actually match it — - // cytoscape selectors can't reach into a data field that's itself an - // object. + // Cytoscape selectors read flat data fields. kind: node.kind, category: categoryOf(node.kind), root: node.id === dag.root, @@ -674,7 +614,6 @@ function laneElements(laneId, laneLabel, dag, query, stage, laneCost) { laneId, stage, queryName: query.name, - lifecycleSummary: query.lifecycle_summary, translations: translationsForNode(query, node, stage), }, classes: node.decision && typeof node.decision.benefit?.value === 'number' @@ -1028,33 +967,6 @@ function showPrePostDetail(data) { : ''; const decisions = data.translations || []; - const planSummary = data.lifecycleSummary; - let planSummaryHtml = ''; - if (planSummary) { - const value = (item) => item === null || item === undefined ? 'unknown' : String(item); - const selected = planSummary.selected_raw_recompute - ? 'Raw recomputation' - : 'Summary maintenance'; - planSummaryHtml = `

Lifecycle plan decision

-
Selected: ${escapeHtml(selected)}
-
summary cost: ${escapeHtml(value(planSummary.summary_total_cost))} · raw recompute cost: ${escapeHtml(value(planSummary.raw_recompute_total_cost))} · deployments: ${escapeHtml(value(planSummary.deployment_count))}
-
horizon: ${escapeHtml(value(planSummary.horizon_seconds))} s · expected reads: ${escapeHtml(value(planSummary.expected_reads))} · evaluation rate: ${escapeHtml(value(planSummary.evaluation_rate_per_second))}/s · update rate: ${escapeHtml(value(planSummary.update_rate_per_second))}/s
-
`; - } - const lifecycle = node.detail && node.detail.summary_maintenance; - let lifecycleHtml = ''; - if (lifecycle) { - const selected = lifecycle.selected; - const selectedText = selected - ? `${selected.lifecycle.kind} · ${selected.maintenance_mode} · ${selected.evaluation_schedule} · ${selected.output_representation}` - : 'No lifecycle selected'; - const alternatives = (lifecycle.alternatives || []).map((alternative) => { - const status = alternative.rejection ? `rejected: ${alternative.rejection}` : `cost: ${alternative.total_cost}`; - const assumptions = (alternative.assumptions || []).join('; ') || 'none'; - return `
${escapeHtml(alternative.lifecycle.kind)}
${escapeHtml(status)}
assumptions: ${escapeHtml(assumptions)}
`; - }).join(''); - lifecycleHtml = `

Summary maintenance lifecycle

Selected: ${escapeHtml(selectedText)}
${alternatives}
`; - } let translationHtml = ''; if (decisions.length > 0) { const cards = decisions.map((entry) => ` @@ -1076,9 +988,7 @@ function showPrePostDetail(data) { ${escapeHtml(chipLabel)}
${escapeHtml(node.label)}
${rootHtml} - ${planSummaryHtml} ${translationHtml} - ${lifecycleHtml}

IR node content

${escapeHtml(JSON.stringify(node.detail, null, 2))}
`; @@ -1172,10 +1082,6 @@ function renderLegend() { Shared workload nodeExplicitly identified by the exporter as shared across selected queries`); rows.push(`
Query root${escapeHtml(ROOT_BADGE.description)}
`); - const panelBg = getComputedStyle(document.documentElement).getPropertyValue('--panel2').trim() || '#f0f2f5'; - const mutedColor = getComputedStyle(document.documentElement).getPropertyValue('--muted').trim() || '#6b7280'; - rows.push(`
- Pass-through (KeepPreAsap)Unchanged pre-ASAP sub-DAG carried into the Summary DAG as-is
`); legendList.innerHTML = rows.join(''); } @@ -1204,7 +1110,7 @@ function loadWorkload(parsed) { const incoming = (parsed && parsed.queries) || []; // One batch id for this whole document — see `sourceBatch`'s own doc above. const sourceBatch = nextSourceBatch++; - incoming.forEach((q) => queries.push({ name: q.name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch })); + incoming.forEach((q) => queries.push({ name: q.name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, sourceBatch })); if (activeIndex === -1 && queries.length > 0) activeIndex = 0; if (participants.size === 0 && activeIndex >= 0) participants.add(activeIndex); }