From ebfcc370f625f367b4c4bf393bec2e14580849d2 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Mon, 5 Oct 2026 00:02:14 +0000 Subject: [PATCH 1/2] feat(accuracy): certify UnivMon L2 from layer 0's F2 UnivMon's L2 readout now reads layer 0's CountSketch row-median F2 (the textbook AMS estimator over the whole stream) instead of the heavy-hitter G-sum heuristic, and the built-in model certifies it: Chebyshev per row (p = 0.1, eps2 = sqrt(20/w)), binomial median tail over odd d rows, relative L2 bound 1 - sqrt(1 - eps2). Fails closed unless d is odd, w a power of two and d*log2(w) + d <= 128. UnivMon is sized by inverting that bound for every readout, and Stage 3 prices it from its rows and Pass 1's state-size function. Co-Authored-By: Claude Opus 5.5 --- crates/devtools/tests/stage_pipeline.rs | 5 +- .../executor/src/summary_kernels/univmon.rs | 45 +++- .../tests/univmon_candidates.rs | 16 +- .../tests/precompute_raw_samples.rs | 8 +- .../src/accuracy/estimators/mod.rs | 7 +- .../src/accuracy/estimators/univmon.rs | 212 +++++++++++++++--- crates/plan-selection/src/lib.rs | 54 +++++ crates/planner/tests/summary_sharing.rs | 24 +- docs/design_docs/concepts/accuracy-models.md | 2 +- .../proposals/univmon-frequency-summary.md | 19 +- 10 files changed, 340 insertions(+), 52 deletions(-) diff --git a/crates/devtools/tests/stage_pipeline.rs b/crates/devtools/tests/stage_pipeline.rs index a3fd44f24..90f0860d7 100644 --- a/crates/devtools/tests/stage_pipeline.rs +++ b/crates/devtools/tests/stage_pipeline.rs @@ -253,8 +253,9 @@ fn selected_label(document: &Value) -> String { } /// Example 2: the three SQL statistics over `flows` lower through the SQL -/// frontend, are reported as SQL, and plan. The built-in models have no -/// UnivMon accuracy model, so the selected plan is the exact one with the +/// frontend, are reported as SQL, and plan. The built-in models certify +/// UnivMon only for the L2 norm (Q3), and a UnivMon sized for ε = 0.01 costs +/// more than exact counting, so the selected plan is the exact one with the /// shared input (recorded, not required by the design). #[test] fn example2_plans_the_sql_workload() { diff --git a/crates/executor/src/summary_kernels/univmon.rs b/crates/executor/src/summary_kernels/univmon.rs index 9d679b5dc..dab550427 100644 --- a/crates/executor/src/summary_kernels/univmon.rs +++ b/crates/executor/src/summary_kernels/univmon.rs @@ -181,7 +181,9 @@ impl AggregateCore for UnivMonAccumulator { value: None, } => self.inner.calc_l1(), SketchStatistic::Cardinality => self.inner.calc_card(), - SketchStatistic::FrequencyL2 => self.inner.calc_l2(), + // Layer 0 sees the whole stream; its row-median F₂ is the readout + // the planner certifies (`calc_l2` is the heavy-hitter G-sum). + SketchStatistic::FrequencyL2 => self.inner.l2_sketch_layers[0].get_l2(), SketchStatistic::FrequencyEntropy => self.inner.calc_entropy(), other => return Err(format!("UnivMon does not answer {other:?}").into()), }) @@ -274,6 +276,47 @@ mod tests { } } + /// L2 reads layer 0's F₂ estimate, and at the planner's (0.01, 0.01) + /// sizing it is within 1% of a Zipf stream's true L2 norm. + #[test] + fn l2_is_layer0_f2_within_the_certified_bound() { + use asap_logical_optimizer::pass1::replacement::default_size_params; + use planner_types::ir::operator::agg_intent::default_cardinality; + use planner_types::ir::schema::{SketchAlgorithm, SketchParams}; + let SketchParams::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + } = default_size_params(SketchAlgorithm::UnivMon, &default_cardinality(), 0.01, 0.01) + else { + unreachable!() + }; + let mut state = UnivMonAccumulator::new( + heap_size as usize, + sketch_rows as usize, + sketch_cols as usize, + usize::from(layers), + ) + .unwrap(); + // Zipf(1) frequencies over 5,000 keys: key i occurs ⌊10,000 / i⌋ times. + let mut f2 = 0.0; + for key in 1..=5_000u32 { + let count = 10_000 / key; + f2 += f64::from(count) * f64::from(count); + for _ in 0..count { + state.insert_sample(f64::from(key)).unwrap(); + } + } + let l2 = state.estimate(&SketchStatistic::FrequencyL2).unwrap(); + assert_eq!(l2, state.sketch().l2_sketch_layers[0].get_l2()); + assert!( + (l2 - f2.sqrt()).abs() <= 0.01 * f2.sqrt(), + "{l2} vs {}", + f2.sqrt() + ); + } + /// Retained variable-length identities contribute to the runtime memory reservation. #[test] fn memory_accounts_for_string_identities() { diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index 3fbf8afb0..3d34603fa 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -68,11 +68,12 @@ fn candidate(query: &str, accuracy: AccuracyTarget) -> Rc { #[test] fn four_evaluations_share_one_value_frequency_state_and_keep_honest_guarantees() { - // Equal data, grouping and window produce one state independently of evaluation. + // Equal data, grouping, window and requirement produce one state + // independently of evaluation: UnivMon is sized for L2 whatever it reads. let accuracy = AccuracyTarget::Epsilon(0.02); let roots: Vec<_> = [ ("distinct_over_time(m[5m])", accuracy.clone()), - ("count_over_time(m[5m])", AccuracyTarget::Exact), + ("count_over_time(m[5m])", accuracy.clone()), ("l2_over_time(m[5m])", accuracy.clone()), ("entropy_over_time(m[5m])", accuracy), ] @@ -114,11 +115,14 @@ fn four_evaluations_share_one_value_frequency_state_and_keep_honest_guarantees() let Operator::ASAP(ASAPOp::SummaryAgg { family, .. }) = &summary_input.operator else { panic!() }; - assert!( + // Production certifies L2 from layer 0's F₂, but has no + // calibrated bound for distinct count or entropy. + assert_eq!( DefaultAccuracyModel .local_guarantee(family, query) - .is_none(), - "production has no calibrated error bound" + .map(|g| g.metric), + (*index == 2).then_some(ErrorMetric::RelativeValue), + "{query:?}" ); } post_asap_dag(root); @@ -129,7 +133,7 @@ fn four_evaluations_share_one_value_frequency_state_and_keep_honest_guarantees() fn uncalibrated_frequency_evaluations_do_not_bypass_accuracy_targets() { // An unmeasured heuristic remains inspectable but is never certified or // automatically selected for a caller-visible bounded-error result. - for query in ["entropy_over_time(m[5m])", "l2_over_time(m[5m])"] { + for query in ["entropy_over_time(m[5m])"] { for target in [ AccuracyTarget::Exact, AccuracyTarget::Epsilon(0.02), diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index 50f808210..1e0fc4df6 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -160,7 +160,13 @@ fn execute( window_end_ms: 6000, revision: 1, }, - Limits::default(), + // A UnivMon sized for L2 at ε = 0.02 holds 16 layers of 5 × 2^14 + // counters (≈ 10 MiB) per population, past the default 64 MiB for + // this test's five series. + Limits { + max_bytes: 1 << 30, + ..Limits::default() + }, ) .unwrap(); block_on(async { diff --git a/crates/logical-optimizer/src/accuracy/estimators/mod.rs b/crates/logical-optimizer/src/accuracy/estimators/mod.rs index 5926e4b0a..a41aba115 100644 --- a/crates/logical-optimizer/src/accuracy/estimators/mod.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/mod.rs @@ -29,7 +29,7 @@ pub(super) fn sketch_guarantee( SketchParams::Kmv { .. } | SketchParams::Theta { .. } => { cardinality::guarantee(algorithm, params, query) } - SketchParams::UnivMon { .. } => univmon::guarantee(query), + SketchParams::UnivMon { .. } => univmon::guarantee(params, query), } } @@ -108,9 +108,8 @@ pub(crate) fn size_params( delta: f64, ) -> SketchParams { match kind { - // Baseline dimensions are candidates, not an inverted error bound. - // Empirical models may size these; no theoretical guarantee is claimed. - SketchAlgorithm::UnivMon => univmon::size_params(), + // Sized for the L2 readout whatever the intent (see `univmon`). + SketchAlgorithm::UnivMon => univmon::size_params(eps, delta), SketchAlgorithm::Kll => SketchParams::Kll { k: kll::kll_k(eps) }, SketchAlgorithm::Cms => SketchParams::Cms { width: cms::cms_width(eps), diff --git a/crates/logical-optimizer/src/accuracy/estimators/univmon.rs b/crates/logical-optimizer/src/accuracy/estimators/univmon.rs index 4a0293144..1adaa9b37 100644 --- a/crates/logical-optimizer/src/accuracy/estimators/univmon.rs +++ b/crates/logical-optimizer/src/accuracy/estimators/univmon.rs @@ -1,8 +1,26 @@ -//! UnivMon certifies only its unit-update total, which the kernel reads -//! exactly (`calc_l1` returns the update count). +//! UnivMon certifies its unit-update total, which the kernel reads exactly +//! (`calc_l1` returns the update count), and the L2 norm read from layer 0's +//! CountSketch. //! -//! Distinct count, L2 norm and entropy are deliberately left uncertified, -//! so Stage 3 rejects them unless a deployment supplies accuracy evidence: +//! **L2.** Layer 0 sees the whole stream. Each of its `d` rows keeps +//! `R = Σ_j C[j]²`, the AMS F₂ estimator: with fully random hashing, +//! `E[R] = F₂` and `Var[R] ≤ 2F₂²/w`. Chebyshev puts a row outside +//! `(1 ± ε₂)·F₂` with probability at most `p = 0.1` when +//! `ε₂ = √(2/(w·p)) = √(20/w)`. The kernel reads `√(median of the rows)`, +//! which is outside the interval only if at least `(d+1)/2` of `d` (odd) +//! independent rows are, so `δ = Σ_{i ≥ (d+1)/2} C(d,i) pⁱ (1−p)^(d−i)`. +//! Inside it the relative L2 error is at most `1 − √(1 − ε₂)`, the lower +//! side; the upper side `√(1 + ε₂) − 1` is smaller since `√` is concave. +//! States merge by adding counters, so the bound holds for merged panes. +//! +//! sketchlib slices every row's column and sign from one 128-bit hash (row +//! `r` uses bits `[r·m, (r+1)·m)` for `m = log₂ w` and sign bit `127 − r`), +//! so the rows are independent only when those slices are disjoint: +//! `d·m + d ≤ 128` with `w` a power of two. Otherwise no L2 bound is given. +//! The contract records the idealized-hash assumption. +//! +//! **Distinct count and entropy** are deliberately left uncertified, so +//! Stage 3 rejects them unless a deployment supplies accuracy evidence: //! //! - Liu et al., "One Sketch to Rule Them All" (SIGCOMM 2016), rely on //! Braverman and Ostrovsky's recursive sketch. With O(log n) layers, each of @@ -21,26 +39,107 @@ //! heap holds every item, a CountSketch estimate may be at most zero for a //! present item, so the distinct count is not certified in that case //! either. -//! - A per-layer CountSketch does give a sound F₂ (AMS) bound, but `calc_l2` -//! reads the recursive sum, not that estimate. //! -//! A sound guarantee needs either a readout with a proven bound (for -//! example, L2 from layer 0's CountSketch) or a per-layer cover guarantee for -//! the heap. Both are kernel changes. +//! A sound guarantee for them needs a per-layer cover guarantee for the +//! heap, a kernel change. use super::*; -pub(super) fn guarantee(query: &SketchStatistic) -> Option { - matches!(query, SketchStatistic::PointCount { value: None, .. }) - .then(|| ResultGuarantee::exact("univmon_unit_update_total")) +/// The Chebyshev failure probability of one row's F₂ estimate. +const ROW_FAILURE: f64 = 0.1; + +/// `asap-executor`'s UnivMon kernel rejects more than 20 rows. +const MAX_ROWS: u32 = 19; + +pub(super) fn guarantee(params: &SketchParams, query: &SketchStatistic) -> Option { + match query { + SketchStatistic::PointCount { value: None, .. } => { + Some(ResultGuarantee::exact("univmon_unit_update_total")) + } + SketchStatistic::FrequencyL2 => l2_guarantee(params), + _ => None, + } } -/// One shape for every readout and requirement: no bound is inverted (see -/// the module docs), so every consumer of a key reads identical states. -pub(super) fn size_params() -> SketchParams { +fn l2_guarantee(params: &SketchParams) -> Option { + let SketchParams::UnivMon { + sketch_rows: d, + sketch_cols: w, + .. + } = params + else { + return None; + }; + let hash_bits = u64::from(*d) * (u64::from(w.checked_ilog2()?) + 1); + if d.is_multiple_of(2) || !w.is_power_of_two() || hash_bits > 128 { + return None; + } + let f2_error = (20.0 / f64::from(*w)).sqrt(); + if f2_error >= 1.0 { + return None; + } + Some(ResultGuarantee { + metric: ErrorMetric::RelativeValue, + bound: BoundExpr::Constant { + value: 1.0 - (1.0 - f2_error).sqrt(), + }, + failure_probability: ProbabilityExpr::Constant { + value: median_failure(*d), + }, + provenance: vec![GuaranteeSource::SketchEvaluation { + algorithm: "UnivMon".into(), + contract: "univmon_layer0_f2_median_chebyshev_v1".into(), + params: serde_json::json!({"sketch_rows": d, "sketch_cols": w, + "row_failure_probability": ROW_FAILURE, + "hash_assumption": "independent_uniform_128_bit_hash", + "readout": "sqrt_median_layer0_row_f2"}), + query: "FrequencyL2".into(), + }], + }) +} + +/// Probability that at least `(d+1)/2` of `d` independent rows fail, each +/// with probability [`ROW_FAILURE`]. +fn median_failure(d: u32) -> f64 { + let (p, n) = (ROW_FAILURE, i32::try_from(d).unwrap_or(i32::MAX)); + let mut binomial = 1.0; // C(d, i) + let mut tail = 0.0; + for i in 0..=n { + if 2 * i > n { + tail += binomial * p.powi(i) * (1.0 - p).powi(n - i); + } + binomial *= f64::from(n - i) / f64::from(i + 1); + } + tail +} + +/// One shape per `(ε, δ)` for every readout, so every consumer of a Pass 2 +/// key reads identical states: the L2 sizing, which the total does not need +/// and distinct count and entropy cannot use. `w` is the smallest power of +/// two with `1 − √(1 − √(20/w)) ≤ ε`, `d` the smallest odd depth with +/// median failure at most δ. Parameters past the hash's bit budget are kept; +/// the guarantee then rejects them. Without a positive ε (an exact count, +/// which reads only the exact total) the shape is a fixed baseline. +pub(super) fn size_params(eps: f64, delta: f64) -> SketchParams { + if eps.is_nan() || eps <= 0.0 { + return SketchParams::UnivMon { + heap_size: 256, + sketch_rows: 5, + sketch_cols: 1024, + layers: 16, + }; + } + let eps = eps.min(1.0); + let f2_error = 2.0 * eps - eps * eps; + let sketch_cols = + saturating_ceil(20.0 / (f2_error * f2_error), 32, 1 << 26).next_power_of_two(); + let sketch_rows = (1..=MAX_ROWS) + .step_by(2) + .find(|&d| median_failure(d) <= delta) + .unwrap_or(MAX_ROWS); SketchParams::UnivMon { heap_size: 256, - sketch_rows: 5, - sketch_cols: 1024, + sketch_rows, + sketch_cols, layers: 16, } } @@ -51,35 +150,98 @@ mod tests { use asap_types::ir::scalar::ColumnRef; use asap_types::ir::schema::{GroupingStrategy, SketchKind}; - fn family() -> FieldDataType { + fn family(params: SketchParams) -> FieldDataType { FieldDataType::Sketch( - SketchKind::new(SketchAlgorithm::UnivMon, size_params()), + SketchKind::new(SketchAlgorithm::UnivMon, params), GroupingStrategy::default(), ) } - /// The exact total is certified; distinct count, L2 and entropy have no - /// sound bound for the shipped kernel and stay uncertified. + fn l2(params: SketchParams) -> Option { + DefaultAccuracyModel.local_guarantee(&family(params), &SketchStatistic::FrequencyL2) + } + + fn shape(params: &SketchParams) -> (u32, u32) { + let SketchParams::UnivMon { + sketch_rows, + sketch_cols, + .. + } = params + else { + unreachable!() + }; + (*sketch_rows, *sketch_cols) + } + + fn with_shape(sketch_rows: u32, sketch_cols: u32) -> SketchParams { + SketchParams::UnivMon { + heap_size: 256, + sketch_rows, + sketch_cols, + layers: 16, + } + } + + /// The exact total and L2 are certified; distinct count and entropy have + /// no sound bound for the shipped kernel and stay uncertified. #[test] - fn only_the_total_is_certified() { + fn the_total_and_l2_are_certified() { + let params = size_params(0.01, 0.01); let total = SketchStatistic::PointCount { key: ColumnRef::SampleValue, value: None, }; assert!(DefaultAccuracyModel - .local_guarantee(&family(), &total) + .local_guarantee(&family(params.clone()), &total) .is_some_and(|g| g.is_exact())); + let guarantee = l2(params.clone()).expect("L2 is certified"); + assert_eq!(guarantee.metric, ErrorMetric::RelativeValue); + assert!(DefaultAccuracyModel.satisfies( + &guarantee, + &AccuracyTarget::EpsilonDelta { + epsilon: 0.01, + delta: 0.01, + } + )); for query in [ SketchStatistic::Cardinality, - SketchStatistic::FrequencyL2, SketchStatistic::FrequencyEntropy, ] { assert!( DefaultAccuracyModel - .local_guarantee(&family(), &query) + .local_guarantee(&family(params.clone()), &query) .is_none(), "{query:?}" ); } } + + /// (0.01, 0.01) sizes 5 rows of 2^16 columns: √(20/2^16) ≈ 0.0175 F₂ + /// error, ≈ 0.0088 L2 error, and a median failure of ≈ 0.0086. + #[test] + fn sizing_inverts_the_bound() { + assert_eq!(shape(&size_params(0.01, 0.01)), (5, 1 << 16)); + let (rows, cols) = shape(&size_params(0.01, 0.01)); + let (tighter_rows, _) = shape(&size_params(0.01, 0.001)); + let (_, wider_cols) = shape(&size_params(0.005, 0.01)); + assert!(tighter_rows > rows && wider_cols > cols); + let guarantee = l2(size_params(0.01, 0.01)).unwrap(); + assert!(guarantee.bound.evaluate().unwrap() <= 0.01); + assert!(guarantee.failure_probability.evaluate().unwrap() <= 0.01); + } + + /// Even depth, a non-power-of-two width, or rows whose hash slices + /// overlap get no L2 bound. + #[test] + fn l2_fails_closed_outside_the_hash_contract() { + assert!(l2(with_shape(5, 1 << 16)).is_some()); + assert!(l2(with_shape(4, 1 << 16)).is_none()); + assert!(l2(with_shape(5, 1000)).is_none()); + // 7 · (16 + 1) = 119 bits fit; 9 · 17 = 153 do not. + assert!(l2(with_shape(7, 1 << 16)).is_some()); + assert!(l2(with_shape(9, 1 << 16)).is_none()); + // (0.01, 0.001) needs 9 rows of 2^16 columns: sized, but rejected. + assert_eq!(shape(&size_params(0.01, 0.001)), (9, 1 << 16)); + assert!(l2(size_params(0.01, 0.001)).is_none()); + } } diff --git a/crates/plan-selection/src/lib.rs b/crates/plan-selection/src/lib.rs index 9f0448d7b..8d661504f 100644 --- a/crates/plan-selection/src/lib.rs +++ b/crates/plan-selection/src/lib.rs @@ -2029,6 +2029,13 @@ fn summary_shape(family: &FieldDataType) -> (u64, u64) { u64::from(*depth) + 1, 8 * u64::from(*width) * u64::from(*depth) + 24 * u64::from(*heap_size), ), + // An insert reaches about two layers; each updates `d` rows' + // counters and sign-checks, then its heap. + params @ SketchParams::UnivMon { sketch_rows, .. } => ( + 2 * (2 * u64::from(*sketch_rows) + 4), + asap_logical_optimizer::pass1::replacement::sketch_state_bytes(params) + .unwrap_or(u64::MAX), + ), _ => (1, 1_024), }, _ => (1, 8), @@ -2466,6 +2473,53 @@ mod tests { assert_eq!(states.len(), 1); } + /// #509 Example 2's requirements on one UnivMon each: the built-in model + /// certifies the L2 norm (Q3) from layer 0's F₂, but neither the + /// distinct count (Q1) nor the entropy (Q2). + #[test] + fn univmon_certifies_l2_but_not_distinct_count_or_entropy() { + for (query, epsilon, certified) in [ + ("distinct_over_time(src[1m])", 0.02, false), + ("entropy_over_time(src[1m])", 0.05, false), + ("l2_over_time(src[1m])", 0.01, true), + ] { + let target = AccuracyTarget::EpsilonDelta { + epsilon, + delta: 0.01, + }; + let root = lower_promql(query, target.clone()); + let root = asap_types::ir::schema_support::with_promql_series_identity(&root).unwrap(); + let inventory = + asap_logical_optimizer::pass1::logical_candidates::enumerate_local_logical_candidates( + vec![(0, QueryRoot::Operator(root))], + &Default::default(), + ) + .unwrap(); + let choice: Vec<_> = inventory + .targets + .iter() + .map(|t| { + t.alternatives + .iter() + .position(|a| matches!(a, asap_logical_optimizer::Realization::Sketch(kind) if *kind.algorithm() == SketchAlgorithm::UnivMon)) + .expect("a UnivMon alternative") + }) + .collect(); + let (_, candidate) = realize_choice(&inventory, &choice).unwrap(); + let violation = accuracy_violation( + &candidate, + &[every_10s(Some(target))], + &PlanningModels::builtin(), + ); + if certified { + assert_eq!(violation, None, "{query}"); + } else { + let reason = violation.expect(query); + assert!(reason.contains("no accuracy model for UnivMon"), "{reason}"); + } + } + } + /// A candidate that cannot be checked is rejected with its reason; the /// others are still selected among. #[test] diff --git a/crates/planner/tests/summary_sharing.rs b/crates/planner/tests/summary_sharing.rs index ba5813d8a..f2404b061 100644 --- a/crates/planner/tests/summary_sharing.rs +++ b/crates/planner/tests/summary_sharing.rs @@ -559,9 +559,10 @@ fn certified_frequency_evaluations_share_one_univmon_state() { /// #509 Example 2 through the stage pipeline: distinct count, entropy and L2 /// of one input over one window, with requirements ε = 0.02, 0.05 and 0.01. /// The summary-capability variant gives the three targets one UnivMon. -/// Certified by the synthetic model, that one state serves all three -/// queries. The built-in model has no sound bound for these readouts, so -/// Stage 3 rejects every UnivMon plan. +/// Certified by the synthetic model, the plan whose three estimates read one +/// UnivMon build is valid; building it (≈ 28 updates per row) still costs +/// more than exact counting, so it is not selected. The built-in model +/// certifies only the L2 readout, and no UnivMon is selected either. #[tokio::test] async fn frequency_moments_share_one_univmon_in_the_stage_pipeline() { let queries = [ @@ -600,10 +601,19 @@ async fn frequency_moments_share_one_univmon_in_the_stage_pipeline() { }; let certified = plan(PlanningModels::builtin().with_accuracy(&UnivMonEvidence)).await; - let states = univmon_states(&certified); - assert_eq!(states.len(), 3, "{:?}", certified.selection); - assert!(states.iter().all(|state| Rc::ptr_eq(state, &states[0]))); - assert_eq!(unique_deployments(&certified), 1); + let selection = certified.selection.as_ref().expect("a selection"); + assert!( + selection.costs.values().any(|cost| { + let count = |prefix: &str| { + cost.per_node + .values() + .filter(|node| node.detail.starts_with(prefix)) + .count() + }; + count("build UnivMon") == 1 && count("estimate") == 3 + }), + "{selection:?}" + ); let builtin = plan(PlanningModels::builtin()).await; assert!(univmon_states(&builtin).is_empty()); diff --git a/docs/design_docs/concepts/accuracy-models.md b/docs/design_docs/concepts/accuracy-models.md index 39e7c468b..d2588edad 100644 --- a/docs/design_docs/concepts/accuracy-models.md +++ b/docs/design_docs/concepts/accuracy-models.md @@ -121,7 +121,7 @@ The built-in models currently include: | CMS | L1-normalized frequency bound from width and depth; does not by itself certify TopK membership | | CountSketch | L2-normalized frequency bound and median concentration bound, requiring valid odd depth | | KMV / Theta | Parameter-derived cardinality bounds using the registered variance/Chebyshev model at 99% confidence | -| UnivMon | Exact unit-update total for the supported evaluation; no universal guarantee for all its statistics | +| UnivMon | Exact unit-update total; relative L2 bound from layer 0's row-median F2 (Chebyshev per row, binomial median tail), requiring odd depth, power-of-two width and disjoint hash slices (`d·log2(w) + d ≤ 128`) under an idealized hash; no guarantee for distinct count or entropy | | Other families/evaluations | No default certificate where no accuracy model is registered | This table describes Planner's registered contracts, not independent diff --git a/docs/design_docs/proposals/univmon-frequency-summary.md b/docs/design_docs/proposals/univmon-frequency-summary.md index 75cfd080b..27a9ac73b 100644 --- a/docs/design_docs/proposals/univmon-frequency-summary.md +++ b/docs/design_docs/proposals/univmon-frequency-summary.md @@ -25,12 +25,21 @@ readout. External differential tests must establish the engine's semantics. The existing `asap_sketchlib::UnivMon` implementation supplies `calc_card`, `calc_l1`, `calc_l2`, and `calc_entropy`. Its entropy uses log base 2. Its L1 -counter is exact for nonnegative updates; the other estimators are heuristic. +counter is exact for nonnegative updates; `calc_card`, `calc_l2` and +`calc_entropy` are heavy-hitter heuristics. The executor therefore reads L2 +from layer 0's CountSketch instead (`get_l2`, the square root of the row +median of `Σ C²`), which sees the whole stream and is the textbook F2 +estimator. With `w` columns each row's F2 is within `√(20/w)` relative error +except with probability 0.1 (Chebyshev), and the median of `d` odd rows fails +with the binomial tail; the L2 bound is `1 − √(1 − √(20/w))`. This needs a +power-of-two `w` and `d·log2(w) + d ≤ 128`, so the rows' slices of +sketchlib's one 128-bit hash are disjoint, and assumes an idealized hash. The parameter contract records heap size, sketch rows, sketch columns and -number of layers. Default dimensions define a candidate configuration, not -an epsilon guarantee. A deployment accuracy model must supply calibrated -evidence before an approximate readout can satisfy an accuracy target. Without -that evidence, the Planner keeps the exact sub-DAG. HLL/Theta/KMV remain +number of layers; rows and columns are sized from the requirement by +inverting that bound for every readout, so consumers that share one state +read identical states. A deployment accuracy model must supply calibrated +evidence before a distinct-count or entropy readout can satisfy an accuracy +target. Without that evidence, the Planner keeps the exact sub-DAG. HLL/Theta/KMV remain cardinality alternatives, and exact count remains the cheaper first count candidate. From e34c85c7f04dfed8c35c86669faa9afd5b51dfb3 Mon Sep 17 00:00:00 2001 From: zzylol <50204836+zzylol@users.noreply.github.com> Date: Mon, 5 Oct 2026 02:24:02 +0000 Subject: [PATCH 2/2] fix: integrate with #620 #620's priced_example2_candidates_bind requires every priced Stage 3 candidate to bind. With UnivMon L2 certified, Example 2's UnivMon over src_ip (unit-weight item update) is now priced, but the physical planner sent it to the keyed weighted-frequency build, which has no UnivMon kernel. Bind it as UnivMon's value-frequency build over the item column, which already counts typed SQL identities. Co-Authored-By: Claude Opus 5.5 --- crates/executor/src/physical_planner/mod.rs | 23 +++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/crates/executor/src/physical_planner/mod.rs b/crates/executor/src/physical_planner/mod.rs index c2389af76..daf25da65 100644 --- a/crates/executor/src/physical_planner/mod.rs +++ b/crates/executor/src/physical_planner/mod.rs @@ -987,6 +987,29 @@ fn summary_build( groups(input, keys)?, ); } + // UnivMon counts one occurrence of each row's item (SQL `COUNT(*) GROUP + // BY item`): its value-frequency build over the item column. + if let ( + FieldDataType::Sketch(kind, _), + Some(SummaryInputExpr::Column(item)), + SummaryInputExpr::Constant(weight), + ) = (family, &update.item, &update.weight) + { + if kind.algorithm() == &planner_types::ir::schema::SketchAlgorithm::UnivMon + && *weight == 1.0 + { + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid("UnivMon requires explicit grouping columns")); + }; + return Operator::summary_build( + input.clone(), + family.clone(), + named_column(input, item)?, + None, + groups(input, keys)?, + ); + } + } if let Some(item) = &update.item { let PlannerReduction::Reduce(keys) = reduction else { return Err(invalid("keyed summary requires explicit partitions"));