spg_engine/aggregate.rs
1//! Aggregate executor.
2//!
3//! Handles `SELECT … <aggs> … [GROUP BY …]` queries. The planning strategy
4//! is straightforward:
5//!
6//! 1. Walk the SELECT (and ORDER BY) expressions to find every aggregate
7//! function call. Dedupe by AST equality and assign each `__agg_<i>`.
8//! 2. Same for every `GROUP BY` expression: assign `__grp_<j>`.
9//! 3. Stream the WHERE-filtered rows, group by the tuple of GROUP BY
10//! values, and update per-group aggregate state.
11//! 4. Materialise a synthetic per-group row containing
12//! `[__grp_0..__grp_K, __agg_0..__agg_N]` and rewrite the user's
13//! SELECT / ORDER BY expressions to reference those synthetic columns
14//! instead of the originals.
15//! 5. Evaluate the rewritten expressions against the synthetic schema and
16//! emit results.
17//!
18//! v1.8 implements `count(*)`, `count(expr)`, `sum`, `min`, `max`, `avg`.
19//! NULL semantics follow PG: aggregates skip NULL inputs (except
20//! `count(*)`, which counts rows). `sum(int)` widens to `BigInt`;
21//! `avg(int|bigint)` returns `Float`.
22
23use alloc::borrow::Cow;
24use alloc::boxed::Box;
25use alloc::collections::BTreeSet;
26use alloc::format;
27use alloc::string::{String, ToString};
28use alloc::vec::Vec;
29
30use spg_sql::ast::{Expr, SelectItem, SelectStatement};
31use spg_storage::{ColumnSchema, DataType, Row, Value};
32
33use crate::eval::{self, EvalContext, EvalError};
34use crate::join::AggRows;
35
36impl crate::Engine {
37 /// v7.39 (round 763, F31-C1) — expand a `*` / `alias.*` SELECT item
38 /// into explicit column refs when the statement takes the aggregate
39 /// path and the FROM is one plain catalog table. Returns `None`
40 /// when nothing applies (the caller keeps the original statement).
41 /// Joined / derived / SRF sources keep the old refusal for now.
42 pub(crate) fn expand_aggregate_wildcard(
43 &self,
44 stmt: &SelectStatement,
45 ) -> Option<SelectStatement> {
46 use spg_sql::ast::SelectItem;
47 if !stmt
48 .items
49 .iter()
50 .any(|i| matches!(i, SelectItem::Wildcard | SelectItem::QualifiedWildcard(_)))
51 {
52 return None;
53 }
54 if !uses_aggregate(stmt) {
55 return None;
56 }
57 let from = stmt.from.as_ref()?;
58 if !from.joins.is_empty()
59 || from.primary.unnest_expr.is_some()
60 || from.primary.lateral_subquery.is_some()
61 || from.primary.generate_series_args.is_some()
62 || from.primary.table_fn_call.is_some()
63 || from.primary.json_table.is_some()
64 || from.primary.jsonb_each_text_arg.is_some()
65 {
66 return None;
67 }
68 let table = self.active_catalog().get(&from.primary.name)?;
69 let alias = from
70 .primary
71 .alias
72 .clone()
73 .unwrap_or_else(|| from.primary.name.clone());
74 let mut items: Vec<SelectItem> = Vec::with_capacity(stmt.items.len());
75 for item in &stmt.items {
76 match item {
77 SelectItem::Wildcard => {
78 for c in &table.schema().columns {
79 items.push(SelectItem::Expr {
80 expr: Expr::Column(spg_sql::ast::ColumnName {
81 qualifier: None,
82 name: c.name.clone(),
83 }),
84 alias: None,
85 });
86 }
87 }
88 SelectItem::QualifiedWildcard(q) => {
89 if !q.eq_ignore_ascii_case(&alias) {
90 return None; // unknown qualifier — keep the old path
91 }
92 // Bare names: the single-table qualifier is
93 // redundant, and the group-expr matcher unifies
94 // bare-to-bare (a qualified ref would miss a bare
95 // GROUP BY id).
96 for c in &table.schema().columns {
97 items.push(SelectItem::Expr {
98 expr: Expr::Column(spg_sql::ast::ColumnName {
99 qualifier: None,
100 name: c.name.clone(),
101 }),
102 alias: None,
103 });
104 }
105 }
106 other => items.push(other.clone()),
107 }
108 }
109 let mut out = stmt.clone();
110 out.items = items;
111 Some(out)
112 }
113}
114
115/// True if this statement should go through the aggregate path.
116pub fn uses_aggregate(stmt: &SelectStatement) -> bool {
117 if stmt.group_by.is_some() || stmt.having.is_some() {
118 return true;
119 }
120 uses_aggregate_ignoring_group_by(stmt)
121}
122
123/// v7.38.13 — the same question with the GROUP BY / HAVING short-circuit
124/// removed: does an aggregate CALL appear anywhere? `baregroup` needs
125/// this to tell a grouped aggregate from a GROUP BY that is a DISTINCT.
126pub(crate) fn uses_aggregate_ignoring_group_by(stmt: &SelectStatement) -> bool {
127 for item in &stmt.items {
128 if let SelectItem::Expr { expr, .. } = item
129 && contains_aggregate(expr)
130 {
131 return true;
132 }
133 }
134 for o in &stmt.order_by {
135 if contains_aggregate(&o.expr) {
136 return true;
137 }
138 }
139 if let Some(h) = &stmt.having
140 && contains_aggregate(h)
141 {
142 return true;
143 }
144 false
145}
146
147pub fn contains_aggregate(e: &Expr) -> bool {
148 match e {
149 Expr::FunctionCall { name, args } => {
150 is_aggregate_name(name) || args.iter().any(contains_aggregate)
151 }
152 Expr::NamedArg { expr, .. } => contains_aggregate(expr),
153 Expr::Variadic(expr) => contains_aggregate(expr),
154 Expr::AggregateOrdered { .. } => true,
155 Expr::Binary { lhs, rhs, .. } => contains_aggregate(lhs) || contains_aggregate(rhs),
156 Expr::Unary { expr, .. }
157 | Expr::Cast { expr, .. }
158 | Expr::IsNull { expr, .. }
159 | Expr::BoolTest { expr, .. }
160 | Expr::FieldAccess { base: expr, .. } => contains_aggregate(expr),
161 Expr::Like { expr, pattern, .. } => contains_aggregate(expr) || contains_aggregate(pattern),
162 Expr::Extract { source, .. } => contains_aggregate(source),
163 // v4.10 subqueries + v4.12 window functions / Literal /
164 // Column — all non-aggregate leaves from the regular
165 // aggregate planner's POV. Window-bearing projections are
166 // routed to exec_select_with_window before this runs.
167 Expr::ScalarSubquery(_)
168 | Expr::Exists { .. }
169 | Expr::InSubquery { .. }
170 | Expr::RowInSubquery { .. }
171 | Expr::RowCmpSubquery { .. }
172 | Expr::WindowFunction { .. }
173 | Expr::Literal(_)
174 | Expr::Placeholder(_)
175 | Expr::Column(_) => false,
176 // v7.10.10 — recurse into array constructor / subscript /
177 // ANY/ALL children. Aggregates inside `ARRAY[SUM(x)]` are
178 // valid PG and must be detected here.
179 Expr::Array(items) => items.iter().any(contains_aggregate),
180 Expr::ArraySubscript { target, index } => {
181 contains_aggregate(target) || contains_aggregate(index)
182 }
183 Expr::ArraySlice { target, lo, hi } => {
184 contains_aggregate(target)
185 || lo.as_deref().is_some_and(contains_aggregate)
186 || hi.as_deref().is_some_and(contains_aggregate)
187 }
188 Expr::AnyAll { expr, array, .. } => contains_aggregate(expr) || contains_aggregate(array),
189 Expr::InList { expr, list, .. } => {
190 contains_aggregate(expr) || list.iter().any(contains_aggregate)
191 }
192 // v7.13.0 — CASE WHEN … END. Recurse into operand,
193 // every (WHEN, THEN) pair, and the ELSE branch.
194 Expr::Case {
195 operand,
196 branches,
197 else_branch,
198 } => {
199 operand.as_deref().is_some_and(contains_aggregate)
200 || branches
201 .iter()
202 .any(|(w, t)| contains_aggregate(w) || contains_aggregate(t))
203 || else_branch.as_deref().is_some_and(contains_aggregate)
204 }
205 }
206}
207
208pub fn is_aggregate_name(name: &str) -> bool {
209 matches!(
210 name.to_ascii_lowercase().as_str(),
211 "count"
212 | "count_star"
213 | "sum"
214 | "min"
215 | "max"
216 | "avg"
217 // v7.17.0 — variadic / collection aggregates. ORM
218 // reports (Hibernate / Rails / Django) emit these in
219 // GROUP BY rollups; pre-7.17 SPG hit "unknown
220 // aggregate".
221 | "string_agg"
222 | "array_agg"
223 // PG 16+ — any_value: an arbitrary non-NULL value from
224 // the group (SPG: the first seen, deterministic for
225 // ordered input).
226 | "any_value"
227 // PG 14+ — range_agg: collect ranges into a multirange
228 // (insertion order, no coalescing — matches the
229 // multirange constructor contract).
230 | "range_agg"
231 // PG 14+ — range_intersect_agg: intersection fold.
232 | "range_intersect_agg"
233 // MySQL group_concat (string_agg with ',' default) +
234 // SQL/XML xmlagg (separator-less concatenation).
235 | "group_concat"
236 | "xmlagg"
237 // v7.17.0 — boolean aggregates. `every` is SQL-standard
238 // alias for `bool_and`.
239 | "bool_and"
240 | "bool_or"
241 | "every"
242 // v7.32 (round-29) — statistical aggregates (every BI /
243 // dashboard emits these in rollups).
244 | "stddev" | "stddev_samp" | "stddev_pop"
245 | "variance" | "var_samp" | "var_pop"
246 // v7.32 (round-29) — bitwise aggregates.
247 | "bit_and" | "bit_or" | "bit_xor"
248 // v7.32 (round-29) — ordered-set aggregates (used with
249 // `WITHIN GROUP (ORDER BY …)`).
250 | "percentile_cont" | "percentile_disc" | "mode"
251 // v7.32 (round-29) — hypothetical-set aggregates (also
252 // `WITHIN GROUP`): the rank the direct args WOULD have.
253 | "rank" | "dense_rank" | "percent_rank" | "cume_dist"
254 // v7.32 (round-29) — two-argument regression family.
255 | "covar_pop" | "covar_samp" | "corr"
256 | "regr_count" | "regr_avgx" | "regr_avgy" | "regr_slope"
257 | "regr_intercept" | "regr_r2" | "regr_sxx" | "regr_syy" | "regr_sxy"
258 // v7.32 (round-29) — JSON aggregates.
259 | "json_agg" | "jsonb_agg" | "json_object_agg" | "jsonb_object_agg"
260 | "json_agg_strict" | "jsonb_agg_strict"
261 | "json_object_agg_strict" | "jsonb_object_agg_strict"
262 | "json_object_agg_unique" | "jsonb_object_agg_unique"
263 | "json_object_agg_unique_strict" | "jsonb_object_agg_unique_strict"
264 // SQL:2016 standard spellings (PG 16+ accepts both).
265 | "json_arrayagg" | "json_objectagg"
266 )
267}
268
269/// v7.32 (round-29) — two-argument regression aggregates `f(Y, X)`.
270fn is_regression_name(name: &str) -> bool {
271 matches!(
272 name,
273 "covar_pop"
274 | "covar_samp"
275 | "corr"
276 | "regr_count"
277 | "regr_avgx"
278 | "regr_avgy"
279 | "regr_slope"
280 | "regr_intercept"
281 | "regr_r2"
282 | "regr_sxx"
283 | "regr_syy"
284 | "regr_sxy"
285 )
286}
287
288/// v7.32 (round-29) — aggregates that consume a second positional
289/// argument: `string_agg(v, sep)`, the regression family `f(Y, X)`, and
290/// `json_object_agg(key, value)`.
291fn agg_uses_second_arg(name: &str) -> bool {
292 // v7.39 (round 354, M12) — group_concat's SEPARATOR is lowered onto the
293 // same second argument string_agg takes; without this the separator was
294 // parsed and then dropped, so `SEPARATOR '|'` silently kept the default
295 // comma.
296 name == "group_concat"
297 || name == "string_agg"
298 || name.starts_with("json_object_agg")
299 || name.starts_with("jsonb_object_agg")
300 || name == "jsonb_object_agg"
301 || name == "json_objectagg"
302 || is_regression_name(name)
303}
304
305/// v7.32 (round-29) — ordered-set aggregates: the value to aggregate
306/// comes from the `WITHIN GROUP (ORDER BY …)` sort spec, and any
307/// in-parens arguments are *direct* arguments (the percentile fraction).
308/// `mode()` takes no direct argument.
309pub fn is_ordered_set_name(name: &str) -> bool {
310 // v7.32 — `eq_ignore_ascii_case` instead of `to_ascii_lowercase()`:
311 // these classifiers run in the aggregate row/group loop, where the
312 // old per-call `String` allocation showed up as ~16% of the inbox's
313 // aggregate path in a sampled profile (the names are constant).
314 ["percentile_cont", "percentile_disc", "mode"]
315 .iter()
316 .any(|k| name.eq_ignore_ascii_case(k))
317}
318
319/// v7.32 (round-29) — hypothetical-set aggregates: `rank(args) WITHIN
320/// GROUP (ORDER BY …)` and friends compute the rank the hypothetical
321/// row would have. Like ordered-set, the value stream comes from the
322/// sort spec and the in-parens args are direct (the hypothetical row).
323pub fn is_hypothetical_set_name(name: &str) -> bool {
324 ["rank", "dense_rank", "percent_rank", "cume_dist"]
325 .iter()
326 .any(|k| name.eq_ignore_ascii_case(k))
327}
328
329/// v7.32 (round-29) — every aggregate that takes its value stream from
330/// a `WITHIN GROUP (ORDER BY …)` clause (ordered-set + hypothetical-set).
331pub fn is_within_group_name(name: &str) -> bool {
332 is_ordered_set_name(name) || is_hypothetical_set_name(name)
333}
334
335/// v7.37.4 (R34) — pre-computed aggregate kind. Replaces per-row
336/// string matches in `update_state` with a single `match` on a
337/// `Copy` enum (compiles to a jump table). For the mailrs prod
338/// `/api/conversations` shape (14 aggregates × 100 k rows = 1.4 M
339/// inner-loop iterations) this is the dominant per-row cost.
340///
341/// Lowered from `AggSpec::name` at spec build time via
342/// [`classify_agg_name`]; populated by the three `AggSpec`
343/// construction sites (window+ORDER, plain, `first_ordered`
344/// `array_agg`).
345#[derive(Copy, Clone, Debug, PartialEq, Eq)]
346pub(crate) enum AggKind {
347 CountStar,
348 Count,
349 Sum,
350 Avg,
351 Min,
352 Max,
353 /// PG 16+ any_value — first non-NULL value seen.
354 AnyValue,
355 /// PG 14+ range_agg — collect ranges into a multirange.
356 RangeAgg,
357 /// PG 14+ range_intersect_agg — intersection fold over ranges.
358 RangeIntersectAgg,
359 StringAgg,
360 ArrayAgg,
361 BoolAnd,
362 BoolOr,
363 /// stddev / stddev_samp / stddev_pop / variance / var_samp / var_pop.
364 StddevFamily,
365 BitAnd,
366 BitOr,
367 BitXor,
368 /// ordered-set (`percentile_cont/disc`, `mode`) +
369 /// hypothetical-set (`rank`/`dense_rank`/etc.) aggregates that
370 /// share the WITHIN-GROUP collection path.
371 WithinGroup,
372 /// covar_samp / covar_pop / corr / regr_*.
373 Regression,
374 JsonAgg,
375 JsonObjectAgg,
376}
377
378/// v7.37.4 (R34) — name → kind, called once per spec at build time.
379/// Hot path (`update_state_kind`) only sees the enum; the canonical
380/// string still travels with the spec so `finalize` and errors can
381/// quote it.
382/// v7.39 (round 231) — the spelling `classify_agg_name` / `update_state` /
383/// `finalize` expect. PG's `every` is a standard-SQL alias for `bool_and`
384/// and every accumulator keys off the latter. The GROUP BY builder folded
385/// it at two of its own call sites; the window path (round 230) reached
386/// `classify_agg_name` without folding and hit its panic arm, so
387/// `every(x) OVER (…)` aborted the query. One entry point now, and
388/// `every_aggregate_name_classifies` keeps the two name lists in step.
389pub(crate) fn canonical_agg_name(name: &str) -> &str {
390 if name.eq_ignore_ascii_case("every") {
391 "bool_and"
392 } else {
393 name
394 }
395}
396
397pub(crate) fn classify_agg_name(name: &str) -> AggKind {
398 match name {
399 "count_star" => AggKind::CountStar,
400 "count" => AggKind::Count,
401 "sum" => AggKind::Sum,
402 "avg" => AggKind::Avg,
403 "min" => AggKind::Min,
404 "max" => AggKind::Max,
405 "any_value" => AggKind::AnyValue,
406 "range_agg" => AggKind::RangeAgg,
407 "range_intersect_agg" => AggKind::RangeIntersectAgg,
408 "string_agg" | "group_concat" | "xmlagg" => AggKind::StringAgg,
409 "array_agg" => AggKind::ArrayAgg,
410 "bool_and" => AggKind::BoolAnd,
411 "bool_or" => AggKind::BoolOr,
412 "stddev" | "stddev_samp" | "stddev_pop" | "variance" | "var_samp" | "var_pop" => {
413 AggKind::StddevFamily
414 }
415 "bit_and" => AggKind::BitAnd,
416 "bit_or" => AggKind::BitOr,
417 "bit_xor" => AggKind::BitXor,
418 "json_agg" | "jsonb_agg" | "json_arrayagg" | "json_agg_strict" | "jsonb_agg_strict" => {
419 AggKind::JsonAgg
420 }
421 "json_object_agg"
422 | "jsonb_object_agg"
423 | "json_objectagg"
424 | "json_object_agg_strict"
425 | "jsonb_object_agg_strict"
426 | "json_object_agg_unique"
427 | "jsonb_object_agg_unique"
428 | "json_object_agg_unique_strict"
429 | "jsonb_object_agg_unique_strict" => AggKind::JsonObjectAgg,
430 n if is_within_group_name(n) => AggKind::WithinGroup,
431 n if is_regression_name(n) => AggKind::Regression,
432 other => panic!("classify_agg_name: unknown aggregate {other}"),
433 }
434}
435
436/// Per-aggregate running state.
437///
438/// The four `use_*` flags are independent observations about which value
439/// shapes have flowed through this accumulator (a single `sum()` can see both
440/// numeric and float inputs), not a discriminant — collapsing them into one
441/// enum would change accumulation semantics, and a bitflags word would hide
442/// which gate each fast path reads.
443#[allow(clippy::struct_excessive_bools)]
444#[derive(Debug, Default, Clone)]
445pub(crate) struct AggState {
446 /// The shared sum/avg running state (see `NumAcc`).
447 num: NumAcc,
448 extreme: Option<Value<'static>>,
449 /// v7.17.0 — running collection for string_agg / array_agg.
450 /// Each entry is one row's contribution (NULL preserved as
451 /// `Value::Null`; string_agg's finalize step drops them, but
452 /// array_agg keeps them). Pushing in insertion order matches
453 /// PG behaviour when no `ORDER BY` is given inside the
454 /// aggregate call.
455 items: Vec<Value<'static>>,
456 /// v7.39 (round 762, F31-C2) — per-item separator, parallel to
457 /// `items`. PG evaluates string_agg's separator PER ROW: element
458 /// i is prefixed by ITS row's separator (`string_agg(v,
459 /// '<'||v||'>')` over a,b,c answers `a<b>b<c>c`; a NULL separator
460 /// renders empty; a skipped-NULL value row's separator is never
461 /// used). Populated only on the general path when the call has a
462 /// second argument; the fused lane is literal-separator only and
463 /// keeps the single `separator` snapshot below.
464 item_seps: Vec<Option<String>>,
465 /// v7.25 (round-17) — per-group dedupe set for DISTINCT
466 /// aggregates (encoded values; NULLs never reach it because
467 /// the caller's skip runs after the per-aggregate NULL rules).
468 /// v7.37.4 measured `hashbrown::HashSet` as worse at this
469 /// shape — the per-(group × distinct-spec) hash table alloc
470 /// overhead beats the lookup-speed gain when each set is
471 /// small. Sticking with `BTreeSet`; the dispatch-side enum
472 /// fix in `update_state` is the R34 win.
473 seen: BTreeSet<String>,
474 /// v7.37.x (docker-fair DISTA attack) — fast-path BigInt seen
475 /// set. The hot DISTINCT path used `encode_key_refs_into` to
476 /// turn `Value::BigInt(n)` into a string key like `"I<n>|"` then
477 /// inserted that into the String BTreeSet — ~100 ns of pure alloc
478 /// + format churn per row × 25 k rows × 1 BigInt DISTINCT spec
479 /// (the DISTA `COUNT(DISTINCT m.id)` shape) ≈ 2.5 ms of waste.
480 /// Direct `BTreeSet<i64>` skips encode entirely; lookups stay
481 /// O(log small) on the per-group set. Lazy-allocated — only the
482 /// BigInt-DISTINCT path constructs it.
483 seen_int: Option<BTreeSet<i64>>,
484 /// v7.24 (round-16 A) — per-item ORDER BY key tuples, parallel
485 /// to `items` (pushed under the same skip/keep conditions).
486 /// Empty when the aggregate carries no internal ordering.
487 /// v7.39 (round 723) — FLAT (SoA): `order_by.len()` key values per
488 /// item, back to back. The per-item `Vec<Vec<Value>>` form allocated
489 /// one heap Vec PER ROW just to hold (usually) one integer — ~20 ms
490 /// of pure allocator traffic on the panel's 500k `string_agg(s, ','
491 /// ORDER BY id)`. The key width is the spec's `order_by.len()`,
492 /// which every consumer already has.
493 item_keys: Vec<Value<'static>>,
494 /// v7.17.0 — captured separator for string_agg: the last
495 /// non-NULL text seen. v7.39 (round 762, F31-C2) — this is the
496 /// CONSTANT-separator snapshot only (fused lane, group_concat
497 /// default, DISTINCT fallback); the per-row truth lives in
498 /// `item_seps` (the old note claimed "use the latest row's
499 /// value" was PG's behaviour — measured false, PG is per-row).
500 separator: Option<String>,
501 /// v7.17.0 — running boolean accumulator for bool_and /
502 /// bool_or / every. `None` until the first non-NULL input;
503 /// at finalize None → SQL NULL.
504 bool_acc: Option<bool>,
505 /// v7.32 (round-29) — sum of squares for the variance / stddev
506 /// family (`sum_float` carries the running sum; `count` the n).
507 sum_sq: f64,
508 /// v7.38 (read01) — exact accumulators for the stddev/variance family.
509 /// PG computes those aggregates in NUMERIC over exact inputs (its float8
510 /// overload only serves float inputs), so an f64 accumulator loses PG's
511 /// exact division scale — `var_pop(1,2,3)` is `0.66666666666666666667`,
512 /// not the 16-digit double. `stddev_saw_float` flips on the first
513 /// float/real input and drops the family back to the f64 accumulators,
514 /// whose result is then double precision, matching PG's float8 overload.
515 stddev_saw_float: bool,
516 stddev_sum: Option<spg_storage::bignum::BigNumeric>,
517 stddev_sum_sq: Option<spg_storage::bignum::BigNumeric>,
518 /// v7.39 (round 615) — the same exact Σx / Σx², accumulated in `i128`
519 /// while every input is an integer and neither sum has overflowed.
520 ///
521 /// The `BigNumeric` pair above is exact and is what the finaliser wants,
522 /// but reaching it cost NINE allocations a row on a plain INTEGER column
523 /// — a boxed value per input, its square, and a fresh box for each of
524 /// the two running totals — where `sum` and `avg` over the same column
525 /// cost none. `i128` holds the same integers exactly: an `int4` squares
526 /// to at most 4.6e18, so the running Σx² has room for 3.7e19 rows before
527 /// it can overflow, and a `bigint` input that does overflow falls back
528 /// below with nothing lost — the pair is folded into the BigNumeric
529 /// accumulator first, so the total is the one it would have had.
530 stddev_i_sum: i128,
531 stddev_i_sum_sq: i128,
532 stddev_i_spent: bool,
533 /// v7.32 (round-29) — running accumulator for bit_and / bit_or /
534 /// bit_xor. `None` until the first non-NULL input → SQL NULL.
535 bit_acc: Option<i64>,
536 /// v7.38 (read01, T4.4) — true once a BIGINT input is seen, so
537 /// bit_and/or/xor finalize as bigint vs integer (PG input-typed).
538 bit_wide: bool,
539 /// v7.39 (round 254/255) — EVERY row fed to a WITHIN GROUP
540 /// aggregate, NULLs included. `items` (and `count`) hold only the
541 /// non-NULL values, which is right for `percentile_*` / `mode` —
542 /// but PG's hypothetical-set fractions divide by the full input
543 /// size: with one extra NULL row, `percent_rank(3)` moves from 2/6
544 /// to 2/7 (probed live). rank / dense_rank are unaffected either
545 /// way, since they only count values sorting before the
546 /// hypothetical row.
547 within_group_rows: usize,
548 /// v7.32 (round-29) — two-argument regression family
549 /// (`covar_*` / `corr` / `regr_*`), PG arg order `f(Y, X)`. Only
550 /// rows where BOTH inputs are non-NULL contribute (`count` is the
551 /// paired n, independent of the single-arg `sum_*`).
552 reg_n: i64,
553 reg_sx: f64,
554 reg_sy: f64,
555 reg_sxx: f64,
556 reg_syy: f64,
557 reg_sxy: f64,
558 /// v7.32 (round-29) — second value stream for `json_object_agg`
559 /// (`items` holds the keys, `aux_items` the values).
560 aux_items: Vec<Value<'static>>,
561 /// v7.33 (array_agg argmax) — for a `first_ordered` spec
562 /// (`(array_agg(x ORDER BY y))[1]`), the running first-by-order
563 /// (sort-key tuple, value). Replaced only when a new row's key sorts
564 /// strictly before the current best (ties keep the earliest row, =
565 /// the stable-sort `[1]`). No items/item_keys array is built.
566 first_best: Option<(Vec<Value<'static>>, Value<'static>)>,
567}
568
569#[derive(Debug, Clone)]
570struct AggSpec {
571 name: String, // lowercased
572 /// First argument (value expression) for every aggregate
573 /// except `count(*)`. `None` for `count_star`.
574 arg: Option<Expr>,
575 /// v7.17.0 — second argument. Only `string_agg(value, sep)`
576 /// uses it today. `None` for every other aggregate (or for
577 /// `array_agg`, which is single-arg). Carried in the spec so
578 /// per-row evaluation can re-use the same separator
579 /// expression across calls.
580 arg2: Option<Expr>,
581 /// v7.25 (round-17) — `COUNT(DISTINCT x)` & friends: dedupe
582 /// the input stream per group before accumulation.
583 distinct: bool,
584 /// v7.24 (round-16 A) — aggregate-internal ORDER BY keys
585 /// (`array_agg(x ORDER BY y DESC NULLS LAST)`). Empty for the
586 /// plain form. Only the collection aggregates honour it;
587 /// other aggregates are order-insensitive and ignore it (PG
588 /// accepts the syntax everywhere too).
589 order_by: Vec<spg_sql::ast::OrderBy>,
590 /// v7.32 (round-29) — `FILTER (WHERE cond)`: a per-row predicate
591 /// evaluated against the source row before accumulation. A row
592 /// whose `cond` is not TRUE (false or NULL) is excluded from this
593 /// aggregate only. `None` for the unfiltered form.
594 filter: Option<Expr>,
595 /// v7.32 (round-29) — ordered-set aggregates only: the *direct*
596 /// argument (the percentile fraction for `percentile_cont/disc`).
597 /// PG requires it constant, so it is evaluated once. `None` for
598 /// `mode()` and for every non-ordered-set aggregate.
599 direct_arg: Option<Expr>,
600 /// v7.39 (read01 orderedsetaggs.c) — the remaining direct arguments
601 /// of a multi-key hypothetical-set call (`rank(5, 'x') WITHIN GROUP
602 /// (ORDER BY a, b)`); one per sort key past the first. Empty
603 /// everywhere else.
604 direct_args_extra: Vec<Expr>,
605 /// v7.33 (array_agg argmax) — set when this spec came from
606 /// `(array_agg(x ORDER BY y))[1]`: accumulate only the first-by-order
607 /// element (a running argmax/argmin) and finalise to that scalar
608 /// value, instead of collecting + sorting + materialising the whole
609 /// per-group array just to take element 1. Returns the element type,
610 /// not the array type.
611 first_ordered: bool,
612 /// v7.37.4 (R34) — derived from `name` at spec build time so the
613 /// per-row inner loop dispatches via a `match` on `Copy` enum
614 /// instead of a string compare for every (row × aggregate)
615 /// iteration.
616 kind: AggKind,
617 /// v7.39 (enum order knife) — member labels when the aggregate's
618 /// argument is enum-typed and the aggregate orders its input
619 /// (min/max): extreme comparisons use member order, not label text.
620 /// Enriched once per query in `run` (spec collection is AST-only and
621 /// has no catalog).
622 enum_labels: Option<Vec<String>>,
623 /// v7.39 (round 690) — the argument column's declared collation, for
624 /// `min`/`max`. Resolved beside `enum_labels` and for the same reason:
625 /// both are facts about the ARGUMENT that the comparison needs and
626 /// cannot look up for itself.
627 arg_collation: Option<alloc::string::String>,
628 /// v7.39 (enum order knife) — per-ORDER-BY-key member labels for the
629 /// ordered collection aggregates (`array_agg(x ORDER BY enum_col)`).
630 /// Parallel to `order_by`; all-None when no key is enum-typed.
631 order_enum_labels: Vec<Option<Vec<String>>>,
632 /// v7.38.18 — per-ORDER-BY-key declared collation, for the ordered
633 /// collection aggregates. Parallel to `order_by`, resolved the same
634 /// way and for the same reason as `order_enum_labels` beside it.
635 ///
636 /// `min`/`max` have read the argument's collation since round 690
637 /// (`arg_collation`), and so does the statement's own ORDER BY, but
638 /// the sort INSIDE an aggregate did not: on a column declared
639 /// `COLLATE "en_US.utf8"`, `SELECT x FROM t ORDER BY x` answered
640 /// `apple, client, DateStyle, Zebra` while `string_agg(x, ' ' ORDER
641 /// BY x)` over the same column answered `DateStyle Zebra apple
642 /// client`. Two orderings of one column in one query.
643 order_collations: Vec<Option<alloc::string::String>>,
644}
645
646/// Output of running the aggregate path. Schema describes one row per
647/// group; rows are not yet ORDER BY-sorted (caller does it).
648#[derive(Debug)]
649pub struct AggResult {
650 pub columns: Vec<ColumnSchema>,
651 pub rows: Vec<Row<'static>>,
652 /// v7.31 (perf — PG lesson #1, post-LIMIT subquery projection):
653 /// select-list items whose rewritten expr carries a subquery and
654 /// is referenced by neither ORDER BY nor HAVING. Their output
655 /// cells hold NULL placeholders; the caller truncates to
656 /// LIMIT+OFFSET first and only then evaluates these for the
657 /// surviving rows (PG runs the same shape with SubPlan loops=50
658 /// instead of loops=24000). `(output_col, rewritten_expr)`.
659 pub deferred: Vec<(usize, Expr)>,
660 /// Synthetic group rows aligned 1:1 with `rows`; populated only
661 /// when `deferred` is non-empty.
662 pub synth_rows: Vec<Row<'static>>,
663 /// Schema the deferred exprs evaluate against.
664 pub synth_schema: Vec<ColumnSchema>,
665}
666
667/// Execute aggregate logic against an already-WHERE-filtered iterator of
668/// rows. `table_alias` is the alias accepted by column resolution.
669#[allow(clippy::too_many_lines)]
670/// v7.25.2 (round-19 A) — caller-injected evaluator for synth-row
671/// expressions that still carry subquery nodes after the rewrite
672/// (correlated subqueries in the select list / HAVING / aggregate
673/// ORDER BY of a GROUP BY query). The engine passes its
674/// correlated-aware evaluator; pure-library callers pass None and
675/// surviving subqueries keep erroring loudly.
676pub type CorrelatedEval<'a> =
677 &'a dyn Fn(&Expr, &Row<'static>, &EvalContext<'_>) -> Result<Value<'static>, EvalError>;
678
679/// Output of the per-group projection stage (`project_groups`): the
680/// output schema, the projected rows, the synth rows kept alongside
681/// them for post-LIMIT deferred evaluation, the deferred subquery
682/// items, and the rewritten ORDER BY exprs (shared with the sort).
683struct Projection {
684 columns: Vec<ColumnSchema>,
685 out_rows: Vec<Row<'static>>,
686 kept_synth: Vec<Row<'static>>,
687 deferred: Vec<(usize, Expr)>,
688 order_rewritten: Vec<Expr>,
689 /// v7.37.x — when `defer_projection` is requested, `out_rows`
690 /// carries empty placeholders and the caller runs the per-item
691 /// eval pass after sort+truncate over the surviving ≤ keep_n
692 /// rows. `None` when projection was performed inline.
693 deferred_project: Option<DeferredProject>,
694}
695
696struct DeferredProject {
697 items_rewritten: Vec<Option<Expr>>,
698 items_compiled: Vec<Option<eval::CompiledExpr>>,
699}
700
701/// v7.35.0 — detect the `SELECT COUNT(*) FROM … [WHERE …]` shape
702/// (single item, no GROUP BY / HAVING / ORDER BY / DISTINCT /
703/// LIMIT WITH TIES / FILTER / window). For this shape the answer
704/// is exactly `rows.len()` as `BigInt`, no group state needed.
705/// Returns `None` for any deviation so the caller's full pipeline
706/// runs verbatim.
707///
708/// v7.35.2 — also short-circuit `COUNT(<literal>)` (e.g.
709/// `COUNT(1)`) and `COUNT(<column>)` when the column is declared
710/// NOT NULL on the input schema. PG handles both cases as
711/// `COUNT(*)` (the non-null filter is a no-op), so doing the same
712/// here keeps every `count this thing` shape on the same fast path
713/// instead of routing the literal / non-null-col variants through
714/// the four-stage aggregate pipeline.
715fn try_pure_count_star_short_circuit(
716 stmt: &SelectStatement,
717 rows: AggRows<'_>,
718 schema_cols: &[ColumnSchema],
719 table_alias: Option<&str>,
720) -> Option<AggResult> {
721 if stmt.distinct
722 || stmt.limit_with_ties
723 || stmt.group_by.is_some()
724 || stmt.having.is_some()
725 || !stmt.order_by.is_empty()
726 {
727 return None;
728 }
729 if stmt.items.len() != 1 {
730 return None;
731 }
732 let SelectItem::Expr { expr, alias } = &stmt.items[0] else {
733 return None;
734 };
735 let Expr::FunctionCall { name, args } = expr else {
736 return None;
737 };
738 if !name.eq_ignore_ascii_case("count") && !name.eq_ignore_ascii_case("count_star") {
739 return None;
740 }
741 let count_star_shape = match args.as_slice() {
742 // `COUNT(*)` parses to `count_star` with no args.
743 [] if name.eq_ignore_ascii_case("count_star") => true,
744 // `COUNT(<literal>)` — the per-row test is "is this literal
745 // non-null?" which is constant, so it's COUNT(*) when the
746 // literal is non-null.
747 [Expr::Literal(lit)] => !matches!(lit, spg_sql::ast::Literal::Null),
748 // `COUNT(<column>)` — same answer as COUNT(*) when the
749 // column is statically declared NOT NULL on the input
750 // schema. Resolve through the alias if one is set.
751 [Expr::Column(c)] => {
752 if let Some(q) = c.qualifier.as_deref()
753 && let Some(alias) = table_alias
754 && !q.eq_ignore_ascii_case(alias)
755 {
756 return None;
757 }
758 schema_cols
759 .iter()
760 .find(|s| s.name.eq_ignore_ascii_case(&c.name))
761 .is_some_and(|s| !s.nullable)
762 }
763 _ => return None,
764 };
765 if !count_star_shape {
766 return None;
767 }
768 let col_name = alias.clone().unwrap_or_else(|| "count".to_string());
769 let count = i64::try_from(rows.len()).unwrap_or(i64::MAX);
770 Some(AggResult {
771 columns: alloc::vec![ColumnSchema::new(col_name, DataType::BigInt, false)],
772 rows: alloc::vec![Row::new(alloc::vec![Value::BigInt(count)])],
773 deferred: Vec::new(),
774 synth_rows: Vec::new(),
775 synth_schema: Vec::new(),
776 })
777}
778
779/// v7.39 (round 528) — a GROUP BY name that names an output column.
780///
781/// `SELECT date_trunc('day', ts) AS d, count(*) FROM t GROUP BY d` is the
782/// canonical daily rollup, and it answered `column "d" does not exist`.
783/// Both PG and MySQL take a GROUP BY identifier that matches an output
784/// alias and group by the expression behind it; only grouping by a real
785/// column or an ordinal worked here.
786///
787/// Precedence is PG's, measured: an INPUT column of that name WINS.
788/// `SELECT v AS ts … GROUP BY ts` on a table that has a `ts` column
789/// groups by the column, which is why PG then rejects the ungrouped `v` —
790/// so the alias is consulted only when nothing else answers to the name.
791fn resolve_group_by_aliases(
792 keys: Vec<Expr>,
793 stmt: &SelectStatement,
794 schema_cols: &[ColumnSchema],
795) -> Result<Vec<Expr>, EvalError> {
796 let mut out = Vec::with_capacity(keys.len());
797 for key in keys {
798 let Expr::Column(c) = &key else {
799 out.push(key);
800 continue;
801 };
802 if c.qualifier.is_some()
803 || schema_cols
804 .iter()
805 .any(|sc| sc.name.eq_ignore_ascii_case(&c.name))
806 {
807 out.push(key);
808 continue;
809 }
810 let target = stmt.items.iter().find_map(|it| match it {
811 SelectItem::Expr {
812 expr,
813 alias: Some(a),
814 } if a.eq_ignore_ascii_case(&c.name) => Some(expr),
815 _ => None,
816 });
817 match target {
818 // PG's wording for the one alias that cannot be grouped by.
819 Some(e) if contains_aggregate(e) => {
820 return Err(EvalError::TypeMismatch {
821 detail: alloc::string::String::from(
822 "aggregate functions are not allowed in GROUP BY",
823 ),
824 });
825 }
826 Some(e) => out.push(e.clone()),
827 // Not an alias either — leave it, so the resolver reports the
828 // missing column as it always did.
829 None => out.push(key),
830 }
831 }
832 Ok(out)
833}
834
835pub(crate) fn run(
836 stmt: &SelectStatement,
837 rows: AggRows<'_>,
838 schema_cols: &[ColumnSchema],
839 table_alias: Option<&str>,
840 correlated_eval: Option<CorrelatedEval<'_>>,
841 // v7.39 (parallel-agg P1) — host-injected executor; None = the
842 // single-threaded paths, byte-identical to pre-P1.
843 runner: Option<&dyn crate::ParallelRunner>,
844 // v7.39 (enum order knife) — catalog for enum member-order metadata
845 // (spec collection is AST-only). None keeps every ordering textual.
846 catalog: Option<&spg_storage::Catalog>,
847 // v7.39 (read01 round 63) — and the engine, so a user function whose body
848 // has its own FROM can run inside an aggregate's argument
849 // (`string_agg(lookup(id), ',')`). The catalog alone is not enough: the body
850 // is a QUERY and has to go through the real executor.
851 engine: Option<&crate::Engine>,
852) -> Result<AggResult, EvalError> {
853 // v7.38 P0 元机制 A — fires at the top of the aggregate
854 // executor with the number of input rows. Tests use this to
855 // block before a hypothetical spill decision; in release it
856 // expands to `let _ = (...);`.
857 let __spg_row_count = rows.len();
858 crate::injection_point!("aggregate_spill_trigger", &__spg_row_count);
859 // v7.35.0 — pure `SELECT COUNT(*) FROM … WHERE …` short-circuit.
860 // The caller already filtered rows by WHERE (we run on the
861 // post-WHERE survivor set), so for the canonical pure-COUNT(*)
862 // shape (no GROUP BY / HAVING / ORDER BY / DISTINCT / FILTER /
863 // window) the answer is simply `rows.len()`. The four-stage
864 // aggregate pipeline below (accumulate_groups → build_synth_schema
865 // → finalize_synth_rows → project_groups) collapses to a single
866 // BigInt cell when there's a single group, but each stage still
867 // pays its own allocation tax — group state map, synth schema
868 // vec, finalize loop. `exists_in_60` (mailrs prod #4 baseline)
869 // is exactly this shape on a 25 k-row JOIN.
870 if let Some(short) = try_pure_count_star_short_circuit(stmt, rows, schema_cols, table_alias) {
871 return Ok(short);
872 }
873 let group_exprs: Vec<Expr> = stmt.group_by.clone().unwrap_or_default();
874 // v7.39 (round 528) — a GROUP BY name that is only an output ALIAS.
875 let group_exprs = resolve_group_by_aliases(group_exprs, stmt, schema_cols)?;
876
877 // v7.39 (round 620) — PG's strict rule, checked BEFORE the pipeline so the
878 // diagnosis names what is actually wrong. Skipped under the MySQL dialect,
879 // which licenses exactly what this rejects (the loose rewrite below), and
880 // skipped when the grouping is by a primary key, which licenses every other
881 // column of that table.
882 // A GROUP BY name that resolves to nothing is reported as the missing
883 // column it is, ahead of this rule — measured against PG, which answers
884 // `column "nosuch" does not exist` for `SELECT v FROM t GROUP BY nosuch`
885 // rather than complaining that `v` is ungrouped.
886 let group_keys_all_resolve = group_exprs.iter().all(|g| match g {
887 Expr::Column(c) => {
888 c.qualifier.is_some()
889 || schema_cols
890 .iter()
891 .any(|sc| sc.name.eq_ignore_ascii_case(&c.name))
892 }
893 _ => true,
894 });
895 let licensed = qualifiers_grouped_by_primary_key(stmt, &group_exprs, schema_cols, catalog);
896 let fd_on_primary_key = !licensed.is_empty();
897 if group_keys_all_resolve && !engine.is_some_and(|e| e.backslash_escapes) {
898 let offender = stmt
899 .items
900 .iter()
901 .find_map(|it| match it {
902 SelectItem::Expr { expr, .. } => {
903 first_ungrouped_column(expr, &group_exprs, schema_cols, &licensed)
904 }
905 _ => None,
906 })
907 .or_else(|| {
908 stmt.order_by.iter().find_map(|o| {
909 first_ungrouped_column(&o.expr, &group_exprs, schema_cols, &licensed)
910 })
911 })
912 .or_else(|| {
913 stmt.having
914 .as_ref()
915 .and_then(|h| first_ungrouped_column(h, &group_exprs, schema_cols, &licensed))
916 });
917 if let Some(c) = offender {
918 // PG qualifies the column with the alias when there is one, and
919 // with the table name otherwise.
920 let qual = c
921 .qualifier
922 .as_deref()
923 .or(table_alias)
924 .or_else(|| stmt.from.as_ref().map(|f| f.primary.name.as_str()))
925 .unwrap_or("");
926 return Err(EvalError::TypeMismatch {
927 detail: alloc::format!(
928 "column \"{qual}.{}\" must appear in the GROUP BY clause or be used in an aggregate function",
929 c.name
930 ),
931 });
932 }
933 }
934
935 // v7.39 (round 405) — MySQL's loose GROUP BY: wrap each non-grouped,
936 // non-aggregated column in `any_value(col)` so the rest of the pipeline
937 // treats it as an aggregate (first-seen value per group). Only under the
938 // dialect and only when there is an explicit GROUP BY; PG keeps the
939 // strict "must appear in GROUP BY / be aggregated" rule.
940 //
941 // v7.39 (round 620) — the same rewrite serves PG's functional dependency.
942 // Letting the ungrouped column PAST the check above is not enough: the
943 // grouped row carries only the keys and the aggregates, so `s` still has
944 // nowhere to be read from and the query failed on `column "s" does not
945 // exist`. Grouping by a primary key means one input row per group, so
946 // "any value in the group" IS the value — the identical rewrite, reached
947 // for a different and much narrower reason.
948 let mysql_loose = engine.is_some_and(|e| e.backslash_escapes);
949 let loose_stmt;
950 let stmt = if (mysql_loose || fd_on_primary_key) && !group_exprs.is_empty() {
951 // The dialect claims every ungrouped column; the functional dependency
952 // claims only what a grouped primary key determines.
953 let claim: Option<&[alloc::string::String]> =
954 if mysql_loose { None } else { Some(&licensed) };
955 let mut s = stmt.clone();
956 for item in &mut s.items {
957 if let SelectItem::Expr { expr, .. } = item {
958 let taken = core::mem::replace(expr, Expr::Literal(spg_sql::ast::Literal::Null));
959 *expr = wrap_loose_group_columns(taken, &group_exprs, schema_cols, claim);
960 }
961 }
962 for o in &mut s.order_by {
963 let taken = core::mem::replace(&mut o.expr, Expr::Literal(spg_sql::ast::Literal::Null));
964 o.expr = wrap_loose_group_columns(taken, &group_exprs, schema_cols, claim);
965 }
966 if let Some(h) = s.having.take() {
967 s.having = Some(wrap_loose_group_columns(
968 h,
969 &group_exprs,
970 schema_cols,
971 claim,
972 ));
973 }
974 loose_stmt = s;
975 &loose_stmt
976 } else {
977 stmt
978 };
979
980 // Collect aggregate sub-expressions across items + order_by.
981 let mut agg_specs: Vec<AggSpec> = Vec::new();
982 for item in &stmt.items {
983 if let SelectItem::Expr { expr, .. } = item {
984 collect_aggregates(expr, &mut agg_specs);
985 }
986 }
987 for o in &stmt.order_by {
988 collect_aggregates(&o.expr, &mut agg_specs);
989 }
990 if let Some(h) = &stmt.having {
991 collect_aggregates(h, &mut agg_specs);
992 }
993 // v7.17.0 — arity validation. The collector tolerates an
994 // arbitrary positional-arg count; here we enforce the
995 // per-aggregate contract so a malformed call (e.g.
996 // `array_agg()` or `string_agg(x)`) surfaces as a SQL error
997 // rather than silently coercing to a degenerate aggregate.
998 validate_agg_arities(stmt, &agg_specs)?;
999 validate_within_group(&agg_specs, schema_cols, stmt.group_by.as_deref())?;
1000
1001 // v7.38.18 (S2) — the database's collation, for the columns that
1002 // declare none. `None` when it is byte order, which is every
1003 // database written before this existed.
1004 let db_collation: Option<&str> = catalog
1005 .map(spg_storage::Catalog::db_collation)
1006 .filter(|d| !crate::collate::is_byte_wise(d));
1007 // v7.39 (round 690) — resolve the argument's declared collation for
1008 // `min`/`max`. This rides beside `enum_labels` in `AggSpec` but NOT
1009 // inside its resolver loop: that loop only runs when the catalog holds
1010 // at least one enum type, and a collation has nothing to do with enums.
1011 for spec in &mut agg_specs {
1012 if matches!(spec.kind, AggKind::Min | AggKind::Max)
1013 && let Some(Expr::Column(c)) = &spec.arg
1014 {
1015 // A bare column argument carries its collation; an expression
1016 // produces a new value and has none (derivation is unbuilt).
1017 spec.arg_collation = schema_cols
1018 .iter()
1019 .find(|sc| sc.name.eq_ignore_ascii_case(&c.name))
1020 .and_then(|sc| sc.collation_name.clone())
1021 // v7.38.18 (S2) — the database's when the column
1022 // declares none. `C` filters out below, so nothing moves
1023 // for a database that has not asked for a locale.
1024 .or_else(|| db_collation.map(alloc::string::String::from))
1025 .filter(|n| crate::collate::is_supported(n));
1026 }
1027 }
1028
1029 // v7.38.18 — the same fact for each ORDER BY key of an ordered
1030 // collection aggregate. Outside the enum resolver below for the
1031 // reason the loop above is: that one only runs when the catalog
1032 // holds an enum type, and a collation has nothing to do with enums.
1033 for spec in &mut agg_specs {
1034 if spec.order_by.is_empty() {
1035 continue;
1036 }
1037 spec.order_collations = spec
1038 .order_by
1039 .iter()
1040 .map(|o| {
1041 // A bare column key carries its collation; an expression
1042 // produces a new value and has none, the same limit
1043 // `min`/`max` has over an expression argument.
1044 let Expr::Column(c) = &o.expr else {
1045 return None;
1046 };
1047 schema_cols
1048 .iter()
1049 .find(|sc| sc.name.eq_ignore_ascii_case(&c.name))
1050 .and_then(|sc| sc.collation_name.clone())
1051 .or_else(|| db_collation.map(alloc::string::String::from))
1052 .filter(|n| crate::collate::is_supported(n))
1053 })
1054 .collect();
1055 }
1056
1057 // v7.39 (enum order knife) — resolve enum member-order metadata once
1058 // per query: min/max extremes and ordered-collection sort keys over
1059 // enum-typed expressions compare by member order (PG enumsortorder).
1060 if let Some(cat) = catalog
1061 && !cat.enum_types().is_empty()
1062 {
1063 for spec in &mut agg_specs {
1064 // v7.39 (round 258) — min/max have always needed the argument's
1065 // enum labels; a DISTINCT aggregate now does too, because its
1066 // dedup sort must follow MEMBER order (round 257 added the sort
1067 // and, deriving labels only here, sorted enum columns by text).
1068 if (matches!(spec.kind, AggKind::Min | AggKind::Max) || spec.distinct)
1069 && let Some(arg) = &spec.arg
1070 {
1071 spec.enum_labels = crate::eval::expr_enum_labels(arg, schema_cols, catalog)
1072 .map(<[String]>::to_vec);
1073 }
1074 if !spec.order_by.is_empty() {
1075 spec.order_enum_labels = spec
1076 .order_by
1077 .iter()
1078 .map(|o| {
1079 crate::eval::expr_enum_labels(&o.expr, schema_cols, catalog)
1080 .map(<[String]>::to_vec)
1081 })
1082 .collect();
1083 }
1084 }
1085 }
1086
1087 // (1) Stream the WHERE-filtered rows into insertion-ordered group state.
1088 let order = accumulate_groups(
1089 rows,
1090 &group_exprs,
1091 &agg_specs,
1092 schema_cols,
1093 table_alias,
1094 correlated_eval,
1095 runner,
1096 catalog,
1097 engine,
1098 )?;
1099
1100 // (2) Build the synthetic per-group schema and finalise each group's row.
1101 let synth_schema = build_synth_schema(
1102 rows,
1103 &group_exprs,
1104 &agg_specs,
1105 schema_cols,
1106 table_alias,
1107 catalog,
1108 engine,
1109 )?;
1110 let synth_rows = finalize_synth_rows(
1111 &order,
1112 &agg_specs,
1113 &synth_schema,
1114 rows,
1115 schema_cols,
1116 table_alias,
1117 catalog,
1118 engine,
1119 runner,
1120 )?;
1121
1122 // v7.37.x (mailrs Track A 100k attack) — defer the bound
1123 // per-item SELECT projection on the synth rows until AFTER
1124 // sort + LIMIT truncation. On a `GROUP BY t ORDER BY agg DESC
1125 // LIMIT 50` with 20 000 groups (the mailrs minimal 100k shape)
1126 // pre-defer ran 20 000 × N_items compiled-VM evals + Row
1127 // allocations before discarding 99.75 % at the sort truncation
1128 // step. HAVING still runs inline on every group because it
1129 // filters BEFORE the LIMIT; we only skip the SELECT-list eval.
1130 //
1131 // v7.37 (round 998) — and so a HAVING no longer stands the deferral
1132 // down. It used to, which cost the mailrs Track A query 11.9 ms of
1133 // 83. Neither clause is expensive alone: HAVING costs 5.0 ms without
1134 // an ORDER BY and 16.9 with one, and an ORDER BY costs MINUS 8.6 ms
1135 // without a HAVING, because ORDER BY + LIMIT is what switches this
1136 // deferral on. The residue of 11.9 ms belonged to neither and
1137 // appeared only together.
1138 //
1139 // What named it: the interaction tracks what the aggregates COST
1140 // rather than how many there are — one expensive aggregate
1141 // reproduces it as fully as twelve cheap ones — and it does not move
1142 // when the LIMIT changes. Both follow from projecting all 20 000
1143 // groups instead of the 50 that survive truncation.
1144 //
1145 // Safe because the clause above runs first: HAVING filters into
1146 // `kept_synth` BEFORE this branch, the sort truncates that survivor
1147 // list, and the completion projects from it. HAVING is rewritten
1148 // against the synthetic group schema, so it never reads a projected
1149 // item.
1150 //
1151 // v7.37 (round 997) — a set-returning item must NOT defer. The
1152 // deferred completion at the end of this function evaluates each item
1153 // scalarly; the expansion that turns one group into one row per
1154 // element lives in the branch the deferral skips. So a deferred
1155 // `unnest(...)` in the select list came back as
1156 // `function unnest(integer[]) does not exist` — the exact error round
1157 // 621 had fixed, reintroduced for the shapes that qualify to defer.
1158 // Differential against PG18.4: the same query answered correctly
1159 // without LIMIT, with LIMIT >= the group count, and — at the time —
1160 // with a HAVING, those being the cases where the deferral was off.
1161 // Round 998 removed the HAVING one from that list, which is why this
1162 // guard carries the SRF rule on its own now.
1163 let any_srf_item = stmt.items.iter().any(|i| match i {
1164 SelectItem::Expr { expr, .. } => crate::select::top_level_srf_kind(expr).is_some(),
1165 _ => false,
1166 });
1167 let defer_projection = !stmt.order_by.is_empty()
1168 && !stmt.distinct
1169 && !stmt.limit_with_ties
1170 && !any_srf_item
1171 && stmt.limit_literal().is_some_and(|l| {
1172 let off = stmt.offset_literal().unwrap_or(0) as usize;
1173 let k = (l as usize).saturating_add(off);
1174 k > 0 && k < synth_rows.len()
1175 });
1176
1177 // (3) Rewrite the user's expressions, filter groups by HAVING and project.
1178 let Projection {
1179 columns,
1180 mut out_rows,
1181 mut kept_synth,
1182 deferred,
1183 order_rewritten,
1184 deferred_project,
1185 } = project_groups(
1186 synth_rows,
1187 stmt,
1188 &group_exprs,
1189 &agg_specs,
1190 &synth_schema,
1191 correlated_eval,
1192 defer_projection,
1193 catalog,
1194 engine.is_some_and(|e| e.backslash_escapes),
1195 )?;
1196
1197 // (4) ORDER BY on the aggregated output (the caller applies LIMIT).
1198 //
1199 // v7.37.3 (mailrs prod /api/contacts 3.21× regression — and the
1200 // general inbox-listing-shape SPG-vs-PG gap) — top-K sink for
1201 // `ORDER BY <agg> [DESC] LIMIT k`. Pre-7.37.3 this stage ran a
1202 // full O(N log N) sort over every surviving group, then the
1203 // caller truncated to `k`. With high-cardinality GROUP BY (a
1204 // sender column with hundreds-thousands of distinct values) the
1205 // truncated set is a tiny fraction of `N` — keep an O(k) top-K
1206 // sink and never sort the discarded majority. Matches PG /
1207 // MySQL / MariaDB's standard "LIMIT k under ORDER BY agg"
1208 // optimisation; SPG previously implemented it only on the
1209 // streamed inner-join path (`try_streamed_inner_join_topn`)
1210 // and not on the aggregate output.
1211 //
1212 // Gate: needs a literal LIMIT (placeholder LIMIT we can't bound
1213 // statically here), no DISTINCT (would need post-dedup, can't
1214 // truncate during sort), no LIMIT WITH TIES (which extends past
1215 // the literal k by run-time tie-key comparison).
1216 let keep_n: Option<usize> =
1217 if !stmt.order_by.is_empty() && !stmt.distinct && !stmt.limit_with_ties {
1218 stmt.limit_literal().map(|l| {
1219 let off = stmt.offset_literal().unwrap_or(0) as usize;
1220 (l as usize).saturating_add(off)
1221 })
1222 } else {
1223 None
1224 };
1225 if !stmt.order_by.is_empty() {
1226 let (sorted_synth, sorted_out) = sort_synth_by_order_by(
1227 &synth_schema,
1228 &columns,
1229 &stmt.order_by,
1230 &order_rewritten,
1231 kept_synth,
1232 out_rows,
1233 correlated_eval,
1234 keep_n,
1235 catalog,
1236 engine.is_some_and(|e| e.backslash_escapes),
1237 )?;
1238 kept_synth = sorted_synth;
1239 out_rows = sorted_out;
1240 }
1241
1242 // v7.37.x — run deferred SELECT-list projection on the truncated
1243 // top-K survivors. For `GROUP BY thread_id ORDER BY MAX(date) DESC
1244 // LIMIT 50` against 20 000 groups, this turns ~40 000 compiled-VM
1245 // evals + Row allocations into 100, saving ~2-3 ms on the mailrs
1246 // minimal 100k shape.
1247 if let Some(DeferredProject {
1248 items_rewritten,
1249 items_compiled,
1250 }) = deferred_project
1251 {
1252 let mut synth_ctx = EvalContext::new(&synth_schema, None);
1253 if let Some(cat) = catalog {
1254 synth_ctx = synth_ctx.with_catalog(cat);
1255 }
1256 let mut stack: Vec<Value<'static>> = Vec::new();
1257 for (idx, srow) in kept_synth.iter().enumerate() {
1258 let mut values: Vec<Value<'static>> = Vec::with_capacity(columns.len());
1259 for (i, rewritten) in items_rewritten.iter().enumerate() {
1260 let Some(rewritten) = rewritten else { continue };
1261 if deferred.iter().any(|(c, _)| *c == i) {
1262 values.push(Value::Null);
1263 continue;
1264 }
1265 values.push(if let Some(cc) = &items_compiled[i] {
1266 eval::eval_compiled(cc, srow, &synth_ctx, &mut stack)?
1267 } else {
1268 match correlated_eval {
1269 Some(f) if crate::expr_has_subquery(rewritten) => {
1270 f(rewritten, srow, &synth_ctx)?
1271 }
1272 _ => eval::eval_expr(rewritten, srow, &synth_ctx)?,
1273 }
1274 });
1275 }
1276 out_rows[idx] = Row::new(values);
1277 }
1278 }
1279
1280 // v7.37 (round 999) — SELECT DISTINCT over a GROUP BY query.
1281 //
1282 // Every other path deduplicates: the scan paths, the window path and
1283 // the set operations all call `dedup_rows`. This one never did, so
1284 // `SELECT DISTINCT count(*) FROM t GROUP BY g` returned one row per
1285 // GROUP — 200 where PG18.4 returns 1, all of them the same value.
1286 // Not an error, not a missing column: 199 extra rows, silently.
1287 //
1288 // The gate on the top-K sink above says it in as many words — "no
1289 // DISTINCT (would need post-dedup, can't truncate during sort)" — so
1290 // the sink correctly declines to truncate, and the post-dedup it
1291 // names was never written. This is it.
1292 //
1293 // After the ORDER BY, like the window path: duplicate rows carry
1294 // identical sort keys, so removing them cannot disturb the order.
1295 // Before the LIMIT, which the caller applies, because PG deduplicates
1296 // and then counts.
1297 //
1298 // Only `out_rows` needs it: `deferred` is empty whenever DISTINCT is
1299 // set (`defer_enabled` requires `!stmt.distinct`), so nothing indexes
1300 // into `kept_synth` alongside these rows.
1301 if stmt.distinct {
1302 // v7.38.14 — masked, not dialect-only. `SELECT DISTINCT` over a
1303 // GROUP BY result folded every text position regardless of what
1304 // the column declared, which is the defect 3b494b6e closed on the
1305 // main scan path. The output schema is in scope here and carries
1306 // the collation, so the mask needs no new plumbing.
1307 out_rows = crate::select::dedup_rows(
1308 out_rows,
1309 crate::select::FoldSpec::of_masks(
1310 engine.is_some_and(|e| e.backslash_escapes),
1311 &crate::select::fold_mask_of_columns(&columns),
1312 &crate::select::pad_mask_of_columns(&columns),
1313 ),
1314 );
1315 }
1316
1317 let (synth_rows_out, synth_schema_out) = if deferred.is_empty() {
1318 (Vec::new(), Vec::new())
1319 } else {
1320 (kept_synth, synth_schema.clone())
1321 };
1322 Ok(AggResult {
1323 columns,
1324 rows: out_rows,
1325 deferred,
1326 synth_rows: synth_rows_out,
1327 synth_schema: synth_schema_out,
1328 })
1329}
1330
1331/// v7.32 (round-29) — validate the structural requirements of WITHIN
1332/// GROUP (ordered-set / hypothetical-set) aggregates up front, so a
1333/// malformed call surfaces as a SQL error rather than a silently
1334/// degenerate aggregate.
1335/// v7.39 (round 255) — PG's name for an expression's type in an
1336/// ordered-set signature error. Only a CAST / COLUMN is trusted (the
1337/// round-237 lesson: `describe_expr` reports a binary operator as its
1338/// left operand's type); an untyped literal is PG's own `unknown`, and
1339/// anything else falls back to `unknown` rather than guessing.
1340fn ordered_set_arg_type_name(e: &Expr, columns: &[ColumnSchema]) -> String {
1341 if matches!(
1342 e,
1343 Expr::Literal(spg_sql::ast::Literal::String(_))
1344 | Expr::Literal(spg_sql::ast::Literal::Null)
1345 ) {
1346 return String::from("unknown");
1347 }
1348 match e {
1349 Expr::Cast { .. } | Expr::Column(_) | Expr::Literal(_) => {
1350 crate::describe::describe_expr(e, columns).map_or_else(
1351 || String::from("unknown"),
1352 |s| crate::conversions::pg_type_name_for_error(s.ty),
1353 )
1354 }
1355 _ => String::from("unknown"),
1356 }
1357}
1358
1359/// v7.39 (round 255) — PG resolves an ordered-set / hypothetical-set
1360/// call as ONE function whose signature is `(direct args…, WITHIN GROUP
1361/// args…)`; anything that does not match a declared overload is a plain
1362/// `function f(…) does not exist` (42883), not a bespoke message. Probed
1363/// live: `percentile_cont(numeric, text)`, `rank(integer, integer,
1364/// text)`, `mode(integer, integer)`.
1365fn ordered_set_signature_error(name: &str, spec: &AggSpec, columns: &[ColumnSchema]) -> EvalError {
1366 let mut parts: Vec<String> = Vec::new();
1367 if let Some(d) = &spec.direct_arg {
1368 parts.push(ordered_set_arg_type_name(d, columns));
1369 }
1370 for d in &spec.direct_args_extra {
1371 parts.push(ordered_set_arg_type_name(d, columns));
1372 }
1373 for o in &spec.order_by {
1374 parts.push(ordered_set_arg_type_name(&o.expr, columns));
1375 }
1376 EvalError::TypeMismatch {
1377 detail: format!("function {name}({}) does not exist", parts.join(", ")),
1378 }
1379}
1380
1381fn validate_within_group(
1382 agg_specs: &[AggSpec],
1383 columns: &[ColumnSchema],
1384 group_by: Option<&[Expr]>,
1385) -> Result<(), EvalError> {
1386 // v7.39 (round 765, F31-D2) — PG requires an ordered-set
1387 // aggregate's DIRECT arguments to use only grouped columns
1388 // (`percentile_cont(x) WITHIN GROUP (ORDER BY x)` refuses with
1389 // "column … must appear in the GROUP BY clause", DETAIL "Direct
1390 // arguments of an ordered-set aggregate must use only grouped
1391 // columns", PG18-measured); SPG evaluated the first row's value
1392 // and answered.
1393 fn first_ungrouped(e: &Expr, group_by: Option<&[Expr]>) -> Option<String> {
1394 let mut found: Option<String> = None;
1395 let mut subs: Vec<&SelectStatement> = Vec::new();
1396 crate::visit_expr_columns_and_subqueries(
1397 e,
1398 &mut |c| {
1399 if found.is_some() {
1400 return;
1401 }
1402 let grouped = group_by.is_some_and(|gs| {
1403 gs.iter().any(|g| match g {
1404 Expr::Column(gc) => gc.name.eq_ignore_ascii_case(&c.name),
1405 _ => false,
1406 })
1407 });
1408 // The visitor's exotic-node BAIL marker is an empty
1409 // name — not a real column; skip it (refusing on it
1410 // would reject constant shapes like ARRAY[…] casts).
1411 if !grouped && !c.name.is_empty() {
1412 found = Some(match &c.qualifier {
1413 Some(q) => format!("{q}.{}", c.name),
1414 None => c.name.clone(),
1415 });
1416 }
1417 },
1418 &mut |s| subs.push(s),
1419 );
1420 found
1421 }
1422 for spec in agg_specs {
1423 if !is_within_group_name(&spec.name) {
1424 continue;
1425 }
1426 for d in spec.direct_arg.iter().chain(spec.direct_args_extra.iter()) {
1427 if let Some(col) = first_ungrouped(d, group_by) {
1428 return Err(EvalError::TypeMismatch {
1429 detail: format!(
1430 "column \"{col}\" must appear in the GROUP BY clause or be used in an aggregate function"
1431 ),
1432 });
1433 }
1434 }
1435 }
1436 // v7.32 (round-29) — WITHIN GROUP aggregates require the clause (PG
1437 // raises a hard error otherwise rather than silently degrading), and
1438 // SPG supports the single-sort-key form only.
1439 for spec in agg_specs {
1440 if is_within_group_name(&spec.name) {
1441 if spec.order_by.is_empty() {
1442 // v7.39 (round 704) — the hypothetical-set names double as
1443 // WINDOW functions, and PG resolves the bare zero-argument
1444 // spelling to the window reading: `SELECT rank() FROM t` is
1445 // `window function rank requires an OVER clause` there, not
1446 // a WITHIN GROUP complaint. With a direct argument the
1447 // ordered-set reading is the one the caller meant, and the
1448 // WITHIN GROUP wording stands.
1449 if spec.direct_arg.is_none() && is_hypothetical_set_name(&spec.name) {
1450 return Err(EvalError::TypeMismatch {
1451 detail: format!("window function {} requires an OVER clause", spec.name),
1452 });
1453 }
1454 return Err(EvalError::TypeMismatch {
1455 detail: format!("{}() requires WITHIN GROUP (ORDER BY …)", spec.name),
1456 });
1457 }
1458 // mode() is the only WITHIN GROUP aggregate with no direct
1459 // argument; the rest carry one (percentile fraction /
1460 // hypothetical value).
1461 if spec.name != "mode" && spec.direct_arg.is_none() {
1462 return Err(EvalError::TypeMismatch {
1463 detail: format!("{}() requires a direct argument", spec.name),
1464 });
1465 }
1466 // …and mode() takes NONE: `mode(1)` used to be accepted with
1467 // the argument silently dropped.
1468 if spec.name == "mode" && spec.direct_arg.is_some() {
1469 return Err(ordered_set_signature_error(&spec.name, spec, columns));
1470 }
1471 // v7.39 (read01 orderedsetaggs.c) — the hypothetical-set
1472 // family supports the multi-key form: one direct argument
1473 // per sort key (PG resolves a mismatch as a missing
1474 // function overload; its HINT carries the real rule).
1475 let hypothetical = matches!(
1476 spec.name.as_str(),
1477 "rank" | "dense_rank" | "percent_rank" | "cume_dist"
1478 );
1479 // Only the hypothetical-set family takes a multi-key sort
1480 // spec, and then it needs exactly one direct argument per
1481 // key. PG reports every mismatch as a missing overload.
1482 if hypothetical {
1483 if 1 + spec.direct_args_extra.len() != spec.order_by.len() {
1484 return Err(ordered_set_signature_error(&spec.name, spec, columns));
1485 }
1486 } else if spec.order_by.len() > 1 || !spec.direct_args_extra.is_empty() {
1487 // `percentile_cont(0.5, 0.6)` and `mode(1)` used to be
1488 // silently accepted (the extra arguments were dropped and
1489 // the aggregate answered anyway).
1490 return Err(ordered_set_signature_error(&spec.name, spec, columns));
1491 }
1492 // v7.39 (round 255) — `percentile_cont` interpolates, so PG
1493 // declares it only over the numeric tower and interval
1494 // (probed: text / date / timestamp / bool are refused, while
1495 // `percentile_disc` and `mode` take any sortable type). SPG
1496 // answered NULL for the refused types. Judged from the
1497 // STATICALLY known type only — an unknown one is let through
1498 // (round 237: refusing a legal query is worse than missing an
1499 // illegal one).
1500 if spec.name == "percentile_cont"
1501 && let Some(o) = spec.order_by.first()
1502 && matches!(o.expr, Expr::Cast { .. } | Expr::Column(_))
1503 && let Some(sch) = crate::describe::describe_expr(&o.expr, columns)
1504 && !matches!(
1505 sch.ty,
1506 spg_storage::DataType::SmallInt
1507 | spg_storage::DataType::Int
1508 | spg_storage::DataType::BigInt
1509 | spg_storage::DataType::Float
1510 | spg_storage::DataType::Real
1511 | spg_storage::DataType::Numeric { .. }
1512 | spg_storage::DataType::Interval
1513 )
1514 {
1515 return Err(ordered_set_signature_error(&spec.name, spec, columns));
1516 }
1517 }
1518 }
1519 Ok(())
1520}
1521
1522/// (1) Stream the WHERE-filtered rows, group by the GROUP BY value
1523/// tuple, and update per-group aggregate state. Returns the groups in
1524/// insertion order. See `run` for the bind-once fast path rationale.
1525/// v7.39 (round 665) — the running numeric state a sum/avg keeps, in ONE
1526/// place.
1527///
1528/// It used to live in four independently written copies: `FusedAcc`'s own
1529/// fields, `AggState`'s own fields, and twice more as loose locals inside
1530/// `accumulate_groups`. `FusedAcc`'s doc comment described that openly —
1531/// "field-for-field the same running state the single-spec sum/avg fast
1532/// path keeps in locals" — so the duplication was deliberate manual
1533/// inlining, not drift.
1534///
1535/// The cost was not abstract. Round 664 measured it: adding one guard to
1536/// the sum/avg family meant editing FOUR sites, and three of the four were
1537/// found only by running a different SQL shape and watching the wrong
1538/// answer come back. Reading the code did not reveal them, because the
1539/// three parallel loops in the fused block are not symmetric — the middle
1540/// one is a `length()` shortcut that accumulates nothing numeric.
1541///
1542/// `count` deliberately stays outside: `count(*)` keeps it too, and it is
1543/// not part of the numeric running state.
1544#[derive(Debug, Default, Clone)]
1545struct NumAcc {
1546 sum_int: i64,
1547 sum_float: f64,
1548 use_float: bool,
1549 float_not_real: bool,
1550 sum_num_scaled: i128,
1551 sum_num_kind: spg_storage::NumericKind,
1552 sum_num_scale: u16,
1553 /// v7.39 (read01 numeric.c) — bignum spill; see `SumBig`.
1554 sum_big: SumBig,
1555 use_numeric: bool,
1556 sum_iv_months: i64,
1557 sum_iv_days: i64,
1558 sum_iv_micros: i128,
1559 use_interval: bool,
1560 sum_money: i128,
1561 use_money: bool,
1562 /// Inside the struct, not beside it. Measured: splitting it out gave
1563 /// `acc_cell` two base pointers where the copy it replaced had one,
1564 /// and `sum(int)` over 500k rows lost ~8% (paired, n=12, p=0.04).
1565 /// `count(*)` reading `st.num.count` is a small price for that.
1566 count: i64,
1567}
1568
1569#[allow(clippy::too_many_lines, clippy::type_complexity)]
1570/// v7.37.16 — per-spec accumulator for the fused multi-spec fast path.
1571/// Field-for-field the same running state the single-spec sum/avg fast
1572/// path keeps in locals; finalized into `AggState` identically.
1573#[derive(Default, Clone)]
1574struct FusedAcc {
1575 /// The shared sum/avg running state (see `NumAcc`).
1576 num: NumAcc,
1577 /// v7.39 (round 568/569) — the min/max lane. `min` and `max` were
1578 /// the only ordinary aggregates the fused layout did not accept, so
1579 /// they fell to the generic per-spec machinery and cost DOUBLE a
1580 /// `sum` over the same scan (500k INTs: sum 13.4 ms, min 26.5,
1581 /// max 27.6, while PG18 is flat at 8.2 for all three). They also
1582 /// missed the shard-parallel scan the fused path runs.
1583 extreme: Option<Value<'static>>,
1584 /// Which way this accumulator's comparison goes, so a shard merge
1585 /// does not need to be told.
1586 extreme_max: bool,
1587 extreme_mysql: bool,
1588 /// v7.39 (round 690) — the argument's declared collation, so a
1589 /// shard merge compares the two extremes the same way the scan did.
1590 extreme_coll: Option<alloc::string::String>,
1591 /// v7.39 (round 724) — the collection lanes: string_agg / array_agg
1592 /// items in ROW order (shard merge concatenates in shard order,
1593 /// which IS row order), plus the flat ORDER BY keys (round 723's
1594 /// layout). The finalize sort/join is the existing AggState path.
1595 items: Vec<Value<'static>>,
1596 item_keys: Vec<Value<'static>>,
1597}
1598
1599/// v7.39 (round 569) — a fresh accumulator per op, carrying each one's
1600/// comparison direction so `merge_fused` stays a two-argument fold.
1601fn fused_accs(ops: &[FusedOp], mysql: bool) -> Vec<FusedAcc> {
1602 ops.iter()
1603 .map(|op| {
1604 let mut a = FusedAcc::default();
1605 if let FusedOp::Extreme { max, coll, .. } | FusedOp::ExtremeExpr { max, coll, .. } = op
1606 {
1607 a.extreme_max = *max;
1608 a.extreme_mysql = mysql;
1609 a.extreme_coll = coll.clone();
1610 }
1611 a
1612 })
1613 .collect()
1614}
1615
1616/// v7.39 (parallel-agg P3) — the fused-op layout shared by the
1617/// single-group fast path and the parallel GROUP BY fast path.
1618/// `spec_src[i]`: None = count(*) (finalize from the group row
1619/// count); Some(slot) = unique_ops[slot]'s accumulator.
1620enum FusedOp {
1621 CountCol(usize),
1622 AccCol(usize),
1623 /// v7.39 (round 569) — min/max over a bound column.
1624 /// v7.39 (round 690) — `coll` is the column's declared collation.
1625 /// Unlike an enum's member order (which sends the spec to the
1626 /// generic path), a collation rides along, so a collated column
1627 /// keeps the fused lane's shard-parallel scan.
1628 Extreme {
1629 pos: usize,
1630 max: bool,
1631 coll: Option<alloc::string::String>,
1632 },
1633 /// v7.39 (round 716, S07) — the same three shapes over a COMPILED
1634 /// argument expression. `count(least(id, 0))` used to fall off this
1635 /// lane entirely — `fused_layout` only accepted bound columns — and
1636 /// landed in the SERIAL generic loop, which is where the whole 7.6×
1637 /// against PG lived: PG runs the identical cell as a parallel seq
1638 /// scan. The payload is the SPEC INDEX whose `arg_compiled` program
1639 /// to run; the accumulator lanes are the ones the column ops use.
1640 CountExpr(usize),
1641 AccExpr(usize),
1642 ExtremeExpr {
1643 spec: usize,
1644 max: bool,
1645 coll: Option<alloc::string::String>,
1646 },
1647 /// v7.39 (round 724) — string_agg / array_agg over a bound column,
1648 /// optional bound ORDER BY keys. The payload is the spec index; the
1649 /// scan reads arg_pos / order_pos through it. Collection was the
1650 /// last per-row aggregate stuck on the serial generic loop — 32 ms
1651 /// single-threaded on the panel's 500k string_agg where PG runs a
1652 /// parallel plan.
1653 Collect {
1654 spec: usize,
1655 string_kind: bool,
1656 },
1657}
1658
1659/// Returns the (spec_src, unique_ops) layout when EVERY aggregate
1660/// spec is fused-eligible (count*/count/sum/avg over bound columns,
1661/// no FILTER/DISTINCT/arg2/ORDER), else None.
1662fn fused_layout(
1663 agg_specs: &[AggSpec],
1664 arg_pos: &[Option<usize>],
1665 // v7.39 (round 716) — a compiled argument keeps a spec on the fused
1666 // lane now; a bound column still takes the (cheaper) column op.
1667 arg_compiled: &[Option<eval::CompiledExpr>],
1668 // v7.39 (round 724) — bound ORDER BY key positions, for Collect.
1669 order_pos: &[Vec<Option<usize>>],
1670 arg2_literal_val: &[Option<Value<'static>>],
1671) -> Option<(Vec<Option<usize>>, Vec<FusedOp>)> {
1672 if agg_specs.is_empty() {
1673 return None;
1674 }
1675 let has_arg = |i: usize| arg_pos[i].is_some() || arg_compiled[i].is_some();
1676 // v7.39 (round 724) — a collection spec: bound argument, literal
1677 // separator (string_agg), every ORDER BY key a bound column. The
1678 // finalize path (sort + join) is the ordinary AggState one, so
1679 // multi-key and DESC orders are the finalizer's business, not ours.
1680 let collectible = |i: usize, s: &AggSpec| -> bool {
1681 !s.distinct
1682 && s.filter.is_none()
1683 && !s.first_ordered
1684 && arg_pos[i].is_some()
1685 && s.order_by
1686 .iter()
1687 .enumerate()
1688 .all(|(k, _)| order_pos[i].get(k).copied().flatten().is_some())
1689 && match s.name.as_str() {
1690 "string_agg" => matches!(&arg2_literal_val[i], Some(Value::Text(_))),
1691 "array_agg" => s.arg2.is_none() && s.enum_labels.is_none(),
1692 _ => false,
1693 }
1694 };
1695 let eligible = agg_specs.iter().enumerate().all(|(i, s)| {
1696 collectible(i, s)
1697 || (s.filter.is_none()
1698 && s.arg2.is_none()
1699 && s.order_by.is_empty()
1700 && !s.distinct
1701 && !s.first_ordered
1702 && match s.name.as_str() {
1703 "count_star" => s.arg.is_none(),
1704 "count" | "sum" | "avg" => has_arg(i),
1705 // v7.39 (round 569) — an enum argument compares by
1706 // catalog member order, which the fused lane does not
1707 // carry; those keep the generic path.
1708 "min" | "max" => has_arg(i) && s.enum_labels.is_none(),
1709 _ => false,
1710 })
1711 });
1712 if !eligible {
1713 return None;
1714 }
1715 let mut unique_ops: Vec<FusedOp> = Vec::new();
1716 // Compiled dedupe key = the source Expr (same rule the executor-time
1717 // CSE uses): two specs share a slot only when their argument TREES
1718 // are equal, which `fully_compilable`'s purity makes sufficient.
1719 let same_arg = |j: usize, i: usize| agg_specs[j].arg == agg_specs[i].arg;
1720 let spec_src: Vec<Option<usize>> = agg_specs
1721 .iter()
1722 .enumerate()
1723 .map(|(i, s)| match s.name.as_str() {
1724 "count_star" => None,
1725 // Collection ops never share slots (each keeps its own
1726 // items), so no dedupe probe.
1727 "string_agg" | "array_agg" => {
1728 unique_ops.push(FusedOp::Collect {
1729 spec: i,
1730 string_kind: s.name.as_str() == "string_agg",
1731 });
1732 Some(unique_ops.len() - 1)
1733 }
1734 "min" | "max" => {
1735 let max = s.name.as_str() == "max";
1736 let slot = if let Some(p) = arg_pos[i] {
1737 unique_ops
1738 .iter()
1739 .position(|o| {
1740 matches!(o, FusedOp::Extreme { pos, max: m, coll }
1741 if *pos == p && *m == max && *coll == s.arg_collation)
1742 })
1743 .unwrap_or_else(|| {
1744 unique_ops.push(FusedOp::Extreme {
1745 pos: p,
1746 max,
1747 coll: s.arg_collation.clone(),
1748 });
1749 unique_ops.len() - 1
1750 })
1751 } else {
1752 unique_ops
1753 .iter()
1754 .position(|o| {
1755 matches!(o, FusedOp::ExtremeExpr { spec, max: m, coll }
1756 if same_arg(*spec, i) && *m == max && *coll == s.arg_collation)
1757 })
1758 .unwrap_or_else(|| {
1759 unique_ops.push(FusedOp::ExtremeExpr {
1760 spec: i,
1761 max,
1762 coll: s.arg_collation.clone(),
1763 });
1764 unique_ops.len() - 1
1765 })
1766 };
1767 Some(slot)
1768 }
1769 "count" => {
1770 let slot = if let Some(p) = arg_pos[i] {
1771 unique_ops
1772 .iter()
1773 .position(|o| matches!(o, FusedOp::CountCol(q) if *q == p))
1774 .unwrap_or_else(|| {
1775 unique_ops.push(FusedOp::CountCol(p));
1776 unique_ops.len() - 1
1777 })
1778 } else {
1779 unique_ops
1780 .iter()
1781 .position(|o| matches!(o, FusedOp::CountExpr(j) if same_arg(*j, i)))
1782 .unwrap_or_else(|| {
1783 unique_ops.push(FusedOp::CountExpr(i));
1784 unique_ops.len() - 1
1785 })
1786 };
1787 Some(slot)
1788 }
1789 _ => {
1790 let slot = if let Some(p) = arg_pos[i] {
1791 unique_ops
1792 .iter()
1793 .position(|o| matches!(o, FusedOp::AccCol(q) if *q == p))
1794 .unwrap_or_else(|| {
1795 unique_ops.push(FusedOp::AccCol(p));
1796 unique_ops.len() - 1
1797 })
1798 } else {
1799 unique_ops
1800 .iter()
1801 .position(|o| matches!(o, FusedOp::AccExpr(j) if same_arg(*j, i)))
1802 .unwrap_or_else(|| {
1803 unique_ops.push(FusedOp::AccExpr(i));
1804 unique_ops.len() - 1
1805 })
1806 };
1807 Some(slot)
1808 }
1809 })
1810 .collect();
1811 Some((spec_src, unique_ops))
1812}
1813
1814/// v7.39 (parallel-agg P1) — fold shard accumulator `b` into `a`.
1815/// Every FusedAcc field is a running sum plus a type-witness flag, so
1816/// the merge is field-wise addition with `numeric_add` aligning the
1817/// decimal scales. Merging in shard order keeps float summation
1818/// deterministic for a given shard count (PG's parallel aggregate
1819/// makes the same no-serial-equivalence tradeoff for floats).
1820fn merge_fused(a: &mut FusedAcc, b: &mut FusedAcc) {
1821 // v7.39 (round 569) — fold the shard's extreme in the direction this
1822 // accumulator was built for.
1823 if let Some(be) = &b.extreme {
1824 let take = match &a.extreme {
1825 None => true,
1826 Some(ae) => {
1827 let ord = extreme_cmp_in(None, a.extreme_coll.as_deref(), be, ae, a.extreme_mysql);
1828 if a.extreme_max {
1829 ord == core::cmp::Ordering::Greater
1830 } else {
1831 ord == core::cmp::Ordering::Less
1832 }
1833 }
1834 };
1835 if take {
1836 a.extreme = Some(be.clone());
1837 }
1838 }
1839 a.num.count += b.num.count;
1840 a.num.sum_int += b.num.sum_int;
1841 a.num.sum_float += b.num.sum_float;
1842 a.num.use_float |= b.num.use_float;
1843 a.num.float_not_real |= b.num.float_not_real;
1844 if b.num.use_numeric {
1845 // v7.39 (read01 numeric.c) — fold the shard's bignum spill first,
1846 // then its i128 lane (zero if the shard promoted).
1847 if let Some(bb) = &b.num.sum_big {
1848 sum_add_bignum(
1849 &mut a.num.sum_num_scaled,
1850 &mut a.num.sum_num_scale,
1851 &mut a.num.sum_big,
1852 bb,
1853 );
1854 }
1855 sum_add_exact(
1856 &mut a.num.sum_num_scaled,
1857 &mut a.num.sum_num_scale,
1858 &mut a.num.sum_big,
1859 b.num.sum_num_scaled,
1860 b.num.sum_num_scale,
1861 );
1862 a.num.sum_num_kind = fold_sum_kind(a.num.sum_num_kind, b.num.sum_num_kind);
1863 a.num.use_numeric = true;
1864 }
1865 a.num.sum_iv_months += b.num.sum_iv_months;
1866 a.num.sum_iv_days += b.num.sum_iv_days;
1867 a.num.sum_iv_micros += b.num.sum_iv_micros;
1868 a.num.use_interval |= b.num.use_interval;
1869 a.num.sum_money += b.num.sum_money;
1870 a.num.use_money |= b.num.use_money;
1871 // v7.39 (round 724) — collection lanes concatenate; shard order is
1872 // row order. The merge takes `b` by reference (both call sites), so
1873 // this clones — the per-shard vectors are moved into place only at
1874 // fill time.
1875 a.items.extend(core::mem::take(&mut b.items));
1876 a.item_keys.extend(core::mem::take(&mut b.item_keys));
1877}
1878
1879/// v7.39 — write fused accumulators into the per-spec AggStates
1880/// (shared by the single-group and parallel-GROUP-BY fast paths).
1881/// `group_rows` finalizes count(*) specs.
1882/// v7.39 (round 724) — one row's contribution to a fused Collect op.
1883/// Mirrors `update_state`'s StringAgg / ArrayAgg arms: string_agg skips
1884/// NULL and renders through the shared helper (a non-renderable type
1885/// errors with the same sentence); array_agg keeps NULL elements.
1886fn collect_cell(
1887 a: &mut FusedAcc,
1888 row: &crate::join::RowRef<'_>,
1889 pos: usize,
1890 key_pos: &[Option<usize>],
1891 string_kind: bool,
1892) -> Result<(), EvalError> {
1893 let v = row.get(pos).unwrap_or(&Value::Null);
1894 if string_kind {
1895 if matches!(v, Value::Null) {
1896 return Ok(());
1897 }
1898 let Some(item) = render_string_agg_item(v) else {
1899 return Err(EvalError::TypeMismatch {
1900 detail: format!(
1901 "string_agg requires text value, got {}",
1902 crate::conversions::pg_type_name_for_error_opt(v.data_type())
1903 ),
1904 });
1905 };
1906 a.items.push(item);
1907 } else {
1908 a.items.push(v.clone().into_owned());
1909 }
1910 a.num.count += 1;
1911 for kp in key_pos {
1912 let kv = row
1913 .get(kp.expect("layout-gated bound key"))
1914 .cloned()
1915 .map(Value::into_owned)
1916 .unwrap_or(Value::Null);
1917 a.item_keys.push(kv);
1918 }
1919 Ok(())
1920}
1921
1922/// The string_agg item rendering, shared by `update_state` and the
1923/// round-724 fused Collect op — one place, so the two paths cannot
1924/// drift. Text collects as-is; other scalars coerce to their text
1925/// rendering (MySQL group_concat semantics — also matches PG's
1926/// cast-then-aggregate idiom for `string_agg(v::text, sep)`).
1927fn render_string_agg_item(v: &Value<'_>) -> Option<Value<'static>> {
1928 match v {
1929 Value::Text(s) => Some(Value::text(s.clone())),
1930 // v7.39 (round 626, S05b/F29) — CHAR(n). PG aggregates a
1931 // bpchar column (`string_agg(c, ',')` -> text) and SPG said
1932 // "string_agg requires text value, got character". The text
1933 // form of a bpchar drops its padding, which is what PG's
1934 // own bpchar->text cast does.
1935 Value::BpChar(s) => Some(Value::text(s.trim_end_matches(' ').to_string())),
1936 // v7.39 (read01 round 111) — xmlagg feeds xml values through this
1937 // shared StringAgg path; render the fragment's text (it joins
1938 // separator-less into the concatenated document).
1939 Value::Xml(s) => Some(Value::text(s.to_string())),
1940 Value::Int(n) => Some(Value::text(n.to_string())),
1941 Value::BigInt(n) => Some(Value::text(n.to_string())),
1942 Value::SmallInt(n) => Some(Value::text(n.to_string())),
1943 Value::Float(f) => Some(Value::text(f.to_string())),
1944 Value::Bool(b) => Some(Value::text(if *b { "1" } else { "0" })),
1945 _ => None,
1946 }
1947}
1948
1949fn fill_states_from_fused(
1950 states: &mut [AggState],
1951 spec_src: &[Option<usize>],
1952 accs: &mut [FusedAcc],
1953 group_rows: i64,
1954 // v7.39 (round 724) — string_agg's literal separator, per spec.
1955 arg2_literal_val: &[Option<Value<'static>>],
1956) {
1957 for (i, src) in spec_src.iter().enumerate() {
1958 let state = &mut states[i];
1959 match src {
1960 None => state.num.count = group_rows,
1961 Some(slot) => {
1962 // Collection lanes MOVE (they are per-spec, never
1963 // shared; see the layout's no-dedupe rule).
1964 {
1965 let a = &mut accs[*slot];
1966 if !a.items.is_empty() {
1967 state.items = core::mem::take(&mut a.items);
1968 state.item_keys = core::mem::take(&mut a.item_keys);
1969 }
1970 }
1971 if let Some(Value::Text(sep)) = &arg2_literal_val[i] {
1972 state.separator = Some(sep.to_string());
1973 }
1974 let a = &accs[*slot];
1975 state.num.count = a.num.count;
1976 state.num.sum_int = a.num.sum_int;
1977 state.num.sum_float = a.num.sum_float;
1978 state.num.use_float = a.num.use_float;
1979 state.num.float_not_real = a.num.float_not_real;
1980 state.num.sum_num_scaled = a.num.sum_num_scaled;
1981 state.num.sum_num_kind = a.num.sum_num_kind;
1982 state.num.sum_num_scale = a.num.sum_num_scale;
1983 state.num.sum_big = a.num.sum_big.clone();
1984 state.num.use_numeric = a.num.use_numeric;
1985 state.num.sum_iv_months = a.num.sum_iv_months;
1986 state.num.sum_iv_days = a.num.sum_iv_days;
1987 state.num.sum_iv_micros = a.num.sum_iv_micros;
1988 state.num.use_interval = a.num.use_interval;
1989 state.num.sum_money = a.num.sum_money;
1990 state.num.use_money = a.num.use_money;
1991 if a.extreme.is_some() {
1992 state.extreme = a.extreme.clone();
1993 }
1994 }
1995 }
1996 }
1997}
1998
1999/// v7.39 (read01 numeric.c) — the bignum spill lane of the NUMERIC sum
2000/// tri-state (i128 mantissa + scale + optional BigNumeric). `None` until the
2001/// i128 lane would overflow; from then on the sum lives in the spill and the
2002/// i128 lane stays frozen at zero (PG's sum(numeric) never saturates).
2003type SumBig = Option<alloc::boxed::Box<spg_storage::bignum::BigNumeric>>;
2004
2005/// Add an exact NUMERIC (mantissa × 10^-scale) into the sum tri-state.
2006fn sum_add_exact(
2007 scaled: &mut i128,
2008 scale: &mut u16,
2009 big: &mut SumBig,
2010 add_scaled: i128,
2011 add_scale: u16,
2012) {
2013 use spg_storage::bignum::BigNumeric;
2014 if let Some(b) = big {
2015 **b = b.add(&BigNumeric::from_i128(add_scaled, add_scale));
2016 return;
2017 }
2018 match crate::numeric::numeric_add_checked(*scaled, *scale, add_scaled, add_scale) {
2019 Some((s, sc)) => {
2020 *scaled = s;
2021 *scale = sc;
2022 }
2023 None => {
2024 *big = Some(alloc::boxed::Box::new(
2025 BigNumeric::from_i128(*scaled, *scale)
2026 .add(&BigNumeric::from_i128(add_scaled, add_scale)),
2027 ));
2028 *scaled = 0;
2029 *scale = 0;
2030 }
2031 }
2032}
2033
2034/// Add a BigNumeric input into the sum tri-state (promotes immediately).
2035fn sum_add_bignum(
2036 scaled: &mut i128,
2037 scale: &mut u16,
2038 big: &mut SumBig,
2039 b_in: &spg_storage::bignum::BigNumeric,
2040) {
2041 use spg_storage::bignum::BigNumeric;
2042 let cur = match big.take() {
2043 Some(b) => *b,
2044 None => {
2045 let c = BigNumeric::from_i128(*scaled, *scale);
2046 *scaled = 0;
2047 *scale = 0;
2048 c
2049 }
2050 };
2051 *big = Some(alloc::boxed::Box::new(cur.add(b_in)));
2052}
2053
2054/// One sum/avg accumulation step — the same variant arms (and the same
2055/// error text) as the single-spec fast path's inline match.
2056#[inline]
2057/// v7.39 (round 569) — one row's contribution to a min/max lane.
2058///
2059/// The same question `accumulate_groups` asks per spec per row, with
2060/// none of the per-spec indexing around it. NULL contributes nothing,
2061/// which is PG's rule and the generic path's.
2062fn fused_extreme_cell(a: &mut FusedAcc, v: &Value<'_>, max: bool) -> Result<(), EvalError> {
2063 if matches!(v, Value::Null) {
2064 return Ok(());
2065 }
2066 // v7.39 (round 626) — the FOURTH place this comparison is made. The
2067 // deny list went onto the dispatched arm and the two inlined grouped
2068 // copies first, and `SELECT min(bool_col) FROM t` — no GROUP BY — still
2069 // answered, because it lands here.
2070 if !a.extreme_mysql && min_max_unsupported_type(v) {
2071 return Err(EvalError::TypeMismatch {
2072 detail: format!(
2073 "function {}({}) does not exist",
2074 if max { "max" } else { "min" },
2075 crate::conversions::pg_type_name_for_error_opt(v.data_type())
2076 ),
2077 });
2078 }
2079 let take = match &a.extreme {
2080 None => true,
2081 Some(prev) => {
2082 let ord = extreme_cmp_in(None, a.extreme_coll.as_deref(), v, prev, a.extreme_mysql);
2083 if max {
2084 ord == core::cmp::Ordering::Greater
2085 } else {
2086 ord == core::cmp::Ordering::Less
2087 }
2088 }
2089 };
2090 if take {
2091 a.extreme = Some(v.clone().into_owned());
2092 }
2093 Ok(())
2094}
2095
2096/// v7.39 (round 626, S05b/F29) — the types PG has no `min`/`max` for.
2097///
2098/// A DENY list, not an allow list, and every entry measured: PG accepts
2099/// min/max over int2 int4 int8 numeric float4 float8 money text varchar
2100/// bpchar name date time timetz timestamp timestamptz interval bytea inet
2101/// cidr and the array types, and refuses exactly these. Writing the allow
2102/// list instead is how round 625's first cut of the string guard managed to
2103/// refuse five overloads PG actually has; a deny list of measured
2104/// rejections cannot over-refuse.
2105fn min_max_unsupported_type(v: &Value<'_>) -> bool {
2106 matches!(
2107 v.data_type(),
2108 Some(
2109 spg_storage::DataType::Bool
2110 | spg_storage::DataType::Uuid
2111 | spg_storage::DataType::Macaddr
2112 | spg_storage::DataType::Macaddr8
2113 | spg_storage::DataType::Json
2114 | spg_storage::DataType::Jsonb
2115 | spg_storage::DataType::Bit(_)
2116 | spg_storage::DataType::BitVarying(_)
2117 | spg_storage::DataType::Xml
2118 | spg_storage::DataType::TsVector
2119 | spg_storage::DataType::TsQuery
2120 // v7.39 (round 641) — a transaction id has no ordering
2121 // operator, so PG has no `min(xid)` / `max(xid)` either:
2122 // "function min(xid) does not exist", measured. SPG
2123 // answered, because a Value::Xid carries a u32 that
2124 // compares perfectly well — which is exactly the trap
2125 // the type exists to avoid.
2126 | spg_storage::DataType::Xid
2127 )
2128 )
2129}
2130
2131/// Fold one value into a running sum/avg. THE accumulator — there is no
2132/// second copy, by design; see `NumAcc` for what four copies cost.
2133///
2134/// No `inline(always)` here, and the reason is measured rather than
2135/// stylistic. The four copies were hand-inlining, so the obvious guess was
2136/// that the collapse would cost a call per row and the attribute would buy
2137/// it back. It did not: with `count` split out of `NumAcc`, `sum(int)`
2138/// over 500k rows lost ~8% WITH the attribute applied. What actually
2139/// mattered was the pointer count — the copy this replaces took one
2140/// `&mut FusedAcc`, and passing `&mut NumAcc` plus a separate `&mut i64`
2141/// made two base pointers. Folding `count` back into the struct closed the
2142/// gap; the attribute never did, so it is not here.
2143fn acc_cell(a: &mut NumAcc, v: &Value<'_>) -> Result<(), EvalError> {
2144 match v {
2145 Value::Null => {}
2146 Value::SmallInt(n) => {
2147 a.sum_int += i64::from(*n);
2148 a.count += 1;
2149 }
2150 Value::Int(n) => {
2151 a.sum_int += i64::from(*n);
2152 a.count += 1;
2153 }
2154 // v7.38 (read01, T4) — BIGINT sums as exact NUMERIC (PG).
2155 Value::BigInt(n) => {
2156 sum_add_exact(
2157 &mut a.sum_num_scaled,
2158 &mut a.sum_num_scale,
2159 &mut a.sum_big,
2160 i128::from(*n),
2161 0,
2162 );
2163 a.use_numeric = true;
2164 a.count += 1;
2165 }
2166 Value::Float(x) => {
2167 a.sum_float += *x;
2168 a.use_float = true;
2169 a.float_not_real = true;
2170 a.count += 1;
2171 }
2172 Value::Real(x) => {
2173 a.sum_float += f64::from(*x);
2174 a.use_float = true;
2175 a.count += 1;
2176 }
2177 Value::Numeric {
2178 scaled,
2179 scale,
2180 kind,
2181 } => {
2182 sum_add_exact(
2183 &mut a.sum_num_scaled,
2184 &mut a.sum_num_scale,
2185 &mut a.sum_big,
2186 *scaled,
2187 *scale,
2188 );
2189 a.sum_num_kind = fold_sum_kind(a.sum_num_kind, *kind);
2190 a.use_numeric = true;
2191 a.count += 1;
2192 }
2193 // v7.39 (read01 numeric.c) — a NumericBig input promotes to the spill.
2194 Value::NumericBig(b) => {
2195 sum_add_bignum(
2196 &mut a.sum_num_scaled,
2197 &mut a.sum_num_scale,
2198 &mut a.sum_big,
2199 b,
2200 );
2201 a.use_numeric = true;
2202 a.count += 1;
2203 }
2204 Value::Interval {
2205 months,
2206 days,
2207 micros,
2208 } => {
2209 a.sum_iv_months += i64::from(*months);
2210 a.sum_iv_days += i64::from(*days);
2211 a.sum_iv_micros += i128::from(*micros);
2212 a.use_interval = true;
2213 a.count += 1;
2214 }
2215 Value::Money(c) => {
2216 a.sum_money += i128::from(*c);
2217 a.use_money = true;
2218 a.count += 1;
2219 }
2220 other => {
2221 return Err(EvalError::TypeMismatch {
2222 detail: format!(
2223 "sum/avg need numeric, got {}",
2224 crate::conversions::pg_type_name_for_error_opt(other.data_type())
2225 ),
2226 });
2227 }
2228 }
2229 Ok(())
2230}
2231
2232/// v7.39 (read01 round 61) — thread the catalog into a stage's context when the
2233/// caller has one. `EvalContext::with_catalog` takes a reference, so this keeps
2234/// the Option handling in one place rather than at four call sites.
2235fn with_catalog<'a>(
2236 ctx: EvalContext<'a>,
2237 catalog: Option<&'a spg_storage::Catalog>,
2238 engine: Option<&'a crate::Engine>,
2239) -> EvalContext<'a> {
2240 let ctx = match catalog {
2241 Some(c) => ctx.with_catalog(c),
2242 None => ctx,
2243 };
2244 match engine {
2245 Some(e) => ctx.with_engine(e),
2246 None => ctx,
2247 }
2248}
2249
2250fn accumulate_groups(
2251 rows: AggRows<'_>,
2252 group_exprs: &[Expr],
2253 agg_specs: &[AggSpec],
2254 schema_cols: &[ColumnSchema],
2255 table_alias: Option<&str>,
2256 correlated_eval: Option<CorrelatedEval<'_>>,
2257 runner: Option<&dyn crate::ParallelRunner>,
2258 // v7.39 (read01 round 61) — the catalog. `run` has carried it since the
2259 // enum-order knife, but the four stages below each built a BARE context and
2260 // dropped it — so a catalog-dependent expression inside an aggregate's
2261 // argument (`string_agg(f1(id), ',')`, a user function) answered "unknown
2262 // function". Same family as rounds 49/53/54/55/56.
2263 catalog: Option<&spg_storage::Catalog>,
2264 engine: Option<&crate::Engine>,
2265) -> Result<Vec<(Vec<Value<'static>>, Vec<AggState>)>, EvalError> {
2266 let ctx = with_catalog(EvalContext::new(schema_cols, table_alias), catalog, engine);
2267 // Map group key (vec of values, encoded as canonical string) -> group state.
2268 // v7.32 (architecture v2, P2b) — insertion-ordered group state in
2269 // a Vec; the hash map only maps key → index. Removes the parallel
2270 // `key_order: Vec<String>` (a second per-group key clone) and the
2271 // per-group re-probe `groups[k]` at finalize (24k hash lookups for
2272 // the inbox shape). The map owns its key once on vacant insert.
2273 let mut order: Vec<(Vec<Value<'static>>, Vec<AggState>)> = Vec::new();
2274 let mut groups: hashbrown::HashMap<String, usize> = hashbrown::HashMap::new();
2275 // v7.37.x (mailrs Track A perf — SPGE ≫ PG18) — single-Text GROUP
2276 // BY column fast path. The canonical-string encode (`S<text>|`)
2277 // + `encode_key_refs_into` reuse-buffer churn dominated the 30 k-
2278 // row mailrs minimal probe (~3-4 ms / 30 k). For `GROUP BY t` on
2279 // a TEXT column (the inbox-listing / conversation-grouping shape)
2280 // the column text IS the canonical key — no encoder, no prefix
2281 // byte, no `refs` Vec rebuild per row. The fallback `groups` map
2282 // above is retained for multi-col / non-Text / collation paths;
2283 // this map only fires when the schema and value structurally
2284 // permit it. `null_group_idx` collects NULL group rows (SQL groups
2285 // all NULLs into one bucket).
2286 let mut groups_text: hashbrown::HashMap<String, usize> = hashbrown::HashMap::new();
2287 // v7.37.16 — raw-i64 group map for the single-INT GROUP BY fast path.
2288 let mut groups_int: hashbrown::HashMap<i64, usize> = hashbrown::HashMap::new();
2289 let mut null_group_idx: Option<usize> = None;
2290 // When there are no GROUP BY exprs *and* there is at least one aggregate,
2291 // every row collapses into a single anonymous group keyed by "".
2292 if rows.is_empty() && group_exprs.is_empty() {
2293 // Single empty-aggregate group: count=0, sum=0, max=NULL, etc.
2294 // No rows follow, so the map is never probed — seed `order` only.
2295 let init: Vec<AggState> = (0..agg_specs.len()).map(|_| AggState::default()).collect();
2296 order.push((Vec::new(), init));
2297 }
2298
2299 // v7.30 (perf campaign) - hoist the per-row work that doesn't
2300 // depend on the row: which group exprs need collation folding
2301 // (none, for most queries - the old code cloned the whole
2302 // group_vals vec per row just in case).
2303 // v7.30 (perf campaign) - the no-tax row loop. When a group
2304 // expr or an aggregate argument is a bare column reference
2305 // (the overwhelmingly common shape), bind its position ONCE
2306 // and read row cells by offset in the loop - no per-row tree
2307 // walk, no owned-Value clone out of resolve_column. Anything
2308 // more complex keeps the eval path.
2309 let col_pos = |e: &Expr| -> Option<usize> {
2310 // v7.37.16 — bind bare names too, via the compiled-WHERE
2311 // resolver: `compile_column_pos` mirrors resolve_column's
2312 // happy layers exactly (composite → prefix/alias gate → bare
2313 // exact → unique suffix) and returns None on anything that
2314 // would reach an ambiguity / whole-row / error path, so the
2315 // eval fallback keeps identical semantics. Previously only
2316 // qualified refs bound (via the looser find_column_pos), so
2317 // single-table `GROUP BY g` / `avg(v)` ran the per-row
2318 // eval_expr tree-walk + Vec + encode_key String alloc — the
2319 // heavy.rs group_by / filter_agg residual loss vs PG18.
2320 if let Expr::Column(c) = e {
2321 eval::compile_column_pos(c, &ctx)
2322 } else {
2323 None
2324 }
2325 };
2326 let group_pos: Vec<Option<usize>> = group_exprs.iter().map(col_pos).collect();
2327 let all_groups_bound = group_pos.iter().all(Option::is_some);
2328 // v7.37.x — single-col GROUP BY on a TEXT-typed column lets the
2329 // hot loop key the hash map by the column text directly. Resolved
2330 // once from the bound position against `schema_cols`.
2331 // v7.39 (round 364, M4 P2) — the raw-text GROUP BY fast path keys
2332 // by the column's bytes, which cannot fold; a MySQL session takes
2333 // the general encoder path (which folds) instead.
2334 let single_text_group_col: bool = !ctx.mysql_dialect
2335 && group_pos.len() == 1
2336 && group_pos[0].is_some_and(|p| {
2337 schema_cols
2338 .get(p)
2339 .is_some_and(|c| matches!(c.ty, spg_storage::DataType::Text))
2340 });
2341 // v7.37.16 (heavy.rs group_500k 1.12× loss) — single-col GROUP BY on
2342 // an INTEGER-typed column keys the map by the raw i64 instead of the
2343 // canonical-string encode ("I{n}|" write! + String-keyed hash probe
2344 // was ~25-40 ns of the 42 ns/row 500k GROUP BY budget). Mirrors the
2345 // single-Text fast path; NULLs share `null_group_idx`; a non-integer
2346 // cell (coercion edge) falls back to the encoded path.
2347 let single_int_group_col: bool = group_pos.len() == 1
2348 && group_pos[0].is_some_and(|p| {
2349 schema_cols.get(p).is_some_and(|c| {
2350 matches!(
2351 c.ty,
2352 spg_storage::DataType::SmallInt
2353 | spg_storage::DataType::Int
2354 | spg_storage::DataType::BigInt
2355 )
2356 })
2357 });
2358 let arg_pos: Vec<Option<usize>> = agg_specs
2359 .iter()
2360 .map(|spec| spec.arg.as_ref().and_then(|e| col_pos(e)))
2361 .collect();
2362 // v7.39 (round 370, M4 P4a) — the MySQL dialect folds GROUP BY /
2363 // DISTINCT text keys (M4 P2), EXCEPT over a column with an explicit
2364 // `COLLATE utf8mb4_bin` (stored `Binary`), which de-dups byte-wise.
2365 // A folding default column stores `CaseInsensitive`, so only an
2366 // explicit binary column suppresses the fold. Multi-column GROUP BY
2367 // mixing a binary and a folding column is treated byte-wise as a whole
2368 // (rare; residual).
2369 let is_binary_key_col = |p: Option<usize>| -> bool {
2370 p.and_then(|i| schema_cols.get(i))
2371 .is_some_and(|c| matches!(c.collation, spg_storage::Collation::Binary))
2372 };
2373 // v7.39 (round 371, M4 P4b) — a per-expression `… COLLATE utf8mb4_bin`
2374 // / `BINARY …` key is byte-wise too, so its GROUP BY / DISTINCT does
2375 // not fold. The clause lowers to a `binary` cast the parser emits.
2376 let mysql_fold_groups: bool = ctx.mysql_dialect
2377 && !group_pos.iter().any(|&p| is_binary_key_col(p))
2378 && !group_exprs
2379 .iter()
2380 .any(|e| crate::eval::is_binary_coerced(e));
2381 // v7.38.18 — the padding mask, built beside the fold mask off the
2382 // same argument column so the two cannot come from different places.
2383 let distinct_pads: Vec<bool> = arg_pos
2384 .iter()
2385 .map(|&p| {
2386 p.and_then(|i| schema_cols.get(i))
2387 .is_some_and(|c| crate::collate::pads_space(c.collation_name.as_deref()))
2388 })
2389 .collect();
2390 let distinct_fold_case: Vec<bool> = arg_pos.iter().map(|&p| !is_binary_key_col(p)).collect();
2391 let distinct_fold: Vec<bool> = agg_specs
2392 .iter()
2393 .enumerate()
2394 .map(|(i, spec)| {
2395 // v7.38.18 — a byte-wise column still needs this step when
2396 // its collation PADS. `utf8mb4_bin` folds no case and
2397 // ignores trailing spaces, which is one flag short of what
2398 // a single boolean can say; the pad mask beside this one
2399 // carries the second half and the fold is skipped by
2400 // `distinct_fold_case` below.
2401 ctx.mysql_dialect
2402 && (!is_binary_key_col(arg_pos[i]) || distinct_pads[i])
2403 && !spec
2404 .arg
2405 .as_ref()
2406 .is_some_and(|e| crate::eval::is_binary_coerced(e))
2407 })
2408 .collect();
2409 // v7.37.x (mailrs Track A 100k attack) — dedicated tight loop
2410 // for the "single-Text GROUP BY + single MAX(bound numeric arg)"
2411 // shape. This is the mailrs `/api/conversations` minimal shape
2412 // (`GROUP BY thread_id, MAX(internal_date)`) and an inbox-listing
2413 // staple across the SPG customer set. Skipping the per-row spec
2414 // loop, FILTER / arg2 / order_keys checks, and the union-typed
2415 // `update_state` enum jump saves ~80-100 ns/row at 100 k input
2416 // — the gap closing the SPGE vs PG18 ratio at this scale.
2417 let dedicated_max_loop: bool = single_text_group_col
2418 && agg_specs.len() == 1
2419 && matches!(agg_specs[0].kind, AggKind::Max)
2420 && agg_specs[0].filter.is_none()
2421 && agg_specs[0].arg2.is_none()
2422 && agg_specs[0].order_by.is_empty()
2423 && !agg_specs[0].distinct
2424 && !agg_specs[0].first_ordered
2425 && arg_pos[0].is_some();
2426 // v7.36 (perf — mailrs Ask 1 SUM(LENGTH(text_body)) 18ms → ?) —
2427 // pre-compile every aggregate arg that's a `fully_compilable`
2428 // PURE expression over bound columns. Without this, `LENGTH(col)`
2429 // / `COALESCE(col, '')` / `CAST(col AS BIGINT)` etc. ALL fell
2430 // through to the `(None, Some(e)) => eval_arg(e, mat, ...)` slow
2431 // path that materialises a Cow<Row> per input row — for a 25k-row
2432 // JOIN that's 25k full-row clones for one column read. The Step
2433 // VM (`eval_compiled_ref`) reads columns by RowRef::get and runs
2434 // the same `apply_function` dispatcher with zero materialisation.
2435 let arg_compiled: Vec<Option<eval::CompiledExpr>> = agg_specs
2436 .iter()
2437 .enumerate()
2438 .map(|(i, spec)| match (&arg_pos[i], &spec.arg) {
2439 (Some(_), _) => None,
2440 (None, Some(e)) if eval::fully_compilable(e) => Some(eval::compile_expr(e, &ctx)),
2441 _ => None,
2442 })
2443 .collect();
2444 // v7.37.4 (L1 — executor-time CSE / mailrs P0) — dedupe
2445 // compiled aggregate-arg expressions across specs. mailrs's
2446 // `/api/conversations` SQL has 14 aggregates whose compiled
2447 // CASE/CAST arg expressions overlap heavily (`m.message_id != ''`
2448 // re-appears 4×, the inner `CASE WHEN m.message_id != '' THEN
2449 // m.message_id ELSE CAST(m.id AS TEXT) END` re-appears 3×). Each
2450 // dup currently costs one Step-VM walk per row — 100k rows ×
2451 // ~3-4 redundant evals = ~300-400k wasted Step-VM runs.
2452 //
2453 // Dedupe key = source `Expr` (PartialEq). `CompiledExpr` itself
2454 // is not `Hash` / `Eq`, but n_specs is small (≤ ~20 in practice);
2455 // O(n²) PartialEq probe cost = ~196 cmp per query, vs millions
2456 // of saved per-row evals. `fully_compilable` requires PURE
2457 // scalars (no NOW / RANDOM / sequence accessors), so an earlier
2458 // eval has identical observable semantics to the original.
2459 //
2460 // `arg_slot[i] = Some(s)` means spec `i`'s compiled arg lives in
2461 // slot `s` of `arg_unique_idx` (which points back into
2462 // `arg_compiled` for the canonical owner). Per-row cache fills
2463 // LAZILY — preserves the current FILTER semantics where an arg
2464 // whose spec is filtered out is never evaluated (and never
2465 // surfaces a type error). Reset to `None` at the top of each row.
2466 let mut arg_unique_idx: Vec<usize> = Vec::new();
2467 let mut arg_slot: Vec<Option<usize>> = Vec::with_capacity(agg_specs.len());
2468 arg_slot.resize(agg_specs.len(), None);
2469 for (i, spec) in agg_specs.iter().enumerate() {
2470 if arg_pos[i].is_some() || arg_compiled[i].is_none() {
2471 continue;
2472 }
2473 let src = spec.arg.as_ref().expect("arg_compiled => spec.arg is Some");
2474 let pos = arg_unique_idx
2475 .iter()
2476 .position(|&j| agg_specs[j].arg.as_ref().is_some_and(|other| other == src));
2477 arg_slot[i] = Some(match pos {
2478 Some(p) => p,
2479 None => {
2480 arg_unique_idx.push(i);
2481 arg_unique_idx.len() - 1
2482 }
2483 });
2484 }
2485 let mut row_eval_cache: Vec<Option<Value>> = Vec::with_capacity(arg_unique_idx.len());
2486 row_eval_cache.resize(arg_unique_idx.len(), None);
2487 // v7.33 (array_agg perf) — bound positions for each spec's internal
2488 // ORDER BY keys, so an ordered aggregate (`array_agg(x ORDER BY y)`)
2489 // reads the sort key by reference (RowRef::get) instead of
2490 // materialising the whole combined join row per input row just to
2491 // eval one bound column. Mirrors arg_pos. On the inbox shape this
2492 // turned 24k full-row (~1 KB each) clones into 24k single-cell reads.
2493 let order_pos: Vec<Vec<Option<usize>>> = agg_specs
2494 .iter()
2495 .map(|spec| spec.order_by.iter().map(|o| col_pos(&o.expr)).collect())
2496 .collect();
2497 // v7.37.43 (DISTA A-3) — precompute the per-spec arg2 when it is a
2498 // bare literal. `string_agg(DISTINCT col, ',')` and every other
2499 // call with a constant separator goes through this path; PG evaluates
2500 // arg2 as a Const once at plan time. SPG was paying a Cow row
2501 // materialisation per input row purely so `eval_arg(literal, &row)`
2502 // could run — but a literal doesn't read the row at all. Hoist the
2503 // literal value into a per-query table; per-row arg2 just clones it.
2504 //
2505 // Sentinel: when arg2 is present but NOT a literal, the entry stays
2506 // `None` and the per-row path still falls into the eval branch
2507 // (which forces `needs_mat`).
2508 let arg2_literal_val: Vec<Option<Value<'static>>> = agg_specs
2509 .iter()
2510 .map(|s| match &s.arg2 {
2511 Some(Expr::Literal(l)) => Some(eval::literal_to_value(l)),
2512 _ => None,
2513 })
2514 .collect();
2515 // Does any spec need the fully-materialised row in the bound fast
2516 // path — a FILTER, a non-bound value arg, a NON-LITERAL second arg,
2517 // or a non-bound ORDER key? When false (every aggregate arg/key is a
2518 // bound column — the inbox shape, and the DISTA shape after A-3)
2519 // the bound fast path never materialises a row.
2520 let needs_mat = agg_specs.iter().enumerate().any(|(i, s)| {
2521 s.filter.is_some()
2522 || (s.arg.is_some() && arg_pos[i].is_none() && arg_compiled[i].is_none())
2523 || (s.arg2.is_some() && arg2_literal_val[i].is_none())
2524 || order_pos[i].iter().any(Option::is_none)
2525 });
2526 let ci_positions: Vec<usize> = group_exprs
2527 .iter()
2528 .enumerate()
2529 .filter(|(_, g)| {
2530 matches!(
2531 eval::column_collation(g, &ctx),
2532 Some(spg_storage::Collation::CaseInsensitive)
2533 )
2534 })
2535 .map(|(i, _)| i)
2536 .collect();
2537 // v7.31 (perf 3e) — per-row scratch buffers. The fast path used
2538 // to allocate a key String (and a refs Vec) for EVERY row just
2539 // to probe the group map; hits — the overwhelming case — now
2540 // touch the allocator zero times.
2541 let mut keybuf_s = String::new();
2542 // v7.36 — reused Step VM eval stack for compiled aggregate args.
2543 // v7.37.9 T3 S2 — elided lifetime so the Vec's `'val` binds to the
2544 // row-borrow lifetime per call (`eval_compiled_ref<'row, 'val>` now
2545 // requires `'row: 'val`). Caller-side Vec<Value<'_>> lets compiler
2546 // infer the shortest lifetime that covers all calls.
2547 let mut eval_stack: Vec<Value<'_>> = Vec::new();
2548 let mut dkeybuf = String::new();
2549 let mut refs: Vec<&Value> = Vec::with_capacity(group_pos.len());
2550 // v7.32 (round-31) — an aggregate's argument / FILTER / second arg /
2551 // ORDER key may itself be a *correlated* subquery, e.g.
2552 // `MAX((SELECT i.v FROM inner i WHERE i.fk = o.id))`. A non-correlated
2553 // subquery is pre-resolved to a literal before this loop, but a
2554 // correlated one survives as a subquery node and must be evaluated per
2555 // outer row through the correlated evaluator — the same hook the
2556 // select-list / HAVING / ORDER finalisers already use below. Plain
2557 // `eval_expr` would hit "subquery reached row eval".
2558 //
2559 // The `any_agg_subquery` gate is computed once here so the common case
2560 // (no subquery anywhere in the aggregate args — including every hot
2561 // scan/group aggregate) short-circuits before the per-row
2562 // `expr_has_subquery` walk: `eval_arg` is then exactly `eval_expr`.
2563 let any_agg_subquery = correlated_eval.is_some()
2564 && agg_specs.iter().any(|s| {
2565 s.filter
2566 .as_ref()
2567 .is_some_and(|e| crate::expr_has_subquery(e))
2568 || s.arg.as_ref().is_some_and(|e| crate::expr_has_subquery(e))
2569 || s.arg2.as_ref().is_some_and(|e| crate::expr_has_subquery(e))
2570 || s.order_by.iter().any(|o| crate::expr_has_subquery(&o.expr))
2571 });
2572 let eval_arg =
2573 |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| -> Result<Value<'static>, EvalError> {
2574 match correlated_eval {
2575 Some(f) if any_agg_subquery && crate::expr_has_subquery(e) => f(e, r, c),
2576 _ => eval::eval_expr(e, r, c),
2577 }
2578 };
2579 // v7.36 (perf — mailrs Phase 1, post u64-hash) — single
2580 // anonymous group fast path. When the query has no GROUP BY
2581 // (`SELECT SUM(LENGTH(col)) FROM ...`, COUNT, AVG, etc.) the
2582 // whole input collapses into one group. The fast path below
2583 // still pays one `groups.get("")` hash probe per row plus
2584 // `entry = &mut order[0]` reindex even when the empty-key
2585 // path encodes nothing — measured ~50 ns/row across 25 k rows
2586 // = ~1.25 ms of pure bookkeeping on the user_storage_usage
2587 // baseline.
2588 //
2589 // Bypass: lift `entry` outside the loop and feed every row
2590 // straight into it. Same `update_state` machinery, zero
2591 // per-row hash work, zero per-row index lookup.
2592 let single_anon_group = group_exprs.is_empty() && !rows.is_empty();
2593 if single_anon_group {
2594 // Seed the single group at idx 0 once.
2595 let init: Vec<AggState> = (0..agg_specs.len()).map(|_| AggState::default()).collect();
2596 order.clear();
2597 order.push((Vec::new(), init));
2598 }
2599 // v7.36 (perf — mailrs Phase 1, count_messages 2.58 → ?) —
2600 // `COUNT(*)` short-circuit. For a single-anon-group `COUNT(*)`
2601 // with no FILTER / DISTINCT, every survivor counts once — the
2602 // answer IS `rows.len()`. Skips the 25 k iterations of
2603 // `update_state("count_star", …)` on the mailrs count_messages
2604 // shape; the JOIN already produced exactly the set of rows
2605 // that must be counted.
2606 if single_anon_group
2607 && agg_specs.len() == 1
2608 && agg_specs[0].name == "count_star"
2609 && agg_specs[0].filter.is_none()
2610 && agg_specs[0].arg.is_none()
2611 && agg_specs[0].arg2.is_none()
2612 && agg_specs[0].order_by.is_empty()
2613 && !agg_specs[0].distinct
2614 {
2615 let state = &mut order[0].1[0];
2616 state.num.count = rows.len() as i64;
2617 return Ok(order);
2618 }
2619 // v7.37.16 (heavy.rs agg_500k 1.6× loss) — fused streaming accumulator
2620 // for ANY number of count(*)/count(col)/sum(col)/avg(col) specs over
2621 // BOUND columns (no FILTER/DISTINCT/arg2/ORDER). The generic per-row
2622 // spec loop paid arg dispatch + union-typed update_state per spec per
2623 // row (~10 ns/spec/row); PG's parallel agg runs the 500k 3-spec shape
2624 // at ~18 ns/row effective. Three cuts:
2625 // - count(*) never enters the row loop — it IS rows.len();
2626 // - sum/avg over the SAME column share one accumulator (identical
2627 // running state), so `count(*), sum(v), avg(v)` does ONE cell read
2628 // and one accumulate per row;
2629 // - remaining ops run in one tight pass, no update_state.
2630 // Finalize writes the same AggState fields as the single-spec path.
2631 if single_anon_group
2632 && let Some((spec_src, unique_ops)) = fused_layout(
2633 agg_specs,
2634 &arg_pos,
2635 &arg_compiled,
2636 &order_pos,
2637 &arg2_literal_val,
2638 )
2639 {
2640 let mut accs: Vec<FusedAcc> = fused_accs(&unique_ops, ctx.mysql_dialect);
2641 // v7.39 (parallel-agg P1) — shard the row scan across the
2642 // host-injected executor when the input is large enough.
2643 // Each shard runs the same tight loop over its row range and
2644 // returns its own Vec<FusedAcc>; the merge is field-wise
2645 // (see merge_fused). Errors inside a shard surface as the
2646 // shard result and re-raise after join.
2647 // v7.39 (round 716) — the scan takes its EvalContext as a
2648 // parameter: `EvalContext` is not Sync (per-eval memo Cells, the
2649 // sequence resolver's plain `&dyn Fn`), so the parallel branch
2650 // hands each shard a locally-built minimal context instead of
2651 // capturing the outer one. The compiled ops only reach the parts
2652 // a shard context carries — columns, alias, dialect, catalog —
2653 // because `fully_compilable` excludes everything else (params,
2654 // sequences, user functions, FTS).
2655 let fused_scan = |range: core::ops::Range<usize>,
2656 accs: &mut Vec<FusedAcc>,
2657 fctx: &EvalContext<'_>|
2658 -> Result<(), EvalError> {
2659 // One Step-VM stack per shard call, reused across every
2660 // row and every compiled op.
2661 let mut stack: Vec<Value<'_>> = Vec::new();
2662 for row in rows.range(range.start, range.end).iter() {
2663 for (si, op) in unique_ops.iter().enumerate() {
2664 match op {
2665 FusedOp::CountCol(p) => {
2666 if !matches!(row.get(*p), Some(Value::Null) | None) {
2667 accs[si].num.count += 1;
2668 }
2669 }
2670 FusedOp::AccCol(p) => {
2671 {
2672 let a = &mut accs[si];
2673 acc_cell(&mut a.num, row.get(*p).unwrap_or(&Value::Null))
2674 }?;
2675 }
2676 FusedOp::Extreme { pos, max, .. } => {
2677 fused_extreme_cell(
2678 &mut accs[si],
2679 row.get(*pos).unwrap_or(&Value::Null),
2680 *max,
2681 )?;
2682 }
2683 FusedOp::CountExpr(sp) => {
2684 let c = arg_compiled[*sp].as_ref().expect("gated compiled");
2685 let v = eval::eval_compiled_ref(c, row, fctx, &mut stack)?;
2686 if !matches!(v, Value::Null) {
2687 accs[si].num.count += 1;
2688 }
2689 }
2690 FusedOp::AccExpr(sp) => {
2691 let c = arg_compiled[*sp].as_ref().expect("gated compiled");
2692 let v = eval::eval_compiled_ref(c, row, fctx, &mut stack)?;
2693 acc_cell(&mut accs[si].num, &v)?;
2694 }
2695 FusedOp::ExtremeExpr { spec, max, .. } => {
2696 let c = arg_compiled[*spec].as_ref().expect("gated compiled");
2697 let v = eval::eval_compiled_ref(c, row, fctx, &mut stack)?;
2698 fused_extreme_cell(&mut accs[si], &v, *max)?;
2699 }
2700 FusedOp::Collect { spec, string_kind } => {
2701 collect_cell(
2702 &mut accs[si],
2703 &row,
2704 arg_pos[*spec].expect("gated bound"),
2705 &order_pos[*spec],
2706 *string_kind,
2707 )?;
2708 }
2709 }
2710 }
2711 }
2712 Ok(())
2713 };
2714 if !unique_ops.is_empty() {
2715 let par = runner.filter(|_| rows.len() >= crate::PARALLEL_MIN_ROWS);
2716 if let Some(r) = par {
2717 crate::PARALLEL_AGG_FIRED.fetch_add(1, core::sync::atomic::Ordering::Relaxed);
2718 let n_shards = (rows.len() / crate::PARALLEL_MIN_ROWS).clamp(2, 8);
2719 let chunk = rows.len().div_ceil(n_shards);
2720 type ShardOut = Result<Vec<FusedAcc>, EvalError>;
2721 let ops = &unique_ops;
2722 let mysql_for_accs = ctx.mysql_dialect;
2723 // v7.39 (round 716) — the whitelisted concat family
2724 // renders through the SESSION's style; a shard context
2725 // built from defaults would silently re-render dates and
2726 // floats the default way. RenderStyle is Copy.
2727 let outer_style = ctx.render_style;
2728 let results = r.run_shards(n_shards, &|i| {
2729 let lo = i * chunk;
2730 let hi = ((i + 1) * chunk).min(rows.len());
2731 let mut local: Vec<FusedAcc> = fused_accs(ops, mysql_for_accs);
2732 // Shard-local minimal context (the outer one is not
2733 // Sync); see the fused_scan comment.
2734 let mut sctx = EvalContext::new(schema_cols, table_alias);
2735 sctx.mysql_dialect = mysql_for_accs;
2736 sctx.render_style = outer_style;
2737 let sctx = match catalog {
2738 Some(c) => sctx.with_catalog(c),
2739 None => sctx,
2740 };
2741 let out: ShardOut = fused_scan(lo..hi, &mut local, &sctx).map(|()| local);
2742 alloc::boxed::Box::new(out)
2743 });
2744 for boxed in results {
2745 let shard = boxed
2746 .downcast::<ShardOut>()
2747 .expect("runner echoes the closure's box");
2748 let mut shard_accs = (*shard)?;
2749 for (si, b) in shard_accs.iter_mut().enumerate() {
2750 merge_fused(&mut accs[si], b);
2751 }
2752 }
2753 } else {
2754 fused_scan(0..rows.len(), &mut accs, &ctx)?;
2755 }
2756 }
2757 fill_states_from_fused(
2758 &mut order[0].1,
2759 &spec_src,
2760 &mut accs,
2761 rows.len() as i64,
2762 &arg2_literal_val,
2763 );
2764 return Ok(order);
2765 }
2766 // v7.39 (parallel-agg P3) — parallel GROUP BY fast path: a single
2767 // bound INT group column with every spec fused-eligible (the
2768 // `GROUP BY g` + count/sum/avg panel shape). Shards build local
2769 // i64-keyed maps of FusedAcc slots; the merge folds maps in shard
2770 // order (first-seen group order across shards — SQL leaves GROUP
2771 // BY output order unspecified). Any non-integer cell under the
2772 // integer schema (coercion edge) aborts the shard and the whole
2773 // scan falls back to the serial path below.
2774 if single_int_group_col
2775 && group_exprs.len() == 1
2776 && rows.len() >= crate::PARALLEL_MIN_ROWS
2777 && let Some(r) = runner
2778 && let Some((spec_src, unique_ops)) = fused_layout(
2779 agg_specs,
2780 &arg_pos,
2781 &arg_compiled,
2782 &order_pos,
2783 &arg2_literal_val,
2784 )
2785 && !unique_ops.is_empty()
2786 {
2787 crate::PARALLEL_AGG_FIRED.fetch_add(1, core::sync::atomic::Ordering::Relaxed);
2788 let gp = group_pos[0].expect("single_int_group_col implies bound");
2789 struct ShardMap {
2790 // first-seen order of keys within the shard.
2791 keys: Vec<(i64, Value<'static>)>,
2792 slots: hashbrown::HashMap<i64, Vec<FusedAcc>>,
2793 null_slot: Option<Vec<FusedAcc>>,
2794 null_rows: i64,
2795 key_rows: hashbrown::HashMap<i64, i64>,
2796 }
2797 // Err(None) = coercion edge -> serial fallback; Err(Some(e)) = real error.
2798 type ShardOut = Result<ShardMap, Option<EvalError>>;
2799 let n_shards = (rows.len() / crate::PARALLEL_MIN_ROWS).clamp(2, 8);
2800 let chunk = rows.len().div_ceil(n_shards);
2801 let ops = &unique_ops;
2802 let mysql_for_accs = ctx.mysql_dialect;
2803 // Same session-style carry as the anonymous-group lane.
2804 let outer_style = ctx.render_style;
2805 let results = r.run_shards(n_shards, &|si| {
2806 let lo = si * chunk;
2807 let hi = ((si + 1) * chunk).min(rows.len());
2808 let mut m = ShardMap {
2809 keys: Vec::new(),
2810 slots: hashbrown::HashMap::new(),
2811 null_slot: None,
2812 null_rows: 0,
2813 key_rows: hashbrown::HashMap::new(),
2814 };
2815 let out: ShardOut = (|| {
2816 // v7.39 (round 716) — per-shard Step-VM stack for the
2817 // compiled-argument ops, reused across rows, plus a
2818 // shard-local minimal context (the outer one is not
2819 // Sync); see the anonymous-group fused_scan comment.
2820 let mut stack: Vec<Value<'_>> = Vec::new();
2821 let mut sctx = EvalContext::new(schema_cols, table_alias);
2822 sctx.mysql_dialect = mysql_for_accs;
2823 sctx.render_style = outer_style;
2824 let sctx = match catalog {
2825 Some(c) => sctx.with_catalog(c),
2826 None => sctx,
2827 };
2828 for row in rows.range(lo, hi).iter() {
2829 let v = row.get(gp).unwrap_or(&Value::Null);
2830 let key: Option<i64> = match v {
2831 Value::SmallInt(n) => Some(i64::from(*n)),
2832 Value::Int(n) => Some(i64::from(*n)),
2833 Value::BigInt(n) => Some(*n),
2834 Value::Null => None,
2835 _ => return Err(None), // coercion edge -> serial
2836 };
2837 let slots = match key {
2838 Some(k) => {
2839 *m.key_rows.entry(k).or_insert(0) += 1;
2840 m.slots.entry(k).or_insert_with(|| {
2841 m.keys.push((k, v.clone().into_owned()));
2842 fused_accs(ops, mysql_for_accs)
2843 })
2844 }
2845 None => {
2846 m.null_rows += 1;
2847 m.null_slot
2848 .get_or_insert_with(|| fused_accs(ops, mysql_for_accs))
2849 }
2850 };
2851 for (oi, op) in ops.iter().enumerate() {
2852 match op {
2853 FusedOp::CountCol(p) => {
2854 if !matches!(row.get(*p), Some(Value::Null) | None) {
2855 slots[oi].num.count += 1;
2856 }
2857 }
2858 FusedOp::AccCol(p) => {
2859 {
2860 let a = &mut slots[oi];
2861 acc_cell(&mut a.num, row.get(*p).unwrap_or(&Value::Null))
2862 }
2863 .map_err(Some)?;
2864 }
2865 FusedOp::Extreme { pos, max, .. } => {
2866 fused_extreme_cell(
2867 &mut slots[oi],
2868 row.get(*pos).unwrap_or(&Value::Null),
2869 *max,
2870 )
2871 .map_err(Some)?;
2872 }
2873 FusedOp::CountExpr(sp) => {
2874 let c = arg_compiled[*sp].as_ref().expect("gated compiled");
2875 let v = eval::eval_compiled_ref(c, row, &sctx, &mut stack)
2876 .map_err(Some)?;
2877 if !matches!(v, Value::Null) {
2878 slots[oi].num.count += 1;
2879 }
2880 }
2881 FusedOp::AccExpr(sp) => {
2882 let c = arg_compiled[*sp].as_ref().expect("gated compiled");
2883 let v = eval::eval_compiled_ref(c, row, &sctx, &mut stack)
2884 .map_err(Some)?;
2885 acc_cell(&mut slots[oi].num, &v).map_err(Some)?;
2886 }
2887 FusedOp::ExtremeExpr { spec, max, .. } => {
2888 let c = arg_compiled[*spec].as_ref().expect("gated compiled");
2889 let v = eval::eval_compiled_ref(c, row, &sctx, &mut stack)
2890 .map_err(Some)?;
2891 fused_extreme_cell(&mut slots[oi], &v, *max).map_err(Some)?;
2892 }
2893 FusedOp::Collect { spec, string_kind } => {
2894 collect_cell(
2895 &mut slots[oi],
2896 &row,
2897 arg_pos[*spec].expect("gated bound"),
2898 &order_pos[*spec],
2899 *string_kind,
2900 )
2901 .map_err(Some)?;
2902 }
2903 }
2904 }
2905 }
2906 Ok(m)
2907 })();
2908 alloc::boxed::Box::new(out)
2909 });
2910 // Merge in shard order; a fallback sentinel drops to serial.
2911 let mut merged_keys: Vec<(i64, Value<'static>)> = Vec::new();
2912 let mut merged: hashbrown::HashMap<i64, (Vec<FusedAcc>, i64)> = hashbrown::HashMap::new();
2913 let mut merged_null: Option<(Vec<FusedAcc>, i64)> = None;
2914 let mut fallback = false;
2915 let mut shard_err: Option<EvalError> = None;
2916 for boxed in results {
2917 let shard = boxed
2918 .downcast::<ShardOut>()
2919 .expect("runner echoes the closure's box");
2920 match *shard {
2921 Ok(mut m) => {
2922 for (k, kv) in m.keys {
2923 // Removed (not borrowed): the slot MOVES into the
2924 // merged map on first sight, and the round-724
2925 // collection lanes move out of it on merge.
2926 let mut accs = m.slots.remove(&k).expect("keyed slot");
2927 let rows_k = m.key_rows[&k];
2928 match merged.get_mut(&k) {
2929 Some((dst, cnt)) => {
2930 for (i, b) in accs.iter_mut().enumerate() {
2931 merge_fused(&mut dst[i], b);
2932 }
2933 *cnt += rows_k;
2934 }
2935 None => {
2936 merged_keys.push((k, kv));
2937 merged.insert(k, (accs, rows_k));
2938 }
2939 }
2940 }
2941 if let Some(mut nb) = m.null_slot.take() {
2942 match &mut merged_null {
2943 Some((dst, cnt)) => {
2944 for (i, b) in nb.iter_mut().enumerate() {
2945 merge_fused(&mut dst[i], b);
2946 }
2947 *cnt += m.null_rows;
2948 }
2949 None => merged_null = Some((nb, m.null_rows)),
2950 }
2951 }
2952 }
2953 Err(None) => fallback = true,
2954 Err(Some(e)) => shard_err = Some(e),
2955 }
2956 }
2957 if let Some(e) = shard_err {
2958 return Err(e);
2959 }
2960 if !fallback {
2961 for (k, kv) in merged_keys {
2962 let (mut accs, group_rows) = merged.remove(&k).expect("key recorded");
2963 let mut states: Vec<AggState> =
2964 (0..agg_specs.len()).map(|_| AggState::default()).collect();
2965 fill_states_from_fused(
2966 &mut states,
2967 &spec_src,
2968 &mut accs,
2969 group_rows,
2970 &arg2_literal_val,
2971 );
2972 order.push((alloc::vec![kv], states));
2973 }
2974 if let Some((mut accs, group_rows)) = merged_null {
2975 let mut states: Vec<AggState> =
2976 (0..agg_specs.len()).map(|_| AggState::default()).collect();
2977 fill_states_from_fused(
2978 &mut states,
2979 &spec_src,
2980 &mut accs,
2981 group_rows,
2982 &arg2_literal_val,
2983 );
2984 order.push((alloc::vec![Value::Null], states));
2985 }
2986 return Ok(order);
2987 }
2988 // fallthrough: serial paths below handle the coercion edge.
2989 }
2990
2991 // v7.36 (perf — mailrs Phase 1) — `COUNT(<bound col>)` (non-`*`)
2992 // collapses to: read the cell, increment when not NULL. Skips
2993 // the per-row spec dispatch + `update_state("count", …)`.
2994 if single_anon_group
2995 && agg_specs.len() == 1
2996 && agg_specs[0].name == "count"
2997 && agg_specs[0].filter.is_none()
2998 && agg_specs[0].arg2.is_none()
2999 && agg_specs[0].order_by.is_empty()
3000 && !agg_specs[0].distinct
3001 && arg_pos[0].is_some()
3002 {
3003 let p = arg_pos[0].unwrap();
3004 let mut count: i64 = 0;
3005 for row in rows.iter() {
3006 if !matches!(row.get(p), Some(Value::Null) | None) {
3007 count += 1;
3008 }
3009 }
3010 let state = &mut order[0].1[0];
3011 state.num.count = count;
3012 return Ok(order);
3013 }
3014 // v7.36 (perf — mailrs Phase 1, user_storage_usage 7.5 → ?) —
3015 // single-aggregate streaming accumulator. For
3016 // `SUM(<compiled-expr>)` / `SUM(<bound col>)` with no GROUP BY,
3017 // no FILTER, no arg2, no ORDER BY, no DISTINCT, the whole
3018 // per-row work collapses to: eval the arg, match the Value
3019 // variant, accumulate. Skips the spec-dispatch loop +
3020 // `update_state` per-row name match. On a 25 k-row JOIN
3021 // (user_storage_usage `SUM(LENGTH(text_body))`) that's
3022 // ~50-100 ns/row of pure spec-dispatch overhead removed.
3023 if single_anon_group
3024 && agg_specs.len() == 1
3025 && agg_specs[0].filter.is_none()
3026 && agg_specs[0].arg2.is_none()
3027 && agg_specs[0].order_by.is_empty()
3028 && !agg_specs[0].distinct
3029 && (agg_specs[0].name == "sum" || agg_specs[0].name == "avg")
3030 && (arg_pos[0].is_some() || arg_compiled[0].is_some())
3031 {
3032 let arg_pos0 = arg_pos[0];
3033 let arg_c0 = &arg_compiled[0];
3034 // v7.39 (round 665) — was fifteen loose locals mirroring
3035 // `NumAcc` field for field; `FusedAcc`'s doc comment even
3036 // said so. One struct now, folded by the one `acc_cell`.
3037 let mut na = NumAcc::default();
3038 // Borrow-aware fast inner: avoid the per-row clone when arg
3039 // is a bound column position.
3040 if let Some(p) = arg_pos0 {
3041 for row in rows.iter() {
3042 let v_ref = row.get(p).unwrap_or(&Value::Null);
3043 acc_cell(&mut na, v_ref)?;
3044 }
3045 } else if let Some(p) = arg_c0.as_ref().and_then(|c| c.as_single_column_length()) {
3046 // v7.36 (perf — mailrs Phase 1, user_storage_usage hot
3047 // inner) — `SUM(LENGTH(<text col>))` collapses to a
3048 // straight scan: read the cell by ref, branch on the
3049 // variant, do an ASCII probe + `len()` (or
3050 // `chars().count()` on non-ASCII), accumulate. No Step
3051 // VM, no stack push/pop, no `BigInt` boxing on the way
3052 // out — pure i64 sum. The original Step VM path keeps
3053 // running for everything outside this shape (`SUM(col)`,
3054 // `SUM(expr)`, multi-step compiled args).
3055 for row in rows.iter() {
3056 let Some(v_ref) = row.get(p) else {
3057 continue;
3058 };
3059 let n = match v_ref {
3060 Value::Null => continue,
3061 Value::Text(s) => {
3062 if s.is_ascii() {
3063 s.len() as i64
3064 } else {
3065 s.chars().count() as i64
3066 }
3067 }
3068 other => {
3069 return Err(EvalError::TypeMismatch {
3070 detail: format!(
3071 "length() needs text, got {}",
3072 crate::conversions::pg_type_name_for_error_opt(other.data_type())
3073 ),
3074 });
3075 }
3076 };
3077 na.sum_int += n;
3078 na.count += 1;
3079 }
3080 } else {
3081 let c = arg_c0.as_ref().unwrap();
3082 for row in rows.iter() {
3083 let v = eval::eval_compiled_ref(c, row, &ctx, &mut eval_stack)?;
3084 acc_cell(&mut na, &v)?;
3085 }
3086 }
3087 let state = &mut order[0].1[0];
3088 state.num = na;
3089 return Ok(order);
3090 }
3091 // v7.37.x (mailrs Track A 100k attack) — tight inlined loop for
3092 // the "single-Text GROUP BY + single MAX(bound numeric arg)"
3093 // shape. See `dedicated_max_loop` above for the gate. Returns
3094 // straight to the caller; the rest of the function (single-anon,
3095 // bound-fast, eval-slow paths) is skipped.
3096 if dedicated_max_loop && !single_anon_group {
3097 let gpos = group_pos[0].expect("dedicated_max_loop gates on Some");
3098 let apos = arg_pos[0].expect("dedicated_max_loop gates on Some");
3099 for row in rows.iter() {
3100 let kv = row.get(gpos).unwrap_or(&Value::Null);
3101 let idx = match kv {
3102 Value::Text(s) => match groups_text.get(s.as_ref()) {
3103 Some(&i) => i,
3104 None => {
3105 let i = order.len();
3106 order.push((
3107 alloc::vec![Value::text(s.clone())],
3108 alloc::vec![AggState::default()],
3109 ));
3110 groups_text.insert(s.to_string(), i);
3111 i
3112 }
3113 },
3114 Value::Null => match null_group_idx {
3115 Some(i) => i,
3116 None => {
3117 let i = order.len();
3118 order.push((alloc::vec![Value::Null], alloc::vec![AggState::default()]));
3119 null_group_idx = Some(i);
3120 i
3121 }
3122 },
3123 _ => {
3124 // Schema said Text but value isn't — fall back to
3125 // the generic encoded path for correctness.
3126 refs.clear();
3127 refs.push(kv);
3128 encode_key_refs_into_in(&refs, &mut keybuf_s, mysql_fold_groups);
3129 match groups.get(keybuf_s.as_str()) {
3130 Some(&i) => i,
3131 None => {
3132 let i = order.len();
3133 order.push((
3134 alloc::vec![kv.clone().into_owned()],
3135 alloc::vec![AggState::default()],
3136 ));
3137 groups.insert(keybuf_s.clone(), i);
3138 i
3139 }
3140 }
3141 }
3142 };
3143 // Inline MAX accumulator — skip the union-typed
3144 // `update_state` enum jump and per-spec arg dispatch.
3145 let av = row.get(apos).unwrap_or(&Value::Null);
3146 if !matches!(av, Value::Null) {
3147 let st = &mut order[idx].1[0];
3148 let upd = match &st.extreme {
3149 None => true,
3150 Some(prev) => {
3151 extreme_cmp_in(
3152 agg_specs[0].enum_labels.as_deref(),
3153 agg_specs[0].arg_collation.as_deref(),
3154 av,
3155 prev,
3156 ctx.mysql_dialect,
3157 ) == core::cmp::Ordering::Greater
3158 }
3159 };
3160 if upd {
3161 st.extreme = Some(av.clone().into_owned());
3162 }
3163 }
3164 }
3165 return Ok(order);
3166 }
3167
3168 for row in rows.iter() {
3169 // v7.37.4 (L1 CSE) — reset per-row cache for shared compiled
3170 // aggregate-arg evals. No-op when no dedupe (empty vec).
3171 for slot in row_eval_cache.iter_mut() {
3172 *slot = None;
3173 }
3174 if single_anon_group {
3175 let entry = &mut order[0];
3176 let mat: Option<Cow<'_, Row>> = if needs_mat { Some(row.as_row()) } else { None };
3177 for (i, spec) in agg_specs.iter().enumerate() {
3178 if let Some(f) = &spec.filter
3179 && !matches!(
3180 eval_arg(f, mat.as_deref().expect("needs_mat for FILTER"), &ctx)?,
3181 Value::Bool(true)
3182 )
3183 {
3184 continue;
3185 }
3186 let arg_owned: Value;
3187 let arg_ref: &Value = match (&arg_pos[i], arg_slot[i], &spec.arg) {
3188 (Some(p), _, _) => {
3189 // v7.37.9 Phase 1A-ext counter — fast position-bound arg.
3190 crate::bump_counter!(AGG_PER_ROW_FAST_POS);
3191 row.get(*p).unwrap_or(&Value::Null)
3192 }
3193 (None, None, None) => {
3194 // COUNT(*) sentinel
3195 crate::bump_counter!(AGG_PER_ROW_COUNT_STAR_SENTINEL);
3196 arg_owned = Value::Bool(true);
3197 &arg_owned
3198 }
3199 (None, Some(s), _) => {
3200 if row_eval_cache[s].is_none() {
3201 // v7.37.9 Phase 1A-ext counter — Step-VM ran (cache miss).
3202 crate::bump_counter!(AGG_PER_ROW_COMPILED_MISS);
3203 let c = arg_compiled[arg_unique_idx[s]]
3204 .as_ref()
3205 .expect("arg_unique_idx points at a compiled spec");
3206 let v = eval::eval_compiled_ref(c, row, &ctx, &mut eval_stack)?;
3207 row_eval_cache[s] = Some(v);
3208 } else {
3209 // v7.37.9 Phase 1A-ext counter — CSE cache hit
3210 // (compiled arg deduped across specs in same row).
3211 crate::bump_counter!(AGG_PER_ROW_COMPILED_HIT);
3212 }
3213 row_eval_cache[s].as_ref().expect("just filled above")
3214 }
3215 (None, None, Some(e)) => {
3216 // v7.37.9 Phase 1A-ext counter — eval_expr fallback
3217 // (uncompilable spec — Cow row materialise per row).
3218 crate::bump_counter!(AGG_PER_ROW_EVAL_FALLBACK);
3219 arg_owned = eval_arg(
3220 e,
3221 mat.as_deref().expect("needs_mat for non-bound arg"),
3222 &ctx,
3223 )?;
3224 &arg_owned
3225 }
3226 };
3227 let arg2_val = match (&spec.arg2, &arg2_literal_val[i]) {
3228 (None, _) => None,
3229 // v7.37.43 (DISTA A-3) — literal arg2: clone the
3230 // precomputed value, skip per-row eval & row mat.
3231 (Some(_), Some(lit)) => {
3232 // v7.37.9 Phase 0 diagnostic — count per-row
3233 // hits of the DISTA A-3 fast path.
3234 crate::bump_counter!(DISTA_LITERAL_ARG2_CACHE_FIRE);
3235 Some(lit.clone())
3236 }
3237 (Some(e), None) => Some(eval_arg(
3238 e,
3239 mat.as_deref().expect("needs_mat for arg2"),
3240 &ctx,
3241 )?),
3242 };
3243 let order_keys: Option<Vec<Value<'static>>> = if spec.order_by.is_empty() {
3244 None
3245 } else {
3246 crate::bump_counter!(AGGREGATE_ARRAY_AGG_ORDER_BY_FIRE);
3247 let mut keys: Vec<Value<'static>> = Vec::with_capacity(spec.order_by.len());
3248 for (k, o) in spec.order_by.iter().enumerate() {
3249 let v: Value<'static> = if let Some(p) = order_pos[i][k] {
3250 row.get(p)
3251 .cloned()
3252 .map(Value::into_owned)
3253 .unwrap_or(Value::Null)
3254 } else {
3255 eval_arg(
3256 &o.expr,
3257 mat.as_deref().expect("needs_mat for ORDER key"),
3258 &ctx,
3259 )?
3260 };
3261 keys.push(v);
3262 }
3263 Some(keys)
3264 };
3265 // v7.36 (perf — bugfix v7.36.1 candidate) — first_ordered
3266 // was missing from the single_anon_group fast path,
3267 // sending `(array_agg(x ORDER BY y))[1]` values into
3268 // `update_state(array_agg, …)` whose finalize ignored
3269 // the absent `first_best` and returned `[]`. The slow
3270 // path below has the same branch — keep them aligned.
3271 if spec.first_ordered {
3272 if let Some(keys) = order_keys {
3273 let st = &mut entry.1[i];
3274 let better = match &st.first_best {
3275 None => true,
3276 Some((bk, _)) => {
3277 cmp_order_keys(
3278 &spec.order_by,
3279 &spec.order_enum_labels,
3280 &spec.order_collations,
3281 &keys,
3282 bk,
3283 ctx.mysql_dialect,
3284 ) == core::cmp::Ordering::Less
3285 }
3286 };
3287 if better {
3288 st.first_best = Some((keys, arg_ref.clone().into_owned()));
3289 }
3290 }
3291 continue;
3292 }
3293 if spec.distinct {
3294 // v7.37.x (mailrs Track A 100k distinct_aggs attack)
3295 // — single-Text DISTINCT fast path. Within a single
3296 // distinct spec all input values come from one
3297 // expression and share one type, so the encode-
3298 // prefix (`S<text>|`) is redundant: the column
3299 // text alone is collision-free within this spec's
3300 // `seen` set. Skips encode_one + 2-walk
3301 // contains+insert; only Text arms apply, others
3302 // ride the encoded path unchanged.
3303 //
3304 // v7.37.x (docker-fair DISTA attack) — extend the
3305 // single-family fast path to BigInt via a parallel
3306 // `seen_int: Option<BTreeSet<i64>>`. The DISTA
3307 // `COUNT(DISTINCT m.id)` shape pumps 25 k BigInt
3308 // probes; skipping `encode_key_refs_into` saves
3309 // ~100 ns of alloc + format churn per row.
3310 if let Value::Text(s) = arg_ref {
3311 // v7.39 (round 364, M4 P2) — a MySQL session folds
3312 // the distinct key (case/accent) so `Foo`/`foo`
3313 // count once. The `seen` set stays internally
3314 // consistent: both probe and insert fold.
3315 // v7.39 (round 370, M4 P4a) — but an explicit
3316 // `COLLATE utf8mb4_bin` column de-dups byte-wise.
3317 if distinct_fold[i] {
3318 // v7.38.18 — and pad when the argument's
3319 // collation says trailing spaces do not
3320 // count. `utf8mb4_general_ci` folds AND
3321 // pads; `utf8mb4_0900_ai_ci` only folds.
3322 let base = if distinct_pads[i] {
3323 s.trim_end_matches(' ')
3324 } else {
3325 s.as_ref()
3326 };
3327 let k = if distinct_fold_case[i] {
3328 spg_storage::mysql_ci_fold(base)
3329 } else {
3330 alloc::string::ToString::to_string(base)
3331 };
3332 if entry.1[i].seen.contains(k.as_str()) {
3333 continue;
3334 }
3335 entry.1[i].seen.insert(k);
3336 } else {
3337 if entry.1[i].seen.contains(s.as_ref()) {
3338 continue;
3339 }
3340 entry.1[i].seen.insert(s.to_string());
3341 }
3342 } else if let Value::BigInt(n) = arg_ref {
3343 let set = entry.1[i].seen_int.get_or_insert_with(BTreeSet::new);
3344 if !set.insert(*n) {
3345 continue;
3346 }
3347 } else if let Value::Int(n) = arg_ref {
3348 let set = entry.1[i].seen_int.get_or_insert_with(BTreeSet::new);
3349 if !set.insert(i64::from(*n)) {
3350 continue;
3351 }
3352 } else {
3353 encode_key_refs_into_in(
3354 core::slice::from_ref(&arg_ref),
3355 &mut dkeybuf,
3356 distinct_fold[i],
3357 );
3358 if entry.1[i].seen.contains(dkeybuf.as_str()) {
3359 continue;
3360 }
3361 entry.1[i].seen.insert(dkeybuf.clone());
3362 }
3363 }
3364 // v7.37.x (mailrs Track A 100k attack) — inline the
3365 // common aggregate kinds (MAX / MIN / Count / CountStar
3366 // / BoolOr / BoolAnd) here instead of dispatching
3367 // through `update_state`'s enum jump + per-kind branch.
3368 // Skipping the function-call overhead saves ~20-30 ns
3369 // per spec per row at 100 k; the slow kinds keep the
3370 // dispatched call.
3371 match spec.kind {
3372 AggKind::Max => {
3373 if !matches!(arg_ref, Value::Null) {
3374 // v7.39 (round 626) — the same deny list the
3375 // dispatched path applies. These inlined copies
3376 // exist for speed and are where `min(TRUE)`
3377 // actually lands, so a guard placed only on the
3378 // dispatched arm never fires.
3379 if !ctx.mysql_dialect && min_max_unsupported_type(arg_ref) {
3380 return Err(EvalError::TypeMismatch {
3381 detail: format!(
3382 "function max({}) does not exist",
3383 crate::conversions::pg_type_name_for_error_opt(
3384 arg_ref.data_type()
3385 )
3386 ),
3387 });
3388 }
3389 let st = &mut entry.1[i];
3390 let upd = match &st.extreme {
3391 None => true,
3392 Some(prev) => {
3393 extreme_cmp_in(
3394 spec.enum_labels.as_deref(),
3395 spec.arg_collation.as_deref(),
3396 arg_ref,
3397 prev,
3398 ctx.mysql_dialect,
3399 ) == core::cmp::Ordering::Greater
3400 }
3401 };
3402 if upd {
3403 st.extreme = Some(arg_ref.clone().into_owned());
3404 }
3405 }
3406 }
3407 AggKind::Min => {
3408 if !matches!(arg_ref, Value::Null) {
3409 // v7.39 (round 626) — see the Max arm above.
3410 if !ctx.mysql_dialect && min_max_unsupported_type(arg_ref) {
3411 return Err(EvalError::TypeMismatch {
3412 detail: format!(
3413 "function min({}) does not exist",
3414 crate::conversions::pg_type_name_for_error_opt(
3415 arg_ref.data_type()
3416 )
3417 ),
3418 });
3419 }
3420 let st = &mut entry.1[i];
3421 let upd = match &st.extreme {
3422 None => true,
3423 Some(prev) => {
3424 extreme_cmp_in(
3425 spec.enum_labels.as_deref(),
3426 spec.arg_collation.as_deref(),
3427 arg_ref,
3428 prev,
3429 ctx.mysql_dialect,
3430 ) == core::cmp::Ordering::Less
3431 }
3432 };
3433 if upd {
3434 st.extreme = Some(arg_ref.clone().into_owned());
3435 }
3436 }
3437 }
3438 AggKind::AnyValue => {
3439 if !matches!(arg_ref, Value::Null) {
3440 let st = &mut entry.1[i];
3441 if st.extreme.is_none() {
3442 st.extreme = Some(arg_ref.clone().into_owned());
3443 }
3444 }
3445 }
3446 AggKind::CountStar => {
3447 entry.1[i].num.count += 1;
3448 }
3449 AggKind::Count => {
3450 if !matches!(arg_ref, Value::Null) {
3451 entry.1[i].num.count += 1;
3452 }
3453 }
3454 AggKind::BoolOr => match arg_ref {
3455 Value::Bool(b) => {
3456 let st = &mut entry.1[i];
3457 st.bool_acc = Some(st.bool_acc.unwrap_or(false) || *b);
3458 }
3459 Value::Null => {}
3460 _ => update_state(
3461 &mut entry.1[i],
3462 spec.kind,
3463 &spec.name,
3464 arg_ref,
3465 arg2_val.as_ref(),
3466 order_keys,
3467 spec.enum_labels.as_deref(),
3468 spec.arg_collation.as_deref(),
3469 ctx.mysql_dialect,
3470 )?,
3471 },
3472 AggKind::BoolAnd => match arg_ref {
3473 Value::Bool(b) => {
3474 let st = &mut entry.1[i];
3475 st.bool_acc = Some(st.bool_acc.unwrap_or(true) && *b);
3476 }
3477 Value::Null => {}
3478 _ => update_state(
3479 &mut entry.1[i],
3480 spec.kind,
3481 &spec.name,
3482 arg_ref,
3483 arg2_val.as_ref(),
3484 order_keys,
3485 spec.enum_labels.as_deref(),
3486 spec.arg_collation.as_deref(),
3487 ctx.mysql_dialect,
3488 )?,
3489 },
3490 _ => {
3491 update_state(
3492 &mut entry.1[i],
3493 spec.kind,
3494 &spec.name,
3495 arg_ref,
3496 arg2_val.as_ref(),
3497 order_keys,
3498 spec.enum_labels.as_deref(),
3499 spec.arg_collation.as_deref(),
3500 ctx.mysql_dialect,
3501 )?;
3502 }
3503 }
3504 }
3505 continue;
3506 }
3507 // Fast key: bound positions + no ci folding -> encode
3508 // straight from borrowed cells; group_vals materialise
3509 // only when the group is NEW.
3510 if all_groups_bound && ci_positions.is_empty() {
3511 // v7.37.x — single-Text fast path uses the raw text as the
3512 // map key (no encode_one's `S<text>|` prefix/suffix push,
3513 // no refs Vec rebuild). NULL values land in a dedicated
3514 // slot so SQL's "all NULLs share one group" semantics hold.
3515 let idx = if single_text_group_col {
3516 let v = row.get(group_pos[0].unwrap()).unwrap_or(&Value::Null);
3517 match v {
3518 Value::Text(s) => match groups_text.get(s.as_ref()) {
3519 Some(&i) => i,
3520 None => {
3521 let i = order.len();
3522 let init: Vec<AggState> =
3523 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3524 order.push((alloc::vec![Value::text(s.clone())], init));
3525 groups_text.insert(s.to_string(), i);
3526 i
3527 }
3528 },
3529 Value::Null => match null_group_idx {
3530 Some(i) => i,
3531 None => {
3532 let i = order.len();
3533 let init: Vec<AggState> =
3534 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3535 order.push((alloc::vec![Value::Null], init));
3536 null_group_idx = Some(i);
3537 i
3538 }
3539 },
3540 _ => {
3541 // Schema says Text but value is something else
3542 // (coercion edge case). Fall back to the encoded
3543 // path for correctness — same logic as the
3544 // non-single-Text branch below.
3545 refs.clear();
3546 refs.push(v);
3547 encode_key_refs_into_in(&refs, &mut keybuf_s, mysql_fold_groups);
3548 match groups.get(keybuf_s.as_str()) {
3549 Some(&i) => i,
3550 None => {
3551 let i = order.len();
3552 let init: Vec<AggState> =
3553 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3554 order.push((alloc::vec![v.clone().into_owned()], init));
3555 groups.insert(keybuf_s.clone(), i);
3556 i
3557 }
3558 }
3559 }
3560 }
3561 } else if single_int_group_col {
3562 // v7.37.16 — raw-i64 keying (see single_int_group_col).
3563 let v = row.get(group_pos[0].unwrap()).unwrap_or(&Value::Null);
3564 let key: Option<i64> = match v {
3565 Value::SmallInt(n) => Some(i64::from(*n)),
3566 Value::Int(n) => Some(i64::from(*n)),
3567 Value::BigInt(n) => Some(*n),
3568 _ => None,
3569 };
3570 match (key, v) {
3571 (Some(k), _) => match groups_int.get(&k) {
3572 Some(&i) => i,
3573 None => {
3574 let i = order.len();
3575 let init: Vec<AggState> =
3576 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3577 order.push((alloc::vec![v.clone().into_owned()], init));
3578 groups_int.insert(k, i);
3579 i
3580 }
3581 },
3582 (None, Value::Null) => match null_group_idx {
3583 Some(i) => i,
3584 None => {
3585 let i = order.len();
3586 let init: Vec<AggState> =
3587 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3588 order.push((alloc::vec![Value::Null], init));
3589 null_group_idx = Some(i);
3590 i
3591 }
3592 },
3593 (None, _) => {
3594 // Non-integer cell under an integer schema
3595 // (coercion edge) — encoded-path fallback.
3596 refs.clear();
3597 refs.push(v);
3598 encode_key_refs_into_in(&refs, &mut keybuf_s, mysql_fold_groups);
3599 match groups.get(keybuf_s.as_str()) {
3600 Some(&i) => i,
3601 None => {
3602 let i = order.len();
3603 let init: Vec<AggState> =
3604 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3605 order.push((alloc::vec![v.clone().into_owned()], init));
3606 groups.insert(keybuf_s.clone(), i);
3607 i
3608 }
3609 }
3610 }
3611 }
3612 } else {
3613 refs.clear();
3614 refs.extend(
3615 group_pos
3616 .iter()
3617 .map(|p| row.get(p.unwrap()).unwrap_or(&Value::Null)),
3618 );
3619 encode_key_refs_into_in(&refs, &mut keybuf_s, mysql_fold_groups);
3620 match groups.get(keybuf_s.as_str()) {
3621 Some(&i) => i,
3622 None => {
3623 let i = order.len();
3624 let init: Vec<AggState> =
3625 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3626 let owned: Vec<Value<'static>> =
3627 refs.iter().map(|v| (*v).clone().into_owned()).collect();
3628 order.push((owned, init));
3629 groups.insert(keybuf_s.clone(), i);
3630 i
3631 }
3632 }
3633 };
3634 let entry = &mut order[idx];
3635 // v7.33 (array_agg perf) — materialise the combined row AT
3636 // MOST once per input row, and only when a spec actually
3637 // needs the eval path (FILTER / non-bound arg / arg2 / non-
3638 // bound ORDER key). Bound args and bound ORDER keys read
3639 // cells by reference below, so the inbox shape (all bound)
3640 // never materialises — killing the per-row ~1 KB clone that
3641 // dominated the ordered-aggregate cost.
3642 let mat: Option<Cow<'_, Row>> = if needs_mat { Some(row.as_row()) } else { None };
3643 for (i, spec) in agg_specs.iter().enumerate() {
3644 // v7.32 (round-29) — FILTER (WHERE cond): exclude rows
3645 // where cond is not TRUE before they reach this
3646 // aggregate's accumulator (and before DISTINCT dedup).
3647 if let Some(f) = &spec.filter
3648 && !matches!(
3649 eval_arg(f, mat.as_deref().expect("needs_mat for FILTER"), &ctx)?,
3650 Value::Bool(true)
3651 )
3652 {
3653 continue;
3654 }
3655 let arg_owned: Value;
3656 let arg_ref: &Value = match (&arg_pos[i], arg_slot[i], &spec.arg) {
3657 (Some(p), _, _) => {
3658 crate::bump_counter!(AGG_PER_ROW_FAST_POS);
3659 row.get(*p).unwrap_or(&Value::Null)
3660 }
3661 (None, None, None) => {
3662 crate::bump_counter!(AGG_PER_ROW_COUNT_STAR_SENTINEL);
3663 arg_owned = Value::Bool(true);
3664 &arg_owned
3665 }
3666 (None, Some(s), _) => {
3667 // v7.37.4 (L1 CSE) — shared compiled-arg slot.
3668 // First spec that needs slot `s` this row pays
3669 // the Step-VM eval; siblings reading the same
3670 // slot get the cached Value for free. Preserves
3671 // FILTER semantics: a spec filtered out above
3672 // never reaches here, so its arg stays unevaled.
3673 if row_eval_cache[s].is_none() {
3674 crate::bump_counter!(AGG_PER_ROW_COMPILED_MISS);
3675 let c = arg_compiled[arg_unique_idx[s]]
3676 .as_ref()
3677 .expect("arg_unique_idx points at a compiled spec");
3678 let v = eval::eval_compiled_ref(c, row, &ctx, &mut eval_stack)?;
3679 row_eval_cache[s] = Some(v);
3680 } else {
3681 crate::bump_counter!(AGG_PER_ROW_COMPILED_HIT);
3682 }
3683 row_eval_cache[s].as_ref().expect("just filled above")
3684 }
3685 (None, None, Some(e)) => {
3686 crate::bump_counter!(AGG_PER_ROW_EVAL_FALLBACK);
3687 arg_owned = eval_arg(
3688 e,
3689 mat.as_deref().expect("needs_mat for non-bound arg"),
3690 &ctx,
3691 )?;
3692 &arg_owned
3693 }
3694 };
3695 let arg2_val = match (&spec.arg2, &arg2_literal_val[i]) {
3696 (None, _) => None,
3697 // v7.37.43 (DISTA A-3) — literal arg2: clone the
3698 // precomputed value, skip per-row eval & row mat.
3699 (Some(_), Some(lit)) => {
3700 // v7.37.9 Phase 0 diagnostic — count per-row
3701 // hits of the DISTA A-3 fast path.
3702 crate::bump_counter!(DISTA_LITERAL_ARG2_CACHE_FIRE);
3703 Some(lit.clone())
3704 }
3705 (Some(e), None) => Some(eval_arg(
3706 e,
3707 mat.as_deref().expect("needs_mat for arg2"),
3708 &ctx,
3709 )?),
3710 };
3711 let order_keys: Option<Vec<Value<'static>>> = if spec.order_by.is_empty() {
3712 None
3713 } else {
3714 crate::bump_counter!(AGGREGATE_ARRAY_AGG_ORDER_BY_FIRE);
3715 let mut keys: Vec<Value<'static>> = Vec::with_capacity(spec.order_by.len());
3716 for (k, o) in spec.order_by.iter().enumerate() {
3717 // Bound ORDER key → read the cell by reference; only
3718 // a non-bound key falls to the materialised eval path.
3719 keys.push(match order_pos[i][k] {
3720 Some(p) => row
3721 .get(p)
3722 .cloned()
3723 .map(Value::into_owned)
3724 .unwrap_or(Value::Null),
3725 None => eval_arg(
3726 &o.expr,
3727 mat.as_deref().expect("needs_mat for non-bound ORDER key"),
3728 &ctx,
3729 )?,
3730 });
3731 }
3732 Some(keys)
3733 };
3734 // v7.33 (array_agg argmax) — first_ordered: keep only the
3735 // running first-by-order element (strict-less replacement
3736 // = ties keep the earliest row, matching the stable-sort
3737 // `[1]`), no array build.
3738 if spec.first_ordered {
3739 if let Some(keys) = order_keys {
3740 let st = &mut entry.1[i];
3741 let better = match &st.first_best {
3742 None => true,
3743 Some((bk, _)) => {
3744 cmp_order_keys(
3745 &spec.order_by,
3746 &spec.order_enum_labels,
3747 &spec.order_collations,
3748 &keys,
3749 bk,
3750 ctx.mysql_dialect,
3751 ) == core::cmp::Ordering::Less
3752 }
3753 };
3754 if better {
3755 st.first_best = Some((keys, arg_ref.clone().into_owned()));
3756 }
3757 }
3758 continue;
3759 }
3760 if spec.distinct {
3761 // v7.37.x — single-Text DISTINCT fast path (see
3762 // bound fast path counterpart above). Per-spec
3763 // type invariance lets us use the column text as
3764 // the `seen` key directly, no `S<text>|` prefix.
3765 // v7.37.x (docker-fair DISTA) — BigInt parallel
3766 // path skips encode_key_refs_into entirely.
3767 if let Value::Text(s) = arg_ref {
3768 if entry.1[i].seen.contains(s.as_ref()) {
3769 continue;
3770 }
3771 entry.1[i].seen.insert(s.to_string());
3772 } else if let Value::BigInt(n) = arg_ref {
3773 let set = entry.1[i].seen_int.get_or_insert_with(BTreeSet::new);
3774 if !set.insert(*n) {
3775 continue;
3776 }
3777 } else if let Value::Int(n) = arg_ref {
3778 let set = entry.1[i].seen_int.get_or_insert_with(BTreeSet::new);
3779 if !set.insert(i64::from(*n)) {
3780 continue;
3781 }
3782 } else {
3783 encode_key_refs_into_in(
3784 core::slice::from_ref(&arg_ref),
3785 &mut dkeybuf,
3786 distinct_fold[i],
3787 );
3788 if entry.1[i].seen.contains(dkeybuf.as_str()) {
3789 continue;
3790 }
3791 entry.1[i].seen.insert(dkeybuf.clone());
3792 }
3793 }
3794 // v7.37.x (mailrs Track A 100k attack) — inline the
3795 // common aggregate kinds (MAX / MIN / Count / CountStar
3796 // / BoolOr / BoolAnd) here instead of dispatching
3797 // through `update_state`'s enum jump + per-kind branch.
3798 // Skipping the function-call overhead saves ~20-30 ns
3799 // per spec per row at 100 k; the slow kinds keep the
3800 // dispatched call.
3801 match spec.kind {
3802 AggKind::Max => {
3803 if !matches!(arg_ref, Value::Null) {
3804 // v7.39 (round 626) — the same deny list the
3805 // dispatched path applies. These inlined copies
3806 // exist for speed and are where `min(TRUE)`
3807 // actually lands, so a guard placed only on the
3808 // dispatched arm never fires.
3809 if !ctx.mysql_dialect && min_max_unsupported_type(arg_ref) {
3810 return Err(EvalError::TypeMismatch {
3811 detail: format!(
3812 "function max({}) does not exist",
3813 crate::conversions::pg_type_name_for_error_opt(
3814 arg_ref.data_type()
3815 )
3816 ),
3817 });
3818 }
3819 let st = &mut entry.1[i];
3820 let upd = match &st.extreme {
3821 None => true,
3822 Some(prev) => {
3823 extreme_cmp_in(
3824 spec.enum_labels.as_deref(),
3825 spec.arg_collation.as_deref(),
3826 arg_ref,
3827 prev,
3828 ctx.mysql_dialect,
3829 ) == core::cmp::Ordering::Greater
3830 }
3831 };
3832 if upd {
3833 st.extreme = Some(arg_ref.clone().into_owned());
3834 }
3835 }
3836 }
3837 AggKind::Min => {
3838 if !matches!(arg_ref, Value::Null) {
3839 // v7.39 (round 626) — see the Max arm above.
3840 if !ctx.mysql_dialect && min_max_unsupported_type(arg_ref) {
3841 return Err(EvalError::TypeMismatch {
3842 detail: format!(
3843 "function min({}) does not exist",
3844 crate::conversions::pg_type_name_for_error_opt(
3845 arg_ref.data_type()
3846 )
3847 ),
3848 });
3849 }
3850 let st = &mut entry.1[i];
3851 let upd = match &st.extreme {
3852 None => true,
3853 Some(prev) => {
3854 extreme_cmp_in(
3855 spec.enum_labels.as_deref(),
3856 spec.arg_collation.as_deref(),
3857 arg_ref,
3858 prev,
3859 ctx.mysql_dialect,
3860 ) == core::cmp::Ordering::Less
3861 }
3862 };
3863 if upd {
3864 st.extreme = Some(arg_ref.clone().into_owned());
3865 }
3866 }
3867 }
3868 AggKind::AnyValue => {
3869 if !matches!(arg_ref, Value::Null) {
3870 let st = &mut entry.1[i];
3871 if st.extreme.is_none() {
3872 st.extreme = Some(arg_ref.clone().into_owned());
3873 }
3874 }
3875 }
3876 AggKind::CountStar => {
3877 entry.1[i].num.count += 1;
3878 }
3879 AggKind::Count => {
3880 if !matches!(arg_ref, Value::Null) {
3881 entry.1[i].num.count += 1;
3882 }
3883 }
3884 AggKind::BoolOr => match arg_ref {
3885 Value::Bool(b) => {
3886 let st = &mut entry.1[i];
3887 st.bool_acc = Some(st.bool_acc.unwrap_or(false) || *b);
3888 }
3889 Value::Null => {}
3890 _ => update_state(
3891 &mut entry.1[i],
3892 spec.kind,
3893 &spec.name,
3894 arg_ref,
3895 arg2_val.as_ref(),
3896 order_keys,
3897 spec.enum_labels.as_deref(),
3898 spec.arg_collation.as_deref(),
3899 ctx.mysql_dialect,
3900 )?,
3901 },
3902 AggKind::BoolAnd => match arg_ref {
3903 Value::Bool(b) => {
3904 let st = &mut entry.1[i];
3905 st.bool_acc = Some(st.bool_acc.unwrap_or(true) && *b);
3906 }
3907 Value::Null => {}
3908 _ => update_state(
3909 &mut entry.1[i],
3910 spec.kind,
3911 &spec.name,
3912 arg_ref,
3913 arg2_val.as_ref(),
3914 order_keys,
3915 spec.enum_labels.as_deref(),
3916 spec.arg_collation.as_deref(),
3917 ctx.mysql_dialect,
3918 )?,
3919 },
3920 _ => {
3921 update_state(
3922 &mut entry.1[i],
3923 spec.kind,
3924 &spec.name,
3925 arg_ref,
3926 arg2_val.as_ref(),
3927 order_keys,
3928 spec.enum_labels.as_deref(),
3929 spec.arg_collation.as_deref(),
3930 ctx.mysql_dialect,
3931 )?;
3932 }
3933 }
3934 }
3935 continue;
3936 }
3937 // v7.32 (P4 increment 2) — eval (non-bound) path: present the
3938 // row as a borrowed Row once (Owned → zero-cost borrow; a join
3939 // tuple materialises here exactly once, never on the bound fast
3940 // path above), then the original eval loop runs unchanged.
3941 let row_materialised = row.as_row();
3942 let row: &Row<'static> = &row_materialised;
3943 let group_vals: Vec<Value<'static>> = group_exprs
3944 .iter()
3945 .map(|g| eval::eval_expr(g, row, &ctx))
3946 .collect::<Result<_, _>>()?;
3947 // v7.17.0 Phase 2.5b — case-insensitive group keying: fold
3948 // only the ci columns, and only when any exist. Display
3949 // value (`group_vals`) stays original — only the key folds.
3950 let key = if ci_positions.is_empty() {
3951 encode_key(&group_vals)
3952 } else {
3953 let mut key_vals = group_vals.clone();
3954 for &i in &ci_positions {
3955 if let Value::Text(s) = &key_vals[i] {
3956 // v7.39 (round 370, M4 P4a) — a MySQL folding column
3957 // (stored CaseInsensitive) folds case AND accent; a PG
3958 // CITEXT column stays ASCII-only.
3959 key_vals[i] = Value::text(if ctx.mysql_dialect {
3960 spg_storage::mysql_compare_fold(s)
3961 } else {
3962 s.to_ascii_lowercase()
3963 });
3964 }
3965 }
3966 encode_key(&key_vals)
3967 };
3968 // Probe by index; the map owns the key once on vacant insert.
3969 let idx = match groups.get(key.as_str()) {
3970 Some(&i) => i,
3971 None => {
3972 let i = order.len();
3973 let init: Vec<AggState> =
3974 (0..agg_specs.len()).map(|_| AggState::default()).collect();
3975 order.push((group_vals.clone(), init));
3976 groups.insert(key, i);
3977 i
3978 }
3979 };
3980 let entry = &mut order[idx];
3981 for (i, spec) in agg_specs.iter().enumerate() {
3982 // v7.32 (round-29) — FILTER (WHERE cond): exclude rows where
3983 // cond is not TRUE before accumulation (and before DISTINCT).
3984 if let Some(f) = &spec.filter
3985 && !matches!(eval_arg(f, row, &ctx)?, Value::Bool(true))
3986 {
3987 continue;
3988 }
3989 let arg_val = match &spec.arg {
3990 None => Value::Bool(true), // count_star: sentinel non-null
3991 Some(e) => eval_arg(e, row, &ctx)?,
3992 };
3993 // v7.17.0 — `string_agg(value, separator)` evaluates the
3994 // separator per row. v7.39 (round 762, F31-C2) — PG uses
3995 // the PER-ROW value (element i prefixed by row i's
3996 // separator, PG18-measured `a<b>b<c>c`); update_state
3997 // records it alongside the item now (the old note claimed
3998 // PG "treats it as constant" — measured false).
3999 let arg2_val = match &spec.arg2 {
4000 None => None,
4001 Some(e) => Some(eval_arg(e, row, &ctx)?),
4002 };
4003 // v7.24 (round-16 A) — aggregate-internal ORDER BY:
4004 // evaluate the key tuple against the source row.
4005 let order_keys: Option<Vec<Value<'static>>> = if spec.order_by.is_empty() {
4006 None
4007 } else {
4008 let mut keys: Vec<Value<'static>> = Vec::with_capacity(spec.order_by.len());
4009 for o in &spec.order_by {
4010 keys.push(eval_arg(&o.expr, row, &ctx)?);
4011 }
4012 Some(keys)
4013 };
4014 // v7.33 (array_agg argmax) — first_ordered: keep the running
4015 // first-by-order element only (mirrors the bound fast path).
4016 if spec.first_ordered {
4017 if let Some(keys) = order_keys {
4018 let st = &mut entry.1[i];
4019 let better = match &st.first_best {
4020 None => true,
4021 Some((bk, _)) => {
4022 cmp_order_keys(
4023 &spec.order_by,
4024 &spec.order_enum_labels,
4025 &spec.order_collations,
4026 &keys,
4027 bk,
4028 ctx.mysql_dialect,
4029 ) == core::cmp::Ordering::Less
4030 }
4031 };
4032 if better {
4033 st.first_best = Some((keys, arg_val.clone().into_owned()));
4034 }
4035 }
4036 continue;
4037 }
4038 // v7.25 (round-17) — DISTINCT: drop repeated inputs
4039 // before they reach the accumulator. NULLs flow through
4040 // (each aggregate's own NULL rule applies; PG also
4041 // treats NULL as a single distinct value for array_agg).
4042 // v7.37.x — single-Text fast path same shape as the
4043 // bound/slow paths above.
4044 if spec.distinct {
4045 // v7.37.x (docker-fair DISTA) — single-family fast
4046 // paths skip encode_key for Text/BigInt/Int.
4047 let inserted = match &arg_val {
4048 Value::Text(s) => entry.1[i].seen.insert(s.to_string()),
4049 Value::BigInt(n) => entry.1[i]
4050 .seen_int
4051 .get_or_insert_with(BTreeSet::new)
4052 .insert(*n),
4053 Value::Int(n) => entry.1[i]
4054 .seen_int
4055 .get_or_insert_with(BTreeSet::new)
4056 .insert(i64::from(*n)),
4057 _ => {
4058 let key = encode_key(core::slice::from_ref(&arg_val));
4059 entry.1[i].seen.insert(key)
4060 }
4061 };
4062 if !inserted {
4063 continue;
4064 }
4065 }
4066 update_state(
4067 &mut entry.1[i],
4068 spec.kind,
4069 &spec.name,
4070 &arg_val,
4071 arg2_val.as_ref(),
4072 order_keys,
4073 spec.enum_labels.as_deref(),
4074 spec.arg_collation.as_deref(),
4075 ctx.mysql_dialect,
4076 )?;
4077 }
4078 }
4079 Ok(order)
4080}
4081
4082/// (2a) Build the synthetic per-group schema: `__grp_0..K` then
4083/// `__agg_0..N`. Group types are probed from the first row; aggregate
4084/// types from each spec.
4085fn build_synth_schema(
4086 rows: AggRows<'_>,
4087 group_exprs: &[Expr],
4088 agg_specs: &[AggSpec],
4089 schema_cols: &[ColumnSchema],
4090 table_alias: Option<&str>,
4091 catalog: Option<&spg_storage::Catalog>,
4092 engine: Option<&crate::Engine>,
4093) -> Result<Vec<ColumnSchema>, EvalError> {
4094 let ctx = with_catalog(EvalContext::new(schema_cols, table_alias), catalog, engine);
4095 // Build synthetic schema: __grp_0..K then __agg_0..N.
4096 let group_types: Vec<DataType> = if rows.is_empty() {
4097 // Use Text as a safe stand-in — empty result means schema isn't
4098 // observable. Avoids needing to evaluate group exprs on no row.
4099 group_exprs.iter().map(|_| DataType::Text).collect()
4100 } else {
4101 let probe = rows.get(0).expect("non-empty checked above");
4102 let probe_row = probe.as_row();
4103 let probe: &Row<'static> = &probe_row;
4104 group_exprs
4105 .iter()
4106 .map(|g| {
4107 eval::eval_expr(g, probe, &ctx).map(|v| v.data_type().unwrap_or(DataType::Text))
4108 })
4109 .collect::<Result<_, _>>()?
4110 };
4111 let agg_types: Vec<DataType> = agg_specs
4112 .iter()
4113 .map(|spec| infer_agg_type(spec, schema_cols))
4114 .collect();
4115 let mut synth_schema: Vec<ColumnSchema> = Vec::new();
4116 for (i, ty) in group_types.iter().enumerate() {
4117 let mut col = ColumnSchema::new(format!("__grp_{i}"), *ty, true);
4118 // v7.39 (enum order knife) — a bare enum-column group key keeps
4119 // its enum identity so HAVING comparisons and the grouped-output
4120 // ORDER BY sort by member order downstream.
4121 if let Some(Expr::Column(c)) = group_exprs.get(i) {
4122 let src = schema_cols.iter().find(|sc| sc.name == c.name);
4123 col.user_enum_type = src.and_then(|sc| sc.user_enum_type.clone());
4124 // v7.39 (round 686) — and its collation, for the same reason and
4125 // by the same route. A `__grp_j` column is where a GROUP BY key
4126 // lives from here on, so anything the downstream ORDER BY needs
4127 // about the original column has to travel with it. Without this
4128 // the resolver looks the key up in the synthetic schema, finds
4129 // `__grp_0` with no collation, and the group-by ordering silently
4130 // stays byte-wise.
4131 col.collation_name = src.and_then(|sc| sc.collation_name.clone());
4132 // v7.38.14 — and the collation ENUM, which is a different field
4133 // and the one every MySQL text comparison actually reads. The
4134 // note above carried the NAME and stopped, exactly as round 688
4135 // did in `join.rs::build_combined_schema`; both left the enum
4136 // behind, and `ColumnSchema::new` defaults it to `Binary`, which
4137 // downstream reads as "byte-wise ON PURPOSE" rather than as
4138 // "unknown". So a `__grp_j` column claimed to be an explicit
4139 // binary column and `SELECT DISTINCT ... GROUP BY` stopped
4140 // folding. Sixth field through this hole, second site with the
4141 // identical shape.
4142 if let Some(sc) = src {
4143 col.collation = sc.collation;
4144 }
4145 }
4146 synth_schema.push(col);
4147 }
4148 for (i, ty) in agg_types.iter().enumerate() {
4149 synth_schema.push(ColumnSchema::new(format!("__agg_{i}"), *ty, true));
4150 }
4151 Ok(synth_schema)
4152}
4153
4154/// (2b) Materialise one synthetic row per group (insertion order):
4155/// apply each aggregate's internal ORDER BY, then finalise the running
4156/// state into the group + aggregate cells.
4157/// v7.33 — compare two aggregate-internal ORDER BY key tuples under the
4158/// per-key DESC / NULLS directives. This is the exact comparator the
4159/// finalize sort uses, factored out so the `first_ordered` argmax
4160/// accumulator's "keep first" decision is provably identical to taking
4161/// element `[1]` of the fully-sorted array.
4162fn cmp_order_keys(
4163 order_by: &[spg_sql::ast::OrderBy],
4164 order_enum_labels: &[Option<Vec<String>>],
4165 order_collations: &[Option<alloc::string::String>],
4166 a: &[Value<'static>],
4167 b: &[Value<'static>],
4168 mysql: bool,
4169) -> core::cmp::Ordering {
4170 for (k, o) in order_by.iter().enumerate() {
4171 // v7.39 (enum order knife) — an enum-typed sort key compares by
4172 // member order; NULLs and non-members keep the generic path.
4173 if let Some(Some(labels)) = order_enum_labels.get(k)
4174 && !matches!(&a[k], Value::Null)
4175 && !matches!(&b[k], Value::Null)
4176 && let Some(ord) = crate::eval::enum_ord_cmp(labels, &a[k], &b[k])
4177 {
4178 let ord = if o.desc { ord.reverse() } else { ord };
4179 if ord != core::cmp::Ordering::Equal {
4180 return ord;
4181 }
4182 continue;
4183 }
4184 // v7.37 (M4 P2) — `ORDER BY BINARY x` forces byte-wise sorting
4185 // even under the folding MySQL dialect, so a per-key BINARY
4186 // coercion turns folding back off for that key alone.
4187 let fold = mysql && !crate::eval::is_binary_coerced(&o.expr);
4188 // v7.38.18 — the key's declared collation, so the sort inside an
4189 // aggregate orders a collated column the way the statement's own
4190 // ORDER BY orders it.
4191 let coll = order_collations.get(k).and_then(Option::as_deref);
4192 let cmp = crate::orderby::order_by_value_cmp_coll(
4193 o.desc,
4194 o.nulls_first,
4195 &a[k],
4196 &b[k],
4197 fold,
4198 coll,
4199 );
4200 if cmp != core::cmp::Ordering::Equal {
4201 return cmp;
4202 }
4203 }
4204 core::cmp::Ordering::Equal
4205}
4206
4207#[allow(clippy::too_many_arguments)]
4208fn finalize_synth_rows(
4209 order: &[(Vec<Value<'static>>, Vec<AggState>)],
4210 agg_specs: &[AggSpec],
4211 synth_schema: &[ColumnSchema],
4212 rows: AggRows<'_>,
4213 schema_cols: &[ColumnSchema],
4214 table_alias: Option<&str>,
4215 catalog: Option<&spg_storage::Catalog>,
4216 engine: Option<&crate::Engine>,
4217 runner: Option<&dyn crate::ParallelRunner>,
4218) -> Result<Vec<Row<'static>>, EvalError> {
4219 let ctx = with_catalog(EvalContext::new(schema_cols, table_alias), catalog, engine);
4220 // v7.39 (round 747) — GROUP-parallel finalize for the collection
4221 // aggregates. `string_agg(s, ',' ORDER BY id) GROUP BY g` sorted
4222 // and joined every group's items serially — the panel's last
4223 // >=2.0x cell. Groups are independent; shards produce their row
4224 // ranges in group order and concatenate. Admission: every spec a
4225 // collection kind (their finalize reads items/keys/separator and
4226 // the dialect only — nothing that needs the engine hook), no
4227 // ordered-set / first_ordered / regression shapes.
4228 let collections_only = agg_specs.iter().all(|s| {
4229 matches!(
4230 classify_agg_name(&s.name),
4231 AggKind::StringAgg | AggKind::ArrayAgg | AggKind::JsonAgg
4232 ) && !s.first_ordered
4233 && !is_within_group_name(&s.name)
4234 });
4235 if collections_only
4236 && order.len() >= 16
4237 && let Some(r) = runner
4238 {
4239 let group_len_probe = order.first().map(|(g, _)| g.len()).unwrap_or(0);
4240 let _ = group_len_probe;
4241 let n_shards = (order.len() / 8).clamp(2, 8);
4242 let chunk = order.len().div_ceil(n_shards);
4243 type ShardOut = Result<Vec<Row<'static>>, EvalError>;
4244 let mysql = ctx.mysql_dialect;
4245 let style = ctx.render_style;
4246 let results = r.run_shards(n_shards, &|si| {
4247 let lo = si * chunk;
4248 let hi = ((si + 1) * chunk).min(order.len());
4249 let mut sctx = EvalContext::new(schema_cols, table_alias);
4250 sctx.mysql_dialect = mysql;
4251 sctx.render_style = style;
4252 let run = || -> ShardOut {
4253 let mut out: Vec<Row<'static>> = Vec::with_capacity(hi - lo);
4254 for (gvals, states) in &order[lo..hi] {
4255 out.push(finalize_one_group(
4256 gvals,
4257 states,
4258 agg_specs,
4259 synth_schema,
4260 &sctx,
4261 )?);
4262 }
4263 Ok(out)
4264 };
4265 alloc::boxed::Box::new(run())
4266 });
4267 let mut synth_rows: Vec<Row<'static>> = Vec::with_capacity(order.len());
4268 for boxed in results {
4269 let shard = boxed
4270 .downcast::<ShardOut>()
4271 .expect("runner echoes the closure's box");
4272 synth_rows.extend((*shard)?);
4273 }
4274 return Ok(synth_rows);
4275 }
4276 // v7.32 (round-29) — ordered-set direct arguments (the percentile
4277 // fraction) are constant per PG, so evaluate each once up front.
4278 let direct_arg_vals: Vec<Option<Value>> = agg_specs
4279 .iter()
4280 .map(|spec| match (&spec.direct_arg, rows.first().as_ref()) {
4281 (Some(e), Some(r)) => eval::eval_expr(e, &r.as_row(), &ctx).map(Some),
4282 _ => Ok(None),
4283 })
4284 .collect::<Result<_, _>>()?;
4285 // v7.39 (read01 orderedsetaggs.c) — the remaining hypothetical direct
4286 // arguments of a multi-key call, evaluated once like the first.
4287 let direct_extra_vals: Vec<Vec<Value>> = agg_specs
4288 .iter()
4289 .map(|spec| match rows.first().as_ref() {
4290 Some(r) if !spec.direct_args_extra.is_empty() => spec
4291 .direct_args_extra
4292 .iter()
4293 .map(|e| eval::eval_expr(e, &r.as_row(), &ctx))
4294 .collect(),
4295 _ => Ok(Vec::new()),
4296 })
4297 .collect::<Result<_, _>>()?;
4298
4299 // Materialise synthetic rows (insertion order = `order`).
4300 let mut synth_rows: Vec<Row<'static>> = Vec::new();
4301 for (gvals, states) in order {
4302 let mut values: Vec<Value<'static>> = Vec::with_capacity(synth_schema.len());
4303 // The synth schema is [group keys…, aggregates…]; the aggregate at
4304 // index `i` therefore sits at `group_len + i`.
4305 let group_len = gvals.len();
4306 values.extend(gvals.iter().cloned());
4307 for (i, st) in states.iter().enumerate() {
4308 // v7.33 (array_agg argmax) — first_ordered: the running
4309 // first-by-order value IS the result; no array build/sort.
4310 if agg_specs[i].first_ordered {
4311 values.push(
4312 st.first_best
4313 .as_ref()
4314 .map_or(Value::Null, |(_, v)| v.clone()),
4315 );
4316 continue;
4317 }
4318 // v7.24 (round-16 A) — order the collected items per the
4319 // aggregate-internal ORDER BY before finalize consumes
4320 // them.
4321 let st_sorted;
4322 let kw = agg_specs[i].order_by.len();
4323 let st_final: &AggState = if kw > 0 && st.item_keys.len() == st.items.len() * kw {
4324 let mut idx: Vec<usize> = (0..st.items.len()).collect();
4325 let ob = &agg_specs[i].order_by;
4326 idx.sort_by(|&x, &y| {
4327 cmp_order_keys(
4328 ob,
4329 &agg_specs[i].order_enum_labels,
4330 &agg_specs[i].order_collations,
4331 &st.item_keys[x * kw..(x + 1) * kw],
4332 &st.item_keys[y * kw..(y + 1) * kw],
4333 ctx.mysql_dialect,
4334 )
4335 });
4336 // Permute by MOVE out of the clone — the old form
4337 // cloned every item a second time on top of
4338 // `st.clone()`'s first (5000 Strings twice per group).
4339 let mut sorted = st.clone();
4340 let mut new_items: Vec<Value<'static>> = Vec::with_capacity(idx.len());
4341 for &j in &idx {
4342 new_items.push(core::mem::replace(&mut sorted.items[j], Value::Null));
4343 }
4344 // v7.39 (round 762, F31-C2) — the per-row separators
4345 // travel with their items through the sort.
4346 if sorted.item_seps.len() == sorted.items.len() {
4347 let mut new_seps: Vec<Option<String>> = Vec::with_capacity(idx.len());
4348 for &j in &idx {
4349 new_seps.push(core::mem::take(&mut sorted.item_seps[j]));
4350 }
4351 sorted.item_seps = new_seps;
4352 }
4353 sorted.items = new_items;
4354 st_sorted = sorted;
4355 &st_sorted
4356 } else if agg_specs[i].distinct && st.items.len() > 1 {
4357 // v7.39 (round 257) — PG dedups a DISTINCT aggregate by
4358 // SORTING its input, so the collection aggregates emit
4359 // their values in sort order (probed across array_agg /
4360 // string_agg / json_agg, ints and text, NULLs last):
4361 // `array_agg(DISTINCT x)` over 2,1,2 is `{1,2}`, where
4362 // SPG kept first-seen order and answered `{2,1}`. An
4363 // explicit ORDER BY takes the branch above instead, and
4364 // the scalar aggregates (count / sum / …) are
4365 // order-insensitive, so this only moves the collections.
4366 // v7.39 (round 258) — an ENUM input sorts by MEMBER
4367 // ORDER, not by its text (`{sad,ok,happy}`, not
4368 // `{happy,ok,sad}`); `spec.enum_labels` already
4369 // carries the aggregate argument's labels for exactly
4370 // this. Round 257 shipped this sort with the generic
4371 // value comparison and regressed enum columns.
4372 let labels = agg_specs[i].enum_labels.as_deref();
4373 let mut sorted = st.clone();
4374 // v7.39 (round 762, F31-C2) — DISTINCT re-sorts items
4375 // alone; per-row separators cannot follow, so the
4376 // constant-separator path applies (the last row's).
4377 sorted.item_seps.clear();
4378 sorted.items.sort_by(|a, b| {
4379 if let Some(labels) = labels
4380 && !matches!(a, Value::Null)
4381 && !matches!(b, Value::Null)
4382 && let Some(ord) = crate::eval::enum_ord_cmp(labels, a, b)
4383 {
4384 return ord;
4385 }
4386 crate::order_by_value_cmp_in(false, Some(false), a, b, ctx.mysql_dialect)
4387 });
4388 st_sorted = sorted;
4389 &st_sorted
4390 } else {
4391 st
4392 };
4393 // Ordered-set aggregates compute from the sorted items + the
4394 // direct fraction; everything else uses the running state.
4395 let v = if is_within_group_name(&agg_specs[i].name) {
4396 finalize_ordered_set(
4397 &agg_specs[i].name,
4398 st_final,
4399 direct_arg_vals[i].as_ref(),
4400 &direct_extra_vals[i],
4401 &agg_specs[i].order_by,
4402 &agg_specs[i].order_collations,
4403 ctx.mysql_dialect,
4404 )?
4405 } else {
4406 finalize(&agg_specs[i].name, st_final, ctx.mysql_dialect)
4407 };
4408 // v7.39 (round 327, V44) — keep the zone identity. SPG carries a
4409 // timestamptz at runtime as `Value::Timestamp`, so the array
4410 // `array_agg` builds is a `TimestampArray` and `pg_typeof`
4411 // answered `timestamp without time zone[]` for
4412 // `array_agg(timestamptz_col)`. The STATIC type in the synth
4413 // schema already knows better (`infer_agg_type` maps
4414 // Timestamptz ⇒ TimestamptzArray); re-tag the value to match
4415 // it. Third code path in this family — V31 fixed the array
4416 // constructor, V43 the literal cast.
4417 let v = match (v, synth_schema.get(group_len + i).map(|c| c.ty)) {
4418 (Value::TimestampArray(items), Some(DataType::TimestamptzArray)) => {
4419 Value::TimestamptzArray(items)
4420 }
4421 (v, _) => v,
4422 };
4423 values.push(v);
4424 }
4425 synth_rows.push(Row::new(values));
4426 }
4427 Ok(synth_rows)
4428}
4429
4430/// v7.39 (round 747) — one group's synth row for the COLLECTION
4431/// aggregates (string_agg / array_agg / json_agg): the ordered/distinct
4432/// sort branches verbatim from the serial loop, then `finalize`. The
4433/// group-parallel path calls this; admission guarantees no
4434/// first_ordered / within-group / timestamptz-retag shapes reach it
4435/// (json/array of timestamptz retag is still applied for safety).
4436fn finalize_one_group(
4437 gvals: &[Value<'static>],
4438 states: &[AggState],
4439 agg_specs: &[AggSpec],
4440 synth_schema: &[ColumnSchema],
4441 ctx: &EvalContext<'_>,
4442) -> Result<Row<'static>, EvalError> {
4443 let group_len = gvals.len();
4444 let mut values: Vec<Value<'static>> = Vec::with_capacity(synth_schema.len());
4445 values.extend(gvals.iter().cloned());
4446 for (i, st) in states.iter().enumerate() {
4447 let st_sorted;
4448 let kw = agg_specs[i].order_by.len();
4449 let st_final: &AggState = if kw > 0 && st.item_keys.len() == st.items.len() * kw {
4450 let mut idx: Vec<usize> = (0..st.items.len()).collect();
4451 let ob = &agg_specs[i].order_by;
4452 idx.sort_by(|&x, &y| {
4453 cmp_order_keys(
4454 ob,
4455 &agg_specs[i].order_enum_labels,
4456 &agg_specs[i].order_collations,
4457 &st.item_keys[x * kw..(x + 1) * kw],
4458 &st.item_keys[y * kw..(y + 1) * kw],
4459 ctx.mysql_dialect,
4460 )
4461 });
4462 let mut sorted = st.clone();
4463 let mut new_items: Vec<Value<'static>> = Vec::with_capacity(idx.len());
4464 for &j in &idx {
4465 new_items.push(core::mem::replace(&mut sorted.items[j], Value::Null));
4466 }
4467 // v7.39 (round 762, F31-C2) — separators travel with items.
4468 if sorted.item_seps.len() == sorted.items.len() {
4469 let mut new_seps: Vec<Option<String>> = Vec::with_capacity(idx.len());
4470 for &j in &idx {
4471 new_seps.push(core::mem::take(&mut sorted.item_seps[j]));
4472 }
4473 sorted.item_seps = new_seps;
4474 }
4475 sorted.items = new_items;
4476 st_sorted = sorted;
4477 &st_sorted
4478 } else if agg_specs[i].distinct && st.items.len() > 1 {
4479 let labels = agg_specs[i].enum_labels.as_deref();
4480 let mut sorted = st.clone();
4481 // v7.39 (round 762, F31-C2) — see the sibling branch above.
4482 sorted.item_seps.clear();
4483 sorted.items.sort_by(|a, b| {
4484 if let Some(labels) = labels
4485 && !matches!(a, Value::Null)
4486 && !matches!(b, Value::Null)
4487 && let Some(ord) = crate::eval::enum_ord_cmp(labels, a, b)
4488 {
4489 return ord;
4490 }
4491 crate::order_by_value_cmp_in(false, Some(false), a, b, ctx.mysql_dialect)
4492 });
4493 st_sorted = sorted;
4494 &st_sorted
4495 } else {
4496 st
4497 };
4498 let v = finalize(&agg_specs[i].name, st_final, ctx.mysql_dialect);
4499 let v = match (v, synth_schema.get(group_len + i).map(|c| c.ty)) {
4500 (Value::TimestampArray(items), Some(DataType::TimestamptzArray)) => {
4501 Value::TimestamptzArray(items)
4502 }
4503 (v, _) => v,
4504 };
4505 values.push(v);
4506 }
4507 Ok(Row::new(values))
4508}
4509
4510/// (3) Rewrite the user's SELECT items + HAVING to reference the
4511/// synthetic columns, filter groups by HAVING, and project each
4512/// surviving group into an output row. The synth rows ride alongside
4513/// (`kept_synth`) so post-LIMIT deferred subqueries can evaluate later.
4514#[allow(clippy::too_many_lines)]
4515fn project_groups(
4516 synth_rows: Vec<Row<'static>>,
4517 stmt: &SelectStatement,
4518 group_exprs: &[Expr],
4519 agg_specs: &[AggSpec],
4520 synth_schema: &[ColumnSchema],
4521 correlated_eval: Option<CorrelatedEval<'_>>,
4522 defer_projection: bool,
4523 catalog: Option<&spg_storage::Catalog>,
4524 mysql: bool,
4525) -> Result<Projection, EvalError> {
4526 // Rewrite the user's SELECT items + ORDER BY to reference synthetic
4527 // columns. After rewriting, every remaining `Expr::Column` must
4528 // resolve against the synthetic schema (i.e. must have been a GROUP
4529 // BY expression).
4530 let columns: Vec<ColumnSchema> = stmt
4531 .items
4532 .iter()
4533 .map(|item| match item {
4534 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => {
4535 Err(EvalError::TypeMismatch {
4536 detail: "SELECT * with aggregates is not supported".into(),
4537 })
4538 }
4539 SelectItem::Expr { expr, alias } => {
4540 let rewritten = rewrite_expr(expr, group_exprs, agg_specs);
4541 let name = alias
4542 .clone()
4543 .unwrap_or_else(|| crate::select::default_output_name(expr, mysql));
4544 // v7.38.14 — the type is looked up in the synthetic schema
4545 // here; the COLLATION has to travel by the same route or the
4546 // output column claims `ColumnSchema::new`'s default, which
4547 // is `Binary` and reads downstream as "byte-wise on
4548 // purpose". That is what made `SELECT DISTINCT ... GROUP BY`
4549 // stop folding: the de-duplication asked the output schema
4550 // and the output schema had forgotten.
4551 //
4552 // Third site with this exact shape in one release, after
4553 // `join.rs::build_combined_schema` and `synth_schema` above.
4554 // Each one hand-picks which attributes survive; none picks
4555 // all of them. See S4 of the v7.38.14 roadmap.
4556 let mut col =
4557 ColumnSchema::new(name, agg_or_group_type(&rewritten, synth_schema), true);
4558 if let Expr::Column(c) = &rewritten
4559 && let Some(sc) = synth_schema
4560 .iter()
4561 .find(|sc| sc.name.eq_ignore_ascii_case(&c.name))
4562 {
4563 col.collation = sc.collation;
4564 col.collation_name.clone_from(&sc.collation_name);
4565 }
4566 Ok(col)
4567 }
4568 })
4569 .collect::<Result<_, _>>()?;
4570
4571 // Project per synthetic row. HAVING filters out groups *before*
4572 // we keep the projected row — same semantics as PG: HAVING runs
4573 // against the aggregated row (so `HAVING count(*) > 1` works) and
4574 // sees only group-by'd columns plus aggregate values.
4575 let mut synth_ctx = EvalContext::new(synth_schema, None);
4576 // v7.39 (enum order knife) — HAVING comparisons over enum group keys
4577 // need the catalog for member-order semantics (both the compile-time
4578 // Subtree fallback witness and the eval hook read it).
4579 if let Some(cat) = catalog {
4580 synth_ctx = synth_ctx.with_catalog(cat);
4581 }
4582 // v7.39 (round 404) — a MySQL session lets HAVING name a SELECT alias.
4583 // Build the (alias, expr) map from renaming SELECT items, then subst
4584 // before the aggregate rewrite.
4585 let having_aliases: Vec<(String, Expr)> = if mysql {
4586 stmt.items
4587 .iter()
4588 .filter_map(|it| match it {
4589 SelectItem::Expr {
4590 expr,
4591 alias: Some(a),
4592 } if !matches!(expr, Expr::Column(c)
4593 if c.qualifier.is_none() && c.name.eq_ignore_ascii_case(a)) =>
4594 {
4595 Some((a.clone(), expr.clone()))
4596 }
4597 _ => None,
4598 })
4599 .collect()
4600 } else {
4601 Vec::new()
4602 };
4603 let having_rewritten = stmt.having.as_ref().map(|h| {
4604 let h = if having_aliases.is_empty() {
4605 h.clone()
4606 } else {
4607 substitute_having_aliases(h.clone(), &having_aliases)
4608 };
4609 rewrite_expr(&h, group_exprs, agg_specs)
4610 });
4611 // v7.30 (phase 3e-1) - rewrite SELECT items ONCE. This ran per
4612 // GROUP (23.5k x 9 items of AST cloning = ~48% of the inbox
4613 // query in sampled stacks); the rewrite is group-independent.
4614 // Stable addresses also let the per-expression subquery plans
4615 // (v7.29 3c) hit across groups instead of rebuilding.
4616 let items_rewritten: alloc::vec::Vec<Option<Expr>> = stmt
4617 .items
4618 .iter()
4619 .map(|item| match item {
4620 SelectItem::Expr { expr, .. } => Some(rewrite_expr(expr, group_exprs, agg_specs)),
4621 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => None,
4622 })
4623 .collect();
4624 // v7.31 (perf — PG lesson #1): subquery-bearing select items
4625 // deferred to post-LIMIT, when no sort/filter key can observe
4626 // them. ORDER BY rewrites are hoisted here so the safety check
4627 // and the sort below share one rewrite pass.
4628 let order_rewritten: Vec<Expr> = stmt
4629 .order_by
4630 .iter()
4631 .map(|o| rewrite_expr(&o.expr, group_exprs, agg_specs))
4632 .collect();
4633 let defer_enabled = correlated_eval.is_some()
4634 && !stmt.distinct
4635 && !having_rewritten
4636 .as_ref()
4637 .is_some_and(crate::expr_has_subquery)
4638 && !order_rewritten.iter().any(crate::expr_has_subquery);
4639 let deferred: Vec<(usize, Expr)> = if defer_enabled {
4640 items_rewritten
4641 .iter()
4642 .enumerate()
4643 .filter_map(|(i, r)| {
4644 r.as_ref()
4645 .filter(|e| crate::expr_has_subquery(e))
4646 .map(|e| (i, e.clone()))
4647 })
4648 .collect()
4649 } else {
4650 Vec::new()
4651 };
4652 // v7.32 (architecture v2, P2) — compile the per-group synth-row
4653 // expressions ONCE. The projection / HAVING here run per GROUP
4654 // (24k for the inbox shape) × per item; the rewritten exprs are
4655 // mostly `Column(__agg_N)` / `Column(__grp_K)` against the synth
4656 // schema — flat step programs, no tree walk per group.
4657 let having_compiled = having_rewritten
4658 .as_ref()
4659 .filter(|h| eval::fully_compilable(h))
4660 .map(|h| eval::compile_expr(h, &synth_ctx));
4661 let items_compiled: Vec<Option<eval::CompiledExpr>> = items_rewritten
4662 .iter()
4663 .enumerate()
4664 .map(|(i, r)| {
4665 r.as_ref()
4666 .filter(|e| !deferred.iter().any(|(c, _)| *c == i) && eval::fully_compilable(e))
4667 .map(|e| eval::compile_expr(e, &synth_ctx))
4668 })
4669 .collect();
4670 // v7.39 (round 621) — which items are set-returning, after the rewrite
4671 // (so `unnest(array_agg(x))` is seen as the SRF it is, over a synthetic
4672 // aggregate column). Only the builtin SRFs are recognised here; a user
4673 // `RETURNS SETOF` function inside an aggregate query keeps the old error,
4674 // because running its body needs the executor and this is not it.
4675 let srf_items: Vec<bool> = items_rewritten
4676 .iter()
4677 .map(|r| {
4678 r.as_ref()
4679 .is_some_and(|e| crate::select::top_level_srf_kind(e).is_some())
4680 })
4681 .collect();
4682 let any_srf = srf_items.iter().any(|b| *b);
4683 let mut kept_synth: Vec<Row<'static>> = Vec::new();
4684 let mut out_rows: Vec<Row<'static>> = Vec::new();
4685 let mut stack: Vec<Value<'static>> = Vec::new();
4686 for srow in synth_rows {
4687 if let Some(hc) = &having_compiled {
4688 let cond = eval::eval_compiled(hc, &srow, &synth_ctx, &mut stack)?;
4689 if !crate::eval::predicate_is_true(&cond, "HAVING", synth_ctx.mysql_dialect)? {
4690 continue;
4691 }
4692 } else if let Some(h) = &having_rewritten {
4693 let cond = match correlated_eval {
4694 Some(f) if crate::expr_has_subquery(h) => f(h, &srow, &synth_ctx)?,
4695 _ => eval::eval_expr(h, &srow, &synth_ctx)?,
4696 };
4697 if !crate::eval::predicate_is_true(&cond, "HAVING", synth_ctx.mysql_dialect)? {
4698 continue;
4699 }
4700 }
4701 // v7.37.x — when caller pre-truncates via ORDER BY+LIMIT, skip
4702 // per-item projection here; the caller fills the placeholder
4703 // out_rows from the top-K survivors below.
4704 if defer_projection {
4705 kept_synth.push(srow);
4706 out_rows.push(Row::new(Vec::new()));
4707 continue;
4708 }
4709 let mut values: Vec<Value<'static>> = Vec::with_capacity(columns.len());
4710 for (i, rewritten) in items_rewritten.iter().enumerate() {
4711 let Some(rewritten) = rewritten else { continue };
4712 if deferred.iter().any(|(c, _)| *c == i) {
4713 values.push(Value::Null);
4714 continue;
4715 }
4716 // v7.39 (round 621) — a SET-RETURNING item is collected as its
4717 // whole list; the rows it makes are built after the loop.
4718 if srf_items[i] {
4719 values.push(Value::Null);
4720 continue;
4721 }
4722 values.push(if let Some(cc) = &items_compiled[i] {
4723 eval::eval_compiled(cc, &srow, &synth_ctx, &mut stack)?
4724 } else {
4725 match correlated_eval {
4726 Some(f) if crate::expr_has_subquery(rewritten) => {
4727 f(rewritten, &srow, &synth_ctx)?
4728 }
4729 _ => eval::eval_expr(rewritten, &srow, &synth_ctx)?,
4730 }
4731 });
4732 }
4733 if any_srf {
4734 // v7.39 (round 621) — the aggregate's own output row is what a
4735 // target-list SRF expands over. `SELECT unnest(ARRAY[1,2]),
4736 // count(*) FROM t` answered `function unnest(integer[]) does not
4737 // exist`, because this projection evaluates each item scalarly and
4738 // there is exactly one row per group to put it in. PG answers two
4739 // rows, both carrying the same count — and the shape that matters
4740 // most is `unnest(array_agg(x))`, where the SRF's ARGUMENT is the
4741 // aggregate.
4742 //
4743 // Several SRFs in one list expand in LOCKSTEP with the shorter
4744 // padded to NULL, which is round 67's rule for every other path.
4745 let mut lists: Vec<Vec<Value<'static>>> = Vec::with_capacity(items_rewritten.len());
4746 for (i, rewritten) in items_rewritten.iter().enumerate() {
4747 match (srf_items[i], rewritten) {
4748 (true, Some(r)) => {
4749 lists.push(
4750 crate::select::top_level_srf_output(r, &srow, &synth_ctx).map_err(
4751 |e| match e {
4752 crate::EngineError::Eval(ev) => ev,
4753 other => EvalError::TypeMismatch {
4754 detail: alloc::format!("{other}"),
4755 },
4756 },
4757 )?,
4758 );
4759 }
4760 _ => lists.push(Vec::new()),
4761 }
4762 }
4763 let n = lists.iter().map(Vec::len).max().unwrap_or(0);
4764 for k in 0..n {
4765 let mut vals = values.clone();
4766 for (i, list) in lists.iter().enumerate() {
4767 if srf_items[i]
4768 && let Some(slot) = vals.get_mut(i)
4769 {
4770 *slot = list.get(k).cloned().unwrap_or(Value::Null);
4771 }
4772 }
4773 kept_synth.push(srow.clone());
4774 out_rows.push(Row::new(vals));
4775 }
4776 continue;
4777 }
4778 kept_synth.push(srow);
4779 out_rows.push(Row::new(values));
4780 }
4781 let deferred_project_state = if defer_projection {
4782 Some(DeferredProject {
4783 items_rewritten,
4784 items_compiled,
4785 })
4786 } else {
4787 None
4788 };
4789 Ok(Projection {
4790 columns,
4791 out_rows,
4792 kept_synth,
4793 deferred,
4794 order_rewritten,
4795 deferred_project: deferred_project_state,
4796 })
4797}
4798
4799/// (4) Sort the projected output by the rewritten ORDER BY keys. The
4800/// synth rows ride through the sort so deferred subqueries evaluate
4801/// against the surviving groups after the caller's LIMIT truncation.
4802fn sort_synth_by_order_by(
4803 synth_schema: &[ColumnSchema],
4804 out_columns: &[ColumnSchema],
4805 order_by: &[spg_sql::ast::OrderBy],
4806 order_rewritten: &[Expr],
4807 mut kept_synth: Vec<Row<'static>>,
4808 mut out_rows: Vec<Row<'static>>,
4809 correlated_eval: Option<CorrelatedEval<'_>>,
4810 keep_n: Option<usize>,
4811 catalog: Option<&spg_storage::Catalog>,
4812 mysql: bool,
4813) -> Result<(Vec<Row<'static>>, Vec<Row<'static>>), EvalError> {
4814 let mut synth_ctx = EvalContext::new(synth_schema, None);
4815 if let Some(cat) = catalog {
4816 synth_ctx = synth_ctx.with_catalog(cat);
4817 }
4818 // v7.39 (enum order knife) — per-key member labels when the rewritten
4819 // sort key is an enum-typed column (`__grp_K` carrying user_enum_type).
4820 let key_enum_labels: Vec<Option<&[String]>> = order_rewritten
4821 .iter()
4822 .map(|e| crate::eval::expr_enum_labels(e, synth_schema, catalog))
4823 .collect();
4824 // v7.39 (round 686) — per-key declared collation, built exactly like the
4825 // enum labels above because it is the same kind of thing: metadata the
4826 // comparator needs, resolved once per sort from the key expression.
4827 //
4828 // Located by forcing this call site to reverse and watching
4829 // `GROUP BY loc ORDER BY loc` flip. Rounds 682 and 685 wired eleven
4830 // sites between them without doing that, and none was on the path.
4831 let key_colls: Vec<Option<alloc::string::String>> = order_rewritten
4832 .iter()
4833 .map(|e| {
4834 let spg_sql::ast::Expr::Column(c) = e else {
4835 return None;
4836 };
4837 let pos = crate::eval::find_column_pos(c, &synth_ctx)?;
4838 let name = synth_schema.get(pos)?.collation_name.clone()?;
4839 crate::collate::is_supported(&name).then_some(name)
4840 })
4841 .collect();
4842 // v6.4.0 — multi-key ORDER BY on aggregate output. Each key
4843 // gets its own rewrite + per-key DESC flag. (Rewrites hoisted
4844 // above as `order_rewritten` — shared with the deferral
4845 // safety check.)
4846 let keys_meta: Vec<(bool, Option<bool>)> =
4847 order_by.iter().map(|o| (o.desc, o.nulls_first)).collect();
4848 // P2: compile order-by keys once (per-group sort keys are
4849 // the same `__agg_N` / `__grp_K` shape as the projection).
4850 let order_compiled: Vec<Option<eval::CompiledExpr>> = order_rewritten
4851 .iter()
4852 .map(|e| {
4853 Some(e)
4854 .filter(|e| eval::fully_compilable(e))
4855 .map(|e| eval::compile_expr(e, &synth_ctx))
4856 })
4857 .collect();
4858 // The synth row rides through the sort so deferred exprs can
4859 // evaluate against the surviving groups after the caller's
4860 // LIMIT truncation.
4861 // v7.37 (round 1000) — a sort key that names an OUTPUT column.
4862 //
4863 // `ORDER BY 1` over a set-returning item does not substitute the
4864 // item's expression: round 80 resolved it to the item's output NAME
4865 // instead, because a positional key means the Nth OUTPUT column and
4866 // substituting the expression would make the key "the whole set",
4867 // evaluated once per group, which silently sorted nothing. The
4868 // non-aggregate paths then evaluate that name against the output
4869 // schema.
4870 //
4871 // This one evaluated it against the SYNTHETIC schema, which carries
4872 // `__agg_N` / `__grp_K` and no output aliases, so
4873 // `SELECT unnest(ARRAY[1,2]) AS u, count(*) … GROUP BY g ORDER BY 1`
4874 // answered `column "u" does not exist` — a query PG18.4 answers.
4875 // Spelling it `ORDER BY u` failed differently and for the same
4876 // reason: the alias resolved to the expression, and a set-returning
4877 // call cannot be evaluated scalarly on a group row.
4878 //
4879 // So: a key that names an output column and NOTHING in the synthetic
4880 // schema is read from the projected row, where expansion has already
4881 // put the per-row value. Synthetic names keep precedence, so nothing
4882 // that resolved before resolves differently now.
4883 let out_key_idx: Vec<Option<usize>> = order_rewritten
4884 .iter()
4885 .map(|e| {
4886 let spg_sql::ast::Expr::Column(c) = e else {
4887 return None;
4888 };
4889 if c.qualifier.is_some() || crate::eval::find_column_pos(c, &synth_ctx).is_some() {
4890 return None;
4891 }
4892 out_columns
4893 .iter()
4894 .position(|oc| oc.name.eq_ignore_ascii_case(&c.name))
4895 })
4896 .collect();
4897 let mut keystack: Vec<Value<'static>> = Vec::new();
4898 let mut tagged: Vec<(Vec<Value<'static>>, Row, Row)> = Vec::with_capacity(kept_synth.len());
4899 for (s, o) in kept_synth.into_iter().zip(out_rows) {
4900 let mut keys = Vec::with_capacity(order_rewritten.len());
4901 for (i, (e, oc)) in order_rewritten.iter().zip(&order_compiled).enumerate() {
4902 if let Some(oi) = out_key_idx[i] {
4903 keys.push(o.values.get(oi).cloned().unwrap_or(Value::Null));
4904 continue;
4905 }
4906 keys.push(if let Some(oc) = oc {
4907 eval::eval_compiled(oc, &s, &synth_ctx, &mut keystack)?
4908 } else {
4909 match correlated_eval {
4910 Some(f) if crate::expr_has_subquery(e) => f(e, &s, &synth_ctx)?,
4911 _ => eval::eval_expr(e, &s, &synth_ctx)?,
4912 }
4913 });
4914 }
4915 tagged.push((keys, s, o));
4916 }
4917 let cmp = |a: &(Vec<Value<'static>>, Row, Row), b: &(Vec<Value<'static>>, Row, Row)| {
4918 use core::cmp::Ordering;
4919 for (i, (ka, kb)) in a.0.iter().zip(b.0.iter()).enumerate() {
4920 let (desc, nf) = keys_meta[i];
4921 // v7.39 (enum order knife) — enum keys sort by member order.
4922 if let Some(Some(labels)) = key_enum_labels.get(i)
4923 && !matches!(ka, Value::Null)
4924 && !matches!(kb, Value::Null)
4925 && let Some(ord) = crate::eval::enum_ord_cmp(labels, ka, kb)
4926 {
4927 let ord = if desc { ord.reverse() } else { ord };
4928 if ord != Ordering::Equal {
4929 return ord;
4930 }
4931 continue;
4932 }
4933 let c = crate::orderby::order_by_value_cmp_coll(
4934 desc,
4935 nf,
4936 ka,
4937 kb,
4938 mysql,
4939 key_colls.get(i).and_then(|c| c.as_deref()),
4940 );
4941 if c != Ordering::Equal {
4942 return c;
4943 }
4944 }
4945 Ordering::Equal
4946 };
4947 // v7.37.3 — top-K partial sort when `keep_n` is small enough to
4948 // matter (`Some(k)` with `k < tagged.len()` and `k > 0`).
4949 // `select_nth_unstable_by` partitions in O(N), then we sort the
4950 // surviving prefix in O(K log K). Total = O(N + K log K) vs
4951 // O(N log N) the full sort would pay — matches the inbox-listing
4952 // shape PG uses.
4953 //
4954 match keep_n {
4955 Some(k) if k < tagged.len() && k > 0 => {
4956 let pivot = k - 1;
4957 tagged.select_nth_unstable_by(pivot, cmp);
4958 tagged[..k].sort_by(cmp);
4959 tagged.truncate(k);
4960 }
4961 _ => {
4962 tagged.sort_by(cmp);
4963 }
4964 }
4965 kept_synth = Vec::with_capacity(tagged.len());
4966 out_rows = Vec::with_capacity(tagged.len());
4967 for (_, s, o) in tagged {
4968 kept_synth.push(s);
4969 out_rows.push(o);
4970 }
4971 Ok((kept_synth, out_rows))
4972}
4973
4974/// v7.17.0 — walk the statement again to validate the positional
4975/// arity of every aggregate call site. Done after AST collection
4976/// rather than inside `collect_aggregates` so the collector stays
4977/// infallible; callers in `run()` can do a single early-error
4978/// exit before any per-row work.
4979fn validate_agg_arities(stmt: &SelectStatement, _specs: &[AggSpec]) -> Result<(), EvalError> {
4980 fn walk(e: &Expr) -> Result<(), EvalError> {
4981 if let Expr::FunctionCall { name, args } = e {
4982 let lower = name.to_ascii_lowercase();
4983 let expected: Option<usize> = match lower.as_str() {
4984 "count_star" => Some(0),
4985 "count" | "sum" | "avg" | "min" | "max" | "array_agg"
4986 | "any_value" | "range_agg" | "range_intersect_agg"
4987 // v7.17.0 — boolean aggregates also take exactly
4988 // one arg. `every` is an alias normalised inside
4989 // collect_aggregates / rewrite_expr.
4990 | "bool_and" | "bool_or" | "every"
4991 // v7.32 (round-29) — statistical + bitwise aggregates
4992 // + single-arg JSON aggregate.
4993 | "stddev" | "stddev_samp" | "stddev_pop"
4994 | "variance" | "var_samp" | "var_pop"
4995 | "bit_and" | "bit_or" | "bit_xor"
4996 | "json_agg" | "jsonb_agg" | "xmlagg"
4997 | "json_arrayagg" | "json_agg_strict" | "jsonb_agg_strict" => Some(1),
4998 // v7.39 (round 354, M12) — GROUP_CONCAT takes any number of
4999 // arguments: MySQL concatenates them PER ROW
5000 // (`GROUP_CONCAT(n, ':', t)` is `3:c,1:a,…`, measured), and
5001 // the parser lowers a `SEPARATOR '<s>'` tail onto the last
5002 // one. Fixing the arity at 1 refused both.
5003 "group_concat" => None,
5004 // v7.32 (round-29) — two-argument aggregates: string_agg,
5005 // the regression family f(Y, X), and json_object_agg.
5006 "string_agg"
5007 | "covar_pop" | "covar_samp" | "corr"
5008 | "regr_count" | "regr_avgx" | "regr_avgy" | "regr_slope"
5009 | "regr_intercept" | "regr_r2" | "regr_sxx" | "regr_syy" | "regr_sxy"
5010 | "json_object_agg" | "jsonb_object_agg"
5011 | "json_objectagg"
5012 | "json_object_agg_strict" | "jsonb_object_agg_strict"
5013 | "json_object_agg_unique" | "jsonb_object_agg_unique"
5014 | "json_object_agg_unique_strict" | "jsonb_object_agg_unique_strict" => Some(2),
5015 _ => None,
5016 };
5017 if let Some(want) = expected
5018 && args.len() != want
5019 {
5020 return Err(EvalError::TypeMismatch {
5021 detail: alloc::format!("{lower}() takes {want} arg(s), got {}", args.len()),
5022 });
5023 }
5024 for a in args {
5025 walk(a)?;
5026 }
5027 } else if let Expr::Binary { lhs, rhs, .. } = e {
5028 walk(lhs)?;
5029 walk(rhs)?;
5030 } else if let Expr::Unary { expr, .. }
5031 | Expr::Cast { expr, .. }
5032 | Expr::IsNull { expr, .. }
5033 | Expr::BoolTest { expr, .. } = e
5034 {
5035 walk(expr)?;
5036 }
5037 Ok(())
5038 }
5039 for item in &stmt.items {
5040 if let SelectItem::Expr { expr, .. } = item {
5041 walk(expr)?;
5042 }
5043 }
5044 for o in &stmt.order_by {
5045 walk(&o.expr)?;
5046 }
5047 if let Some(h) = &stmt.having {
5048 walk(h)?;
5049 }
5050 Ok(())
5051}
5052
5053/// v7.33 (array_agg argmax) — recognise `(array_agg(x ORDER BY y))[1]`,
5054/// the argmax/argmin idiom: a non-DISTINCT ordered `array_agg`
5055/// subscripted by the constant 1. Returns `(value_arg, order_by,
5056/// filter)` on a match. When matched, the whole per-group array build +
5057/// sort + materialise is replaced by a running first-by-order scalar
5058/// accumulator and the subscript node is consumed (replaced by the
5059/// synthetic column). collect_aggregates and rewrite_expr share this one
5060/// matcher so their `__agg_<i>` assignment stays in lockstep.
5061fn first_ordered_array_agg(e: &Expr) -> Option<(&Expr, &[spg_sql::ast::OrderBy], Option<&Expr>)> {
5062 let Expr::ArraySubscript { target, index } = e else {
5063 return None;
5064 };
5065 if !matches!(
5066 index.as_ref(),
5067 Expr::Literal(spg_sql::ast::Literal::Integer(1))
5068 ) {
5069 return None;
5070 }
5071 let Expr::AggregateOrdered {
5072 call,
5073 order_by,
5074 distinct,
5075 filter,
5076 } = target.as_ref()
5077 else {
5078 return None;
5079 };
5080 if *distinct || order_by.is_empty() {
5081 return None;
5082 }
5083 let Expr::FunctionCall { name, args } = call.as_ref() else {
5084 return None;
5085 };
5086 if !name.eq_ignore_ascii_case("array_agg") || args.len() != 1 {
5087 return None;
5088 }
5089 Some((&args[0], order_by, filter.as_deref()))
5090}
5091
5092/// v7.39 (round 615) — the exact pair the finaliser reads: the BigNumeric
5093/// accumulator combined with whatever the i128 one still holds. Read-only,
5094/// because finalisation only borrows the state.
5095fn stddev_exact_pair(
5096 st: &AggState,
5097) -> Option<(
5098 spg_storage::bignum::BigNumeric,
5099 spg_storage::bignum::BigNumeric,
5100)> {
5101 use spg_storage::bignum::BigNumeric as BN;
5102 let fast =
5103 (!st.stddev_i_spent && (st.stddev_i_sum != 0 || st.stddev_i_sum_sq != 0)).then(|| {
5104 (
5105 BN::from_i128(st.stddev_i_sum, 0),
5106 BN::from_i128(st.stddev_i_sum_sq, 0),
5107 )
5108 });
5109 match (st.stddev_sum.as_ref(), st.stddev_sum_sq.as_ref(), fast) {
5110 (Some(s), Some(sq), Some((fs, fsq))) => Some((s.add(&fs), sq.add(&fsq))),
5111 (Some(s), Some(sq), None) => Some((s.clone(), sq.clone())),
5112 (None, None, Some(pair)) => Some(pair),
5113 _ => None,
5114 }
5115}
5116
5117/// v7.39 (round 615) — fold the i128 Σx / Σx² into the exact BigNumeric
5118/// pair and retire the fast accumulator. Called once when an input needs the
5119/// slow path, and once at finalisation; both are idempotent because the fast
5120/// pair is zeroed as it is spent.
5121fn spend_stddev_i128(st: &mut AggState) {
5122 if st.stddev_i_spent {
5123 return;
5124 }
5125 st.stddev_i_spent = true;
5126 if st.stddev_i_sum == 0 && st.stddev_i_sum_sq == 0 {
5127 // Nothing accumulated: leave the pair as it was (None means "no
5128 // exact input yet", which the finaliser reads).
5129 return;
5130 }
5131 use spg_storage::bignum::BigNumeric as BN;
5132 let sum = BN::from_i128(st.stddev_i_sum, 0);
5133 let sum_sq = BN::from_i128(st.stddev_i_sum_sq, 0);
5134 st.stddev_sum = Some(st.stddev_sum.as_ref().map_or(sum.clone(), |s| s.add(&sum)));
5135 st.stddev_sum_sq = Some(
5136 st.stddev_sum_sq
5137 .as_ref()
5138 .map_or(sum_sq.clone(), |s| s.add(&sum_sq)),
5139 );
5140}
5141
5142fn collect_aggregates(e: &Expr, out: &mut Vec<AggSpec>) {
5143 match e {
5144 Expr::NamedArg { expr, .. } => collect_aggregates(expr, out),
5145 Expr::Variadic(expr) => collect_aggregates(expr, out),
5146 // v7.24 (round-16 A) — ordered aggregate: register the inner
5147 // call's spec with the ordering attached.
5148 Expr::AggregateOrdered {
5149 call,
5150 order_by,
5151 distinct,
5152 filter,
5153 } => {
5154 if let Expr::FunctionCall { name, args } = call.as_ref() {
5155 let lower = name.to_ascii_lowercase();
5156 if is_aggregate_name(&lower) {
5157 let canonical = if lower == "every" {
5158 "bool_and".to_string()
5159 } else {
5160 lower
5161 };
5162 // Ordered-set aggregates (`percentile_cont(f)
5163 // WITHIN GROUP (ORDER BY x)`) take the value to
5164 // aggregate from the sort spec and the in-parens
5165 // arg as the direct (fraction) argument.
5166 let ordered_set = is_within_group_name(&canonical);
5167 let (arg, direct_arg, direct_args_extra) = if ordered_set {
5168 (
5169 order_by.first().map(|o| o.expr.clone()),
5170 args.first().cloned(),
5171 args.iter().skip(1).cloned().collect(),
5172 )
5173 } else {
5174 (args.first().cloned(), None, Vec::new())
5175 };
5176 let spec = AggSpec {
5177 kind: classify_agg_name(&canonical),
5178 enum_labels: None,
5179 arg_collation: None,
5180 order_enum_labels: Vec::new(),
5181 order_collations: Vec::new(),
5182 name: canonical.clone(),
5183 arg,
5184 arg2: if agg_uses_second_arg(&canonical) {
5185 args.get(1).cloned()
5186 } else {
5187 None
5188 },
5189 distinct: *distinct,
5190 order_by: order_by.clone(),
5191 filter: filter.as_deref().cloned(),
5192 direct_arg,
5193 direct_args_extra,
5194 first_ordered: false,
5195 };
5196 if !out.iter().any(|s| {
5197 s.name == spec.name
5198 && s.arg == spec.arg
5199 && s.arg2 == spec.arg2
5200 && s.distinct == spec.distinct
5201 && s.order_by == spec.order_by
5202 && s.filter == spec.filter
5203 && s.direct_arg == spec.direct_arg
5204 && s.direct_args_extra == spec.direct_args_extra
5205 && s.first_ordered == spec.first_ordered
5206 }) {
5207 out.push(spec);
5208 }
5209 return;
5210 }
5211 }
5212 collect_aggregates(call, out);
5213 for o in order_by {
5214 collect_aggregates(&o.expr, out);
5215 }
5216 }
5217 Expr::FunctionCall { name, args } => {
5218 let lower = name.to_ascii_lowercase();
5219 if is_aggregate_name(&lower) {
5220 let arg = if lower == "count_star" {
5221 None
5222 } else {
5223 args.first().cloned()
5224 };
5225 // v7.17.0 — second positional arg for
5226 // `string_agg(value, separator)`; v7.32 — also the
5227 // regression family `f(Y, X)` and `json_object_agg`.
5228 let arg2 = if agg_uses_second_arg(&lower) {
5229 args.get(1).cloned()
5230 } else {
5231 None
5232 };
5233 // v7.17.0 — `every` is the SQL-standard alias for
5234 // `bool_and`; collapse at collection time so
5235 // update_state / finalize need only one arm.
5236 let canonical = if lower == "every" {
5237 "bool_and".to_string()
5238 } else {
5239 lower
5240 };
5241 let spec = AggSpec {
5242 kind: classify_agg_name(&canonical),
5243 enum_labels: None,
5244 arg_collation: None,
5245 order_enum_labels: Vec::new(),
5246 order_collations: Vec::new(),
5247 name: canonical,
5248 arg: arg.clone(),
5249 arg2: arg2.clone(),
5250 distinct: false,
5251 order_by: Vec::new(),
5252 filter: None,
5253 direct_arg: None,
5254 direct_args_extra: Vec::new(),
5255 first_ordered: false,
5256 };
5257 if !out.iter().any(|s| {
5258 s.name == spec.name
5259 && s.arg == spec.arg
5260 && s.arg2 == spec.arg2
5261 && !s.distinct
5262 && s.order_by == spec.order_by
5263 && s.filter.is_none()
5264 && !s.first_ordered
5265 }) {
5266 out.push(spec);
5267 }
5268 // Don't recurse into the arg — nested aggregates are
5269 // illegal in standard SQL.
5270 } else {
5271 for a in args {
5272 collect_aggregates(a, out);
5273 }
5274 }
5275 }
5276 Expr::Binary { lhs, rhs, .. } => {
5277 collect_aggregates(lhs, out);
5278 collect_aggregates(rhs, out);
5279 }
5280 Expr::Unary { expr, .. }
5281 | Expr::Cast { expr, .. }
5282 | Expr::IsNull { expr, .. }
5283 | Expr::BoolTest { expr, .. }
5284 | Expr::FieldAccess { base: expr, .. } => {
5285 collect_aggregates(expr, out);
5286 }
5287 Expr::Like { expr, pattern, .. } => {
5288 collect_aggregates(expr, out);
5289 collect_aggregates(pattern, out);
5290 }
5291 Expr::InList { expr, list, .. } => {
5292 collect_aggregates(expr, out);
5293 for item in list {
5294 collect_aggregates(item, out);
5295 }
5296 }
5297 Expr::Extract { source, .. } => collect_aggregates(source, out),
5298 // v4.10 subquery + v4.12 window / Literal / Column —
5299 // non-recursing leaves for the aggregate collector.
5300 Expr::ScalarSubquery(_)
5301 | Expr::Exists { .. }
5302 | Expr::InSubquery { .. }
5303 | Expr::RowInSubquery { .. }
5304 | Expr::RowCmpSubquery { .. }
5305 | Expr::WindowFunction { .. }
5306 | Expr::Literal(_)
5307 | Expr::Placeholder(_)
5308 | Expr::Column(_) => {}
5309 // v7.10.10 — recurse into array constructor children +
5310 // subscript / ANY/ALL operands.
5311 Expr::Array(items) => {
5312 for elem in items {
5313 collect_aggregates(elem, out);
5314 }
5315 }
5316 Expr::ArraySubscript { target, index } => {
5317 // v7.33 (array_agg argmax) — `(array_agg(x ORDER BY y))[1]`
5318 // collects as a first_ordered spec; the subscript is consumed
5319 // here (do NOT recurse into the array_agg, or it would also
5320 // register a plain full-array spec).
5321 if let Some((arg, order_by, filter)) = first_ordered_array_agg(e) {
5322 let spec = AggSpec {
5323 kind: AggKind::ArrayAgg,
5324 enum_labels: None,
5325 arg_collation: None,
5326 order_enum_labels: Vec::new(),
5327 order_collations: Vec::new(),
5328 name: "array_agg".to_string(),
5329 arg: Some(arg.clone()),
5330 arg2: None,
5331 distinct: false,
5332 order_by: order_by.to_vec(),
5333 filter: filter.cloned(),
5334 direct_arg: None,
5335 direct_args_extra: Vec::new(),
5336 first_ordered: true,
5337 };
5338 if !out.iter().any(|s| {
5339 s.name == spec.name
5340 && s.arg == spec.arg
5341 && s.order_by == spec.order_by
5342 && s.filter == spec.filter
5343 && s.first_ordered
5344 }) {
5345 out.push(spec);
5346 }
5347 return;
5348 }
5349 collect_aggregates(target, out);
5350 collect_aggregates(index, out);
5351 }
5352 Expr::ArraySlice { target, lo, hi } => {
5353 collect_aggregates(target, out);
5354 if let Some(l) = lo {
5355 collect_aggregates(l, out);
5356 }
5357 if let Some(h) = hi {
5358 collect_aggregates(h, out);
5359 }
5360 }
5361 Expr::AnyAll { expr, array, .. } => {
5362 collect_aggregates(expr, out);
5363 collect_aggregates(array, out);
5364 }
5365 Expr::Case {
5366 operand,
5367 branches,
5368 else_branch,
5369 } => {
5370 if let Some(o) = operand {
5371 collect_aggregates(o, out);
5372 }
5373 for (w, t) in branches {
5374 collect_aggregates(w, out);
5375 collect_aggregates(t, out);
5376 }
5377 if let Some(e) = else_branch {
5378 collect_aggregates(e, out);
5379 }
5380 }
5381 }
5382}
5383
5384pub(crate) fn update_state(
5385 st: &mut AggState,
5386 kind: AggKind,
5387 name: &str,
5388 v: &Value<'_>,
5389 arg2: Option<&Value<'_>>,
5390 order_keys: Option<Vec<Value<'static>>>,
5391 enum_labels: Option<&[String]>,
5392 // v7.39 (round 690) — the argument column's collation, beside
5393 // `enum_labels` because it is the same kind of fact about the argument.
5394 arg_collation: Option<&str>,
5395 mysql: bool,
5396) -> Result<(), EvalError> {
5397 let is_null = matches!(v, Value::Null);
5398 // v7.37.4 (R34) — dispatch by pre-classified `kind` (`Copy`
5399 // enum), not by per-row string match. Hot inner loop on
5400 // multi-aggregate queries (mailrs `/api/conversations`: 14
5401 // aggregates × 100 k rows = 1.4 M dispatches) sees an enum
5402 // jump table instead of a sequence of `eq_str` checks. `name`
5403 // is still threaded through for error messages so the user-
5404 // facing wording is unchanged.
5405 match kind {
5406 AggKind::CountStar => st.num.count += 1,
5407 AggKind::Count => {
5408 if !is_null {
5409 st.num.count += 1;
5410 }
5411 }
5412 AggKind::Sum | AggKind::Avg => {
5413 // v7.39 (round 665) — was a hand-copied duplicate of `acc_cell`,
5414 // arm for arm, down to the wording of the type error. Verified
5415 // equivalent before collapsing: same nine variants, same error,
5416 // and the two apparent differences are both unobservable — this
5417 // one counted before the match so a value that errors bumped the
5418 // count first (the error aborts the query, so it is discarded),
5419 // and its `is_null` early return is literally
5420 // `matches!(v, Value::Null)`, which is the arm `acc_cell` has.
5421 //
5422 // Round 626 had to add a SMALLINT arm HERE that the other three
5423 // copies already carried; `SELECT sum(x)` over a smallint column
5424 // answered "sum/avg need numeric, got smallint" until then. That
5425 // is the failure mode this collapse removes.
5426 acc_cell(&mut st.num, v)?;
5427 }
5428 AggKind::Min => {
5429 if is_null {
5430 return Ok(());
5431 }
5432 if !mysql && min_max_unsupported_type(v) {
5433 return Err(EvalError::TypeMismatch {
5434 detail: format!(
5435 "function min({}) does not exist",
5436 crate::conversions::pg_type_name_for_error_opt(v.data_type())
5437 ),
5438 });
5439 }
5440 match &st.extreme {
5441 None => st.extreme = Some(v.clone().into_owned()),
5442 Some(cur) => {
5443 if extreme_cmp_in(enum_labels, arg_collation, v, cur, mysql)
5444 == core::cmp::Ordering::Less
5445 {
5446 st.extreme = Some(v.clone().into_owned());
5447 }
5448 }
5449 }
5450 }
5451 AggKind::AnyValue => {
5452 if is_null {
5453 return Ok(());
5454 }
5455 if st.extreme.is_none() {
5456 st.extreme = Some(v.clone().into_owned());
5457 }
5458 }
5459 AggKind::RangeAgg => {
5460 if is_null {
5461 return Ok(());
5462 }
5463 let Value::Range {
5464 kind,
5465 lower,
5466 upper,
5467 lower_inc,
5468 upper_inc,
5469 empty,
5470 } = v
5471 else {
5472 return Err(EvalError::TypeMismatch {
5473 detail: format!(
5474 "range_agg requires a range value, got {}",
5475 crate::conversions::pg_type_name_for_error_opt(v.data_type())
5476 ),
5477 });
5478 };
5479 // Initialise the accumulator on first sight (even for
5480 // an empty range, so all-empty groups finalize to {}).
5481 if st.extreme.is_none() {
5482 st.extreme = Some(Value::Multirange {
5483 kind: *kind,
5484 ranges: alloc::vec::Vec::new(),
5485 });
5486 }
5487 if !empty && let Some(Value::Multirange { ranges, .. }) = &mut st.extreme {
5488 ranges.push(spg_storage::RangeSpan {
5489 lower: lower.clone(),
5490 upper: upper.clone(),
5491 lower_inc: *lower_inc,
5492 upper_inc: *upper_inc,
5493 empty: false,
5494 });
5495 }
5496 }
5497 AggKind::RangeIntersectAgg => {
5498 if is_null {
5499 return Ok(());
5500 }
5501 if !matches!(v, Value::Range { .. }) {
5502 return Err(EvalError::TypeMismatch {
5503 detail: format!(
5504 "range_intersect_agg requires a range value, got {}",
5505 crate::conversions::pg_type_name_for_error_opt(v.data_type())
5506 ),
5507 });
5508 }
5509 match &st.extreme {
5510 None => st.extreme = Some(v.clone().into_owned()),
5511 Some(prev) => {
5512 st.extreme = Some(range_intersect(prev, &v.clone().into_owned()));
5513 }
5514 }
5515 }
5516 AggKind::Max => {
5517 if is_null {
5518 return Ok(());
5519 }
5520 if !mysql && min_max_unsupported_type(v) {
5521 return Err(EvalError::TypeMismatch {
5522 detail: format!(
5523 "function max({}) does not exist",
5524 crate::conversions::pg_type_name_for_error_opt(v.data_type())
5525 ),
5526 });
5527 }
5528 match &st.extreme {
5529 None => st.extreme = Some(v.clone().into_owned()),
5530 Some(cur) => {
5531 if extreme_cmp_in(enum_labels, arg_collation, v, cur, mysql)
5532 == core::cmp::Ordering::Greater
5533 {
5534 st.extreme = Some(v.clone().into_owned());
5535 }
5536 }
5537 }
5538 }
5539 // v7.17.0 — string_agg(value, separator). NULL value is
5540 // skipped (PG aggregate-skip-null). v7.39 (round 762,
5541 // F31-C2) — the separator is PER ROW in PG (the old note's
5542 // "using the last value at finalize" claim was measured
5543 // false): each surviving item records its own row's
5544 // separator in `item_seps`; the `separator` snapshot stays
5545 // for the constant-path consumers. count is bumped so we can
5546 // distinguish "empty group → NULL" from "all-NULL group →
5547 // NULL".
5548 AggKind::StringAgg => {
5549 let has_arg2 = arg2.is_some();
5550 if let Some(sep) = arg2
5551 && let Value::Text(s) = sep
5552 {
5553 st.separator = Some(s.to_string());
5554 }
5555 if is_null {
5556 return Ok(());
5557 }
5558 // Text collects as-is; other scalars coerce to their
5559 // text rendering (MySQL group_concat semantics — also
5560 // matches PG's cast-then-aggregate idiom for
5561 // string_agg(v::text, sep)).
5562 let rendered = render_string_agg_item(v);
5563 if let Some(item) = rendered {
5564 st.items.push(item);
5565 // v7.39 (round 762, F31-C2) — the row's own separator
5566 // rides with its item (NULL separator → None → empty).
5567 if has_arg2 {
5568 st.item_seps.push(match arg2 {
5569 Some(Value::Text(sp)) => Some(sp.to_string()),
5570 _ => None,
5571 });
5572 }
5573 if let Some(k) = order_keys {
5574 st.item_keys.extend(k);
5575 }
5576 st.num.count += 1;
5577 } else {
5578 return Err(EvalError::TypeMismatch {
5579 detail: format!(
5580 "string_agg requires text value, got {}",
5581 crate::conversions::pg_type_name_for_error_opt(v.data_type())
5582 ),
5583 });
5584 }
5585 }
5586 // v7.17.0 — array_agg(value). Unlike string_agg, NULL
5587 // elements are KEPT in the array (PG behaviour); the
5588 // result is NULL only when ZERO rows fed in. Element type
5589 // is locked from the first row's value type; subsequent
5590 // rows must match (PG also rejects mixed-type array_agg).
5591 AggKind::ArrayAgg => {
5592 st.items.push(v.clone().into_owned());
5593 if let Some(k) = order_keys {
5594 st.item_keys.extend(k);
5595 }
5596 st.num.count += 1;
5597 }
5598 // v7.17.0 — bool_and(p): TRUE iff every non-NULL input is
5599 // TRUE. NULL skipped; running accumulator stays at TRUE
5600 // until the first non-NULL FALSE.
5601 AggKind::BoolAnd => {
5602 if is_null {
5603 return Ok(());
5604 }
5605 let b = match v {
5606 Value::Bool(b) => *b,
5607 other => {
5608 return Err(EvalError::TypeMismatch {
5609 detail: format!(
5610 "bool_and requires bool, got {}",
5611 crate::conversions::pg_type_name_for_error_opt(other.data_type())
5612 ),
5613 });
5614 }
5615 };
5616 st.bool_acc = Some(st.bool_acc.map_or(b, |acc| acc && b));
5617 }
5618 // v7.17.0 — bool_or(p): TRUE iff any non-NULL input is
5619 // TRUE. NULL skipped.
5620 AggKind::BoolOr => {
5621 if is_null {
5622 return Ok(());
5623 }
5624 let b = match v {
5625 Value::Bool(b) => *b,
5626 other => {
5627 return Err(EvalError::TypeMismatch {
5628 detail: format!(
5629 "bool_or requires bool, got {}",
5630 crate::conversions::pg_type_name_for_error_opt(other.data_type())
5631 ),
5632 });
5633 }
5634 };
5635 st.bool_acc = Some(st.bool_acc.map_or(b, |acc| acc || b));
5636 }
5637 // v7.32 (round-29) — variance / stddev family. Accumulate the
5638 // running sum (sum_float) and sum of squares (sum_sq) over the
5639 // non-NULL numeric inputs; finalize divides by n or n-1.
5640 AggKind::StddevFamily => {
5641 if is_null {
5642 return Ok(());
5643 }
5644 // v7.38 (read01) — keep an exact NUMERIC Σx / Σx² alongside the f64
5645 // pair for as long as every input is exact; a float input abandons it.
5646 if !st.stddev_saw_float {
5647 // v7.39 (round 615) — an integer input stays in i128, which is
5648 // exact and allocates nothing. Anything else, or an overflow,
5649 // spends the fast accumulator into the BigNumeric pair and
5650 // takes the old path from there.
5651 let as_int = match v {
5652 Value::SmallInt(n) => Some(i128::from(*n)),
5653 Value::Int(n) => Some(i128::from(*n)),
5654 Value::BigInt(n) => Some(i128::from(*n)),
5655 _ => None,
5656 };
5657 let folded = if st.stddev_i_spent {
5658 None
5659 } else if let Some(x) = as_int {
5660 match (
5661 st.stddev_i_sum.checked_add(x),
5662 x.checked_mul(x)
5663 .and_then(|xx| st.stddev_i_sum_sq.checked_add(xx)),
5664 ) {
5665 (Some(s), Some(sq)) => {
5666 st.stddev_i_sum = s;
5667 st.stddev_i_sum_sq = sq;
5668 Some(())
5669 }
5670 _ => None,
5671 }
5672 } else {
5673 None
5674 };
5675 if folded.is_none() {
5676 spend_stddev_i128(st);
5677 match crate::eval::binop::value_to_bignum(v) {
5678 Some(b) => {
5679 let sq = b.mul(&b);
5680 st.stddev_sum = Some(
5681 st.stddev_sum
5682 .as_ref()
5683 .map_or_else(|| b.clone(), |s| s.add(&b)),
5684 );
5685 st.stddev_sum_sq = Some(
5686 st.stddev_sum_sq
5687 .as_ref()
5688 .map_or_else(|| sq.clone(), |s| s.add(&sq)),
5689 );
5690 }
5691 None => st.stddev_saw_float = true,
5692 }
5693 }
5694 }
5695 let Some(x) = agg_value_to_f64(v) else {
5696 return Err(EvalError::TypeMismatch {
5697 detail: format!(
5698 "{name} needs numeric, got {}",
5699 crate::conversions::pg_type_name_for_error_opt(v.data_type())
5700 ),
5701 });
5702 };
5703 st.num.count += 1;
5704 st.num.sum_float += x;
5705 st.sum_sq += x * x;
5706 }
5707 // v7.32 (round-29) — bitwise aggregates over integer inputs.
5708 AggKind::BitAnd | AggKind::BitOr | AggKind::BitXor => {
5709 if is_null {
5710 return Ok(());
5711 }
5712 let n = match v {
5713 Value::Int(n) => i64::from(*n),
5714 Value::SmallInt(n) => i64::from(*n),
5715 Value::BigInt(n) => *n,
5716 other => {
5717 return Err(EvalError::TypeMismatch {
5718 detail: format!(
5719 "{name} needs integer, got {}",
5720 crate::conversions::pg_type_name_for_error_opt(other.data_type())
5721 ),
5722 });
5723 }
5724 };
5725 if matches!(v, Value::BigInt(_)) {
5726 st.bit_wide = true;
5727 }
5728 st.bit_acc = Some(match (st.bit_acc, kind) {
5729 (None, _) => n,
5730 (Some(acc), AggKind::BitAnd) => acc & n,
5731 (Some(acc), AggKind::BitOr) => acc | n,
5732 (Some(acc), _) => acc ^ n, // BitXor
5733 });
5734 }
5735 // v7.32 (round-29) — WITHIN GROUP aggregates (ordered-set +
5736 // hypothetical-set) collect the sort value (NULLs ignored, per
5737 // PG) into `items`, sorted at finalize by the parallel
5738 // `item_keys`.
5739 AggKind::WithinGroup => {
5740 // Counted before the NULL skip: the hypothetical-set
5741 // fractions divide by the full input size (PG).
5742 st.within_group_rows += 1;
5743 if is_null {
5744 return Ok(());
5745 }
5746 st.items.push(v.clone().into_owned());
5747 if let Some(k) = order_keys {
5748 st.item_keys.extend(k);
5749 }
5750 st.num.count += 1;
5751 }
5752 // v7.32 (round-29) — regression family f(Y, X). Only rows with
5753 // BOTH inputs non-NULL contribute (PG semantics). `v` is Y,
5754 // `arg2` is X.
5755 AggKind::Regression => {
5756 let (Some(y), Some(x)) = (agg_value_to_f64(v), arg2.and_then(agg_value_to_f64)) else {
5757 return Ok(()); // NULL (or non-numeric) in either input
5758 };
5759 // v7.39 (read01 round 115) — accumulate the sums of squared
5760 // deviations (Sxx / Syy / Sxy) incrementally via the Youngs-Cramer
5761 // update, matching PG's float8 regression aggregates to the last
5762 // ULP. The old naive form (`Σx² − (Σx)²/n` at finalize time) is
5763 // mathematically equal but rounds differently, so `corr` drifted in
5764 // the 16th digit. reg_sx / reg_sy stay raw sums (for the averages).
5765 st.reg_n += 1;
5766 let new_n = st.reg_n as f64;
5767 let new_sx = st.reg_sx + x;
5768 let new_sy = st.reg_sy + y;
5769 if st.reg_n > 1 {
5770 let n_prev = new_n - 1.0;
5771 let tmp_x = x * new_n - new_sx;
5772 let tmp_y = y * new_n - new_sy;
5773 let scale = 1.0 / (n_prev * new_n);
5774 st.reg_sxx += tmp_x * tmp_x * scale;
5775 st.reg_syy += tmp_y * tmp_y * scale;
5776 st.reg_sxy += tmp_x * tmp_y * scale;
5777 }
5778 st.reg_sx = new_sx;
5779 st.reg_sy = new_sy;
5780 }
5781 // v7.32 (round-29) — json_agg / jsonb_agg collect every input
5782 // (NULL becomes JSON null, per PG) in row order.
5783 AggKind::JsonAgg => {
5784 // v7.39 (read01 json.c) — the _strict variants skip NULLs.
5785 if is_null && name.ends_with("_strict") {
5786 return Ok(());
5787 }
5788 st.items.push(v.clone().into_owned());
5789 // Attach the ORDER BY key so finalize_synth_rows sorts the
5790 // elements (`json_agg(x ORDER BY x DESC)`), the same way
5791 // string_agg / array_agg do.
5792 if let Some(k) = order_keys {
5793 st.item_keys.extend(k);
5794 }
5795 st.num.count += 1;
5796 }
5797 // v7.32 (round-29) — json_object_agg(key, value): keys in
5798 // `items`, values in `aux_items`. A NULL key is skipped (PG
5799 // raises; we drop it rather than abort the whole query).
5800 AggKind::JsonObjectAgg => {
5801 if is_null {
5802 return Ok(());
5803 }
5804 // v7.39 (read01 json.c) — _strict skips NULL VALUES; _unique
5805 // raises PG's duplicate-key error.
5806 let val = arg2.cloned().map(Value::into_owned).unwrap_or(Value::Null);
5807 if matches!(val, Value::Null) && name.contains("_strict") {
5808 return Ok(());
5809 }
5810 if name.contains("_unique") {
5811 let kt = match v {
5812 Value::Text(s) | Value::Json(s) => s.to_string(),
5813 other => crate::json::value_to_json_text(other),
5814 };
5815 let dup = st.items.iter().any(|k| match k {
5816 Value::Text(s) | Value::Json(s) => *s == kt,
5817 other => crate::json::value_to_json_text(other) == kt,
5818 });
5819 if dup {
5820 return Err(EvalError::TypeMismatch {
5821 detail: alloc::format!("duplicate JSON object key value: {kt:?}"),
5822 });
5823 }
5824 }
5825 st.items.push(v.clone().into_owned());
5826 st.aux_items.push(val);
5827 st.num.count += 1;
5828 }
5829 }
5830 Ok(())
5831}
5832
5833#[allow(clippy::cast_precision_loss, clippy::cast_possible_truncation)]
5834pub(crate) fn finalize(name: &str, st: &AggState, mysql: bool) -> Value<'static> {
5835 match name {
5836 "count" | "count_star" => Value::BigInt(st.num.count),
5837 "sum" => {
5838 if st.num.count == 0 {
5839 Value::Null
5840 } else if st.num.use_interval {
5841 Value::Interval {
5842 months: st.num.sum_iv_months as i32,
5843 days: st.num.sum_iv_days as i32,
5844 micros: st.num.sum_iv_micros as i64,
5845 }
5846 } else if st.num.use_money {
5847 Value::Money(st.num.sum_money as i64)
5848 } else if st.num.use_numeric {
5849 // v7.38 (read01, T6.P3) — a NaN / ±Infinity input propagates.
5850 if st.num.sum_num_kind != spg_storage::NumericKind::Finite {
5851 Value::numeric_special(st.num.sum_num_kind)
5852 } else if let Some(big) = &st.num.sum_big {
5853 // v7.39 (read01 numeric.c) — the sum spilled past i128;
5854 // fold in the int lane and render exactly.
5855 let tot = big.add(&spg_storage::bignum::BigNumeric::from_i128(
5856 i128::from(st.num.sum_int),
5857 0,
5858 ));
5859 crate::eval::binop::bignum_to_value(tot)
5860 } else {
5861 let (scaled, scale) = crate::numeric::numeric_add(
5862 st.num.sum_num_scaled,
5863 st.num.sum_num_scale,
5864 i128::from(st.num.sum_int),
5865 0,
5866 );
5867 Value::Numeric {
5868 scaled,
5869 scale,
5870 kind: spg_storage::NumericKind::Finite,
5871 }
5872 }
5873 } else if st.num.use_float {
5874 let total = st.num.sum_float + (st.num.sum_int as f64);
5875 // v7.39 (round 269) — sum over REAL input stays real in
5876 // PG; it widens only when something wider joined the
5877 // accumulation. avg is deliberately not the same:
5878 // avg(real) IS double precision (measured on 18.4).
5879 if st.num.float_not_real {
5880 Value::Float(total)
5881 } else {
5882 #[allow(clippy::cast_possible_truncation)]
5883 Value::Real(total as f32)
5884 }
5885 } else {
5886 Value::BigInt(st.num.sum_int)
5887 }
5888 }
5889 "avg" => {
5890 if st.num.count == 0 {
5891 Value::Null
5892 } else if st.num.use_interval {
5893 // PG interval_div: the month quotient truncates and its
5894 // remainder spills into DAYS (a month = 30 days), taking the
5895 // whole-day part into the day field and only the sub-day
5896 // fraction into time; the day remainder then spills into time.
5897 let n = i128::from(st.num.count);
5898 let day_us = 86_400_000_000i128;
5899 let months = i128::from(st.num.sum_iv_months);
5900 let days = i128::from(st.num.sum_iv_days);
5901 let month_out = months / n;
5902 let mrem_days_total = (months % n) * 30; // days (still over n)
5903 let days_from_month = mrem_days_total / n;
5904 let mrem_frac_us = (mrem_days_total % n) * day_us / n;
5905 let day_out = days / n;
5906 let drem_us = (days % n) * day_us / n;
5907 let micros = st.num.sum_iv_micros / n + mrem_frac_us + drem_us;
5908 Value::Interval {
5909 months: month_out as i32,
5910 days: (day_out + days_from_month) as i32,
5911 micros: micros as i64,
5912 }
5913 } else if st.num.use_money {
5914 // PG has no avg(money); we accept it as a sensible superset —
5915 // average of the cent totals, rounded half-away-from-zero.
5916 //
5917 // DELIBERATE. Round 664 read "PG refuses, SPG answers" off
5918 // the F29 list and wrote guards on four accumulators to
5919 // remove this before a test caught it. Per the round-641
5920 // policy such a divergence is judged by correctness risk,
5921 // and this one carries none: money IS cents, so rounding is
5922 // the type's granularity rather than a loss introduced
5923 // here, and no PG application can reach the shape, because
5924 // PG rejects it. Pinned at eight shapes in
5925 // `e2e_avg_money_round664`.
5926 let n = i128::from(st.num.count);
5927 let q =
5928 (st.num.sum_money * 2 + if st.num.sum_money >= 0 { n } else { -n }) / (2 * n);
5929 Value::Money(q as i64)
5930 } else if st.num.use_numeric {
5931 // v7.38 (read01, T6.P3) — avg of a special is that special
5932 // (NaN→NaN, ±Inf→±Inf); PG matches.
5933 if st.num.sum_num_kind != spg_storage::NumericKind::Finite {
5934 Value::numeric_special(st.num.sum_num_kind)
5935 } else if let Some(big) = &st.num.sum_big {
5936 // v7.39 (read01 numeric.c) — bignum avg = spilled sum /
5937 // count at PG's division display scale.
5938 use spg_storage::bignum::BigNumeric;
5939 let sum_tot = big.add(&BigNumeric::from_i128(i128::from(st.num.sum_int), 0));
5940 let cnt = BigNumeric::from_i128(i128::from(st.num.count), 0);
5941 let rscale = crate::numeric::division_display_scale_big(&sum_tot, &cnt);
5942 match sum_tot.div(&cnt, rscale) {
5943 Some(q) => crate::eval::binop::bignum_to_value(q),
5944 None => Value::Null,
5945 }
5946 } else {
5947 let (sum_scaled, sum_scale) = crate::numeric::numeric_add(
5948 st.num.sum_num_scaled,
5949 st.num.sum_num_scale,
5950 i128::from(st.num.sum_int),
5951 0,
5952 );
5953 let (scaled, scale) = crate::numeric::numeric_avg(
5954 sum_scaled,
5955 sum_scale,
5956 i128::from(st.num.count),
5957 );
5958 Value::Numeric {
5959 scaled,
5960 scale,
5961 kind: spg_storage::NumericKind::Finite,
5962 }
5963 }
5964 } else if st.num.use_float {
5965 Value::Float((st.num.sum_float + (st.num.sum_int as f64)) / (st.num.count as f64))
5966 } else {
5967 // v7.38 (read01, T4) — avg over integer input is exact NUMERIC
5968 // (PG: avg(int)/avg(bigint) → numeric), at PG's division display
5969 // scale. sum(int) is unaffected (it reads sum_int as BigInt).
5970 let (scaled, scale) = crate::numeric::numeric_avg(
5971 i128::from(st.num.sum_int),
5972 0,
5973 i128::from(st.num.count),
5974 );
5975 Value::Numeric {
5976 scaled,
5977 scale,
5978 kind: spg_storage::NumericKind::Finite,
5979 }
5980 }
5981 }
5982 "min" | "max" | "any_value" => st.extreme.clone().unwrap_or(Value::Null),
5983 // PG: range_agg over an empty group is NULL; all-empty
5984 // ranges finalize to the empty multirange {}.
5985 // v7.39 (round 231) — range_agg collects its inputs verbatim while
5986 // accumulating; PG's result is a *normalized* multirange, so the
5987 // spans are sorted, merged where they overlap or abut, and emptied
5988 // ones dropped exactly once, here. Without this
5989 // `range_agg` over `[1,3),[5,9),[2,6)` answered all three spans
5990 // where PG answers the single `{[1,9)}` they cover.
5991 "range_agg" => match st.extreme.clone() {
5992 Some(Value::Multirange { kind, ranges }) => Value::Multirange {
5993 kind,
5994 ranges: crate::eval::binop::normalize_multirange_spans(kind, &ranges),
5995 },
5996 other => other.unwrap_or(Value::Null),
5997 },
5998 "range_intersect_agg" => st.extreme.clone().unwrap_or(Value::Null),
5999 // v7.17.0 — string_agg: join all collected text items with
6000 // the captured separator. Empty / all-NULL group → NULL
6001 // (PG semantics).
6002 "string_agg" | "group_concat" | "xmlagg" => {
6003 if st.items.is_empty() {
6004 return Value::Null;
6005 }
6006 // group_concat defaults to ',' (MySQL); xmlagg and a
6007 // separator-less string_agg join bare.
6008 let sep = st.separator.clone().unwrap_or_else(|| {
6009 if name == "group_concat" {
6010 ",".into()
6011 } else {
6012 String::new()
6013 }
6014 });
6015 // v7.39 (round 762, F31-C2) — per-row separators, when the
6016 // accumulate path carried them (aligned with items).
6017 let per_row: Option<&[Option<String>]> =
6018 if !st.item_seps.is_empty() && st.item_seps.len() == st.items.len() {
6019 Some(&st.item_seps)
6020 } else {
6021 None
6022 };
6023 let mut out = String::new();
6024 for (i, item) in st.items.iter().enumerate() {
6025 if i > 0 {
6026 match per_row {
6027 Some(seps) => {
6028 if let Some(sp) = &seps[i] {
6029 out.push_str(sp);
6030 }
6031 }
6032 None => out.push_str(&sep),
6033 }
6034 }
6035 match item {
6036 Value::Text(s) => out.push_str(s),
6037 // MySQL group_concat coerces scalars to text;
6038 // harmless for string_agg (typed inputs are
6039 // Text already).
6040 Value::Int(n) => out.push_str(&n.to_string()),
6041 Value::BigInt(n) => out.push_str(&n.to_string()),
6042 Value::SmallInt(n) => out.push_str(&n.to_string()),
6043 Value::Float(f) => out.push_str(&f.to_string()),
6044 Value::Bool(b) => {
6045 out.push_str(if *b { "1" } else { "0" });
6046 }
6047 _ => {}
6048 }
6049 }
6050 Value::text(out)
6051 }
6052 // v7.17.0 — array_agg: collect into a typed array. NULL
6053 // elements are preserved per PG. Result type is decided
6054 // by the first non-NULL element seen (or Text fallback
6055 // when the whole group is NULL — PG would surface the
6056 // declared input type, but SPG hasn't yet wired the
6057 // aggregate's static input-type from `describe`).
6058 // v7.39 (read01 round 73) — ONE builder, shared with the `ARRAY[…]`
6059 // literal. This finalize used to dispatch on the first non-NULL element
6060 // with arms for int and bigint and a text fallback for everything else,
6061 // so `array_agg(bool_col)` came back as text[] — the same fallback-in-
6062 // place-of-a-decision that rounds 71/72 dug out of the literal path and
6063 // the array functions. Fifth site; now there is only one.
6064 "array_agg" => {
6065 if st.items.is_empty() {
6066 return Value::Null;
6067 }
6068 crate::eval::values::build_array_from_values(&st.items)
6069 }
6070 "bool_and" | "bool_or" => st.bool_acc.map_or(Value::Null, Value::Bool),
6071 // v7.32 (round-29) — variance / stddev. PG: `variance` ==
6072 // `var_samp`, `stddev` == `stddev_samp`. samp needs n >= 2
6073 // (n < 2 → NULL); pop needs n >= 1 (n == 1 → 0).
6074 "variance" | "var_samp" | "var_pop" | "stddev" | "stddev_samp" | "stddev_pop" => {
6075 let n = st.num.count;
6076 if n == 0 {
6077 return Value::Null;
6078 }
6079 let nf = n as f64;
6080 // v7.39 (round 381) — MySQL's bare STDDEV / VARIANCE are the
6081 // POPULATION statistics (`STDDEV` = `STDDEV_POP`, `VARIANCE` =
6082 // `VAR_POP` on MariaDB 11), where PG's bare forms are the
6083 // SAMPLE ones. `_samp` / `_pop` are explicit and unchanged.
6084 let pop = name.ends_with("_pop") || (mysql && (name == "stddev" || name == "variance"));
6085 if !pop && n < 2 {
6086 // var_samp / stddev (samp) with n == 1 → NULL.
6087 return Value::Null;
6088 }
6089 // v7.38 (read01) — over exact inputs PG's numeric overload applies:
6090 // variance = (N·Σx² − (Σx)²) / (N² | N·(N−1)) using numeric division's
6091 // display scale, and stddev is its numeric sqrt. Falls through to the
6092 // f64 path (a double result, PG's float8 overload) on a float input.
6093 if !st.stddev_saw_float {
6094 // v7.39 (round 615) — fold whatever the i128 accumulator holds
6095 // into the exact pair, once, here.
6096 if let Some((sum, sum_sq)) = stddev_exact_pair(st) {
6097 let (sum, sum_sq) = (&sum, &sum_sq);
6098 use spg_storage::bignum::BigNumeric as BN;
6099 let nb = BN::from_i128(i128::from(n), 0);
6100 let numerator = nb.mul(sum_sq).sub(&sum.mul(sum));
6101 let divisor = if pop {
6102 nb.mul(&nb)
6103 } else {
6104 nb.mul(&BN::from_i128(i128::from(n - 1), 0))
6105 };
6106 // PG returns a bare `0` (scale 0) for a zero / clamped-negative
6107 // numerator rather than the division's padded zero.
6108 if numerator.is_zero() || numerator.parts().0 {
6109 return Value::Numeric {
6110 scaled: 0,
6111 scale: 0,
6112 kind: spg_storage::NumericKind::Finite,
6113 };
6114 }
6115 let rscale = crate::numeric::division_display_scale_big(&numerator, &divisor);
6116 if let Some(var) = numerator.div(&divisor, rscale) {
6117 let out = if name.starts_with("stddev") {
6118 var.sqrt(crate::numeric::sqrt_display_scale_big(&var))
6119 } else {
6120 Some(var)
6121 };
6122 if let Some(o) = out {
6123 return crate::eval::binop::bignum_to_value(o);
6124 }
6125 }
6126 }
6127 }
6128 // Match PG's float8 accumulator operation order exactly
6129 // (utils/adt/float.c float8_var_pop / _samp): the numerator
6130 // is `N*Σx² - (Σx)²` and the divisor is `N²` (pop) or
6131 // `N*(N-1)` (samp). SPG previously used the algebraically
6132 // equal `(Σx² - (Σx)²/N) / denom`, whose different float
6133 // rounding drifted a ULP from PG on stddev (only masked
6134 // before by an imprecise hand-rolled sqrt).
6135 let numerator = (nf * st.sum_sq - st.num.sum_float * st.num.sum_float).max(0.0);
6136 let divisor = if pop { nf * nf } else { nf * (nf - 1.0) };
6137 let var = numerator / divisor;
6138 let result = if name.starts_with("stddev") {
6139 crate::eval::f64_sqrt(var)
6140 } else {
6141 var
6142 };
6143 // A float input resolves PG's float8 overload → double precision.
6144 Value::Float(result)
6145 }
6146 // v7.32 (round-29) — bitwise aggregates: None (empty / all-NULL)
6147 // → SQL NULL.
6148 "bit_and" | "bit_or" | "bit_xor" => st.bit_acc.map_or(Value::Null, |acc| {
6149 if st.bit_wide {
6150 Value::BigInt(acc)
6151 } else {
6152 Value::Int(acc as i32)
6153 }
6154 }),
6155 // v7.32 (round-29) — regression family. `regr_count` is the
6156 // paired n; everything else is NULL over an empty set. Terms
6157 // are the mean-centred sums of squares / cross-products.
6158 "regr_count" => Value::BigInt(st.reg_n),
6159 "covar_pop" | "covar_samp" | "corr" | "regr_avgx" | "regr_avgy" | "regr_slope"
6160 | "regr_intercept" | "regr_r2" | "regr_sxx" | "regr_syy" | "regr_sxy" => {
6161 let n = st.reg_n;
6162 if n == 0 {
6163 return Value::Null;
6164 }
6165 let nf = n as f64;
6166 // v7.39 (read01 round 115) — Sxx / Syy / Sxy are now the
6167 // Youngs-Cramer running deviation sums (accumulated above), so they
6168 // are used directly rather than re-derived from the raw squares.
6169 let sxx = st.reg_sxx;
6170 let syy = st.reg_syy;
6171 let sxy = st.reg_sxy;
6172 let avgx = st.reg_sx / nf;
6173 let avgy = st.reg_sy / nf;
6174 let out = match name {
6175 "regr_avgx" => Some(avgx),
6176 "regr_avgy" => Some(avgy),
6177 "regr_sxx" => Some(sxx),
6178 "regr_syy" => Some(syy),
6179 "regr_sxy" => Some(sxy),
6180 "covar_pop" => Some(sxy / nf),
6181 "covar_samp" => (n >= 2).then(|| sxy / (nf - 1.0)),
6182 "regr_slope" => (sxx != 0.0).then(|| sxy / sxx),
6183 "regr_intercept" => (sxx != 0.0).then(|| avgy - (sxy / sxx) * avgx),
6184 "corr" => {
6185 let d = sxx * syy;
6186 (d > 0.0).then(|| sxy / crate::eval::f64_sqrt(d))
6187 }
6188 // PG: NULL when sxx==0; 1 when syy==0 (and sxx>0).
6189 "regr_r2" => {
6190 if sxx == 0.0 {
6191 None
6192 } else if syy == 0.0 {
6193 Some(1.0)
6194 } else {
6195 Some((sxy * sxy) / (sxx * syy))
6196 }
6197 }
6198 _ => None,
6199 };
6200 out.map_or(Value::Null, Value::Float)
6201 }
6202 // v7.32 (round-29) — json_agg / jsonb_agg: a JSON array of every
6203 // collected element in row order; empty set → SQL NULL.
6204 "json_agg" | "jsonb_agg" | "json_arrayagg" | "json_agg_strict" | "jsonb_agg_strict" => {
6205 if st.items.is_empty() {
6206 return Value::Null;
6207 }
6208 let mut out = String::from("[");
6209 for (i, item) in st.items.iter().enumerate() {
6210 if i > 0 {
6211 out.push_str(", ");
6212 }
6213 out.push_str(&crate::json::value_to_json_text(item));
6214 }
6215 out.push(']');
6216 // jsonb_agg yields canonical jsonb (nested object keys sorted,
6217 // numbers normalised); json_agg keeps the input verbatim.
6218 let result = Value::json(out);
6219 if name.starts_with("jsonb_agg") {
6220 crate::json::canonicalize_value(result)
6221 } else {
6222 result
6223 }
6224 }
6225 // v7.32 (round-29) — json_object_agg: a JSON object built from
6226 // the parallel key (`items`) / value (`aux_items`) streams.
6227 "json_object_agg"
6228 | "jsonb_object_agg"
6229 | "json_objectagg"
6230 | "json_object_agg_strict"
6231 | "jsonb_object_agg_strict"
6232 | "json_object_agg_unique"
6233 | "jsonb_object_agg_unique"
6234 | "json_object_agg_unique_strict"
6235 | "jsonb_object_agg_unique_strict" => {
6236 if st.items.is_empty() {
6237 return Value::Null;
6238 }
6239 // Object keys are always JSON strings (PG coerces).
6240 let key_text = |key: &Value| -> String {
6241 match key {
6242 Value::Text(s) | Value::Json(s) => s.to_string(),
6243 other => crate::json::value_to_json_text(other),
6244 }
6245 };
6246 // jsonb dedups keys keeping the last value (jsonb is a
6247 // map); json preserves every pair including duplicates.
6248 let dedup = name.starts_with("jsonb_object_agg");
6249 // (key, value-index) pairs in first-seen key order; for
6250 // jsonb a repeated key updates its value-index in place.
6251 let mut pairs: Vec<(String, usize)> = Vec::with_capacity(st.items.len());
6252 for (i, key) in st.items.iter().enumerate() {
6253 let kt = key_text(key);
6254 if dedup {
6255 if let Some(slot) = pairs.iter_mut().find(|(k, _)| *k == kt) {
6256 slot.1 = i;
6257 continue;
6258 }
6259 }
6260 pairs.push((kt, i));
6261 }
6262 // v7.39 (read01 json.c) — PG's json_object_agg emits the
6263 // distinctive "{ \"k\" : v, ... }" spacing (jsonb variants
6264 // canonicalize it away below).
6265 let mut out = String::from("{ ");
6266 for (n, (kt, i)) in pairs.iter().enumerate() {
6267 if n > 0 {
6268 out.push_str(", ");
6269 }
6270 out.push_str(&crate::json::value_to_json_text(&Value::text(kt.clone())));
6271 out.push_str(" : ");
6272 let val = st.aux_items.get(*i).unwrap_or(&Value::Null);
6273 out.push_str(&crate::json::value_to_json_text(val));
6274 }
6275 out.push_str(" }");
6276 // jsonb_object_agg emits canonical jsonb — keys sorted by PG's
6277 // (length, byte) order; json_object_agg keeps first-seen order.
6278 let result = Value::json(out);
6279 if dedup {
6280 crate::json::canonicalize_value(result)
6281 } else {
6282 result
6283 }
6284 }
6285 // Ordered-set aggregates are finalized in `run` (they need the
6286 // sorted items + the direct fraction argument), never here.
6287 _ => unreachable!(),
6288 }
6289}
6290
6291/// v7.32 (round-29) — numeric coercion for the percentile interpolation.
6292fn agg_value_to_f64(v: &Value) -> Option<f64> {
6293 match v {
6294 Value::Int(n) => Some(f64::from(*n)),
6295 Value::SmallInt(n) => Some(f64::from(*n)),
6296 Value::BigInt(n) => Some(*n as f64),
6297 Value::Float(x) => Some(*x),
6298 Value::Real(x) => Some(f64::from(*x)),
6299 Value::Numeric { scaled, scale, .. } => Some(numeric_to_f64(*scaled, *scale)),
6300 _ => None,
6301 }
6302}
6303
6304/// The array form of a `percentile_cont/disc` direct argument
6305/// (`percentile_cont(ARRAY[0.25,0.5,0.75])`), as f64 fractions. `None` when the
6306/// direct argument is a plain scalar fraction. A NULL element stays `None` —
6307/// PG yields a NULL result element for it.
6308fn percentile_fraction_array(v: Option<&Value>) -> Option<Vec<Option<f64>>> {
6309 match v? {
6310 Value::FloatArray(a) => Some(a.clone()),
6311 Value::NumericArray(a) => Some(
6312 a.iter()
6313 .map(|x| x.map(|(scaled, scale)| numeric_to_f64(scaled, scale)))
6314 .collect(),
6315 ),
6316 Value::IntArray(a) => Some(a.iter().map(|x| x.map(f64::from)).collect()),
6317 // Array literals (`ARRAY[0.25,0.5,0.75]`) evaluate to a TextArray of the
6318 // element renderings; parse each back to f64.
6319 Value::TextArray(a) => Some(
6320 a.iter()
6321 .map(|x| x.as_deref().and_then(|s| s.parse::<f64>().ok()))
6322 .collect(),
6323 ),
6324 _ => None,
6325 }
6326}
6327
6328/// Build an array Value from a list of scalar values, dispatching on the first
6329/// non-NULL element's type (mirrors array_agg's finalize). Used by the array
6330/// form of `percentile_disc`, whose result is an array of the ordered-column
6331/// element type.
6332fn values_to_array(picked: &[Value<'_>]) -> Value<'static> {
6333 let owned: alloc::vec::Vec<Value<'static>> =
6334 picked.iter().map(|v| v.clone().into_owned()).collect();
6335 crate::eval::values::build_array_from_values(&owned)
6336}
6337
6338/// NUMERIC → f64 for the float-math aggregates (stddev / variance / corr /
6339/// percentile_cont). `scaled × 10^-scale`; `10^scale` fits in i128 for the
6340/// NUMERIC scale range, so no `f64::powi` (unavailable under no_std) is needed.
6341#[allow(clippy::cast_precision_loss)]
6342fn numeric_to_f64(scaled: i128, scale: u16) -> f64 {
6343 (scaled as f64) / (10i128.pow(u32::from(scale)) as f64)
6344}
6345
6346/// v7.32 (round-29) — finalize a WITHIN GROUP aggregate. `st.items` is
6347/// already sorted by the `WITHIN GROUP (ORDER BY …)` spec. `direct` is
6348/// the evaluated direct argument: the fraction for `percentile_*`, the
6349/// first hypothetical value for the hypothetical-set family (`rank`
6350/// etc. — `direct_extra` carries the rest of a multi-key call), and
6351/// unused by `mode`. `order_by` is the sort spec; the hypothetical-set
6352/// family compares in the sort direction (multi-key via `st.item_keys`).
6353#[allow(
6354 clippy::cast_precision_loss,
6355 clippy::cast_possible_truncation,
6356 clippy::cast_sign_loss,
6357 clippy::too_many_lines
6358)]
6359fn finalize_ordered_set(
6360 name: &str,
6361 st: &AggState,
6362 direct: Option<&Value>,
6363 direct_extra: &[Value<'static>],
6364 order_by: &[spg_sql::ast::OrderBy],
6365 order_collations: &[Option<alloc::string::String>],
6366 mysql: bool,
6367) -> Result<Value<'static>, EvalError> {
6368 let fraction = direct;
6369 // v7.39 (read01 orderedsetaggs.c) — PG validates the percentile
6370 // fraction before looking at the rows (an out-of-range fraction
6371 // errors even over an empty group), and a NULL fraction is NULL.
6372 let check_fraction = |f: f64| -> Result<f64, EvalError> {
6373 if !(0.0..=1.0).contains(&f) || f.is_nan() {
6374 return Err(EvalError::TypeMismatch {
6375 detail: format!("percentile value {f} is not between 0 and 1"),
6376 });
6377 }
6378 Ok(f)
6379 };
6380 let scalar_fraction: Option<Result<f64, EvalError>> =
6381 if matches!(name, "percentile_cont" | "percentile_disc") {
6382 match fraction {
6383 None | Some(Value::Null) => return Ok(Value::Null),
6384 Some(v) => match percentile_fraction_array(Some(v)) {
6385 Some(fracs) => {
6386 for f in fracs.iter().flatten() {
6387 check_fraction(*f)?;
6388 }
6389 None
6390 }
6391 None => Some(
6392 agg_value_to_f64(v)
6393 .ok_or_else(|| EvalError::TypeMismatch {
6394 detail: format!(
6395 "percentile fraction must be numeric, got {}",
6396 crate::conversions::pg_type_name_for_error_opt(v.data_type())
6397 ),
6398 })
6399 .and_then(check_fraction),
6400 ),
6401 },
6402 }
6403 } else {
6404 None
6405 };
6406 let items = &st.items;
6407 if items.is_empty() {
6408 // A hypothetical row ranks first over an empty group; the
6409 // distribution functions are 0 / divide-by-(n+1).
6410 return Ok(match name {
6411 "rank" | "dense_rank" => Value::BigInt(1),
6412 "percent_rank" => Value::Float(0.0),
6413 "cume_dist" => Value::Float(1.0),
6414 _ => Value::Null,
6415 });
6416 }
6417 let n = items.len();
6418 Ok(match name {
6419 // v7.32 (round-29) — hypothetical-set: the rank the direct value
6420 // would have if inserted into the group, in the sort direction.
6421 "rank" | "dense_rank" | "percent_rank" | "cume_dist" => {
6422 let Some(h) = fraction else {
6423 return Ok(Value::Null);
6424 };
6425 // v7.39 (read01 orderedsetaggs.c) — the multi-key form
6426 // compares the hypothetical tuple against the collected
6427 // `item_keys` tuples with the full sort spec.
6428 let kw = order_by.len();
6429 let multi = kw > 1 && st.item_keys.len() == items.len() * kw;
6430 let hv: Vec<Value<'static>> = core::iter::once(h.clone().into_owned())
6431 .chain(direct_extra.iter().cloned())
6432 .collect();
6433 let (desc, nulls_first) = order_by
6434 .first()
6435 .map_or((false, None), |o| (o.desc, o.nulls_first));
6436 let cmp_i = |i: usize| -> core::cmp::Ordering {
6437 if multi {
6438 cmp_order_keys(
6439 order_by,
6440 &[],
6441 order_collations,
6442 &st.item_keys[i * kw..(i + 1) * kw],
6443 &hv,
6444 mysql,
6445 )
6446 } else {
6447 crate::order_by_value_cmp_in(desc, nulls_first, &items[i], h, mysql)
6448 }
6449 };
6450 let mut before: Vec<usize> = Vec::new(); // sort strictly before h
6451 let mut before_or_eq = 0usize; // sort before-or-peer with h
6452 for i in 0..n {
6453 match cmp_i(i) {
6454 core::cmp::Ordering::Less => {
6455 before.push(i);
6456 before_or_eq += 1;
6457 }
6458 core::cmp::Ordering::Equal => before_or_eq += 1,
6459 core::cmp::Ordering::Greater => {}
6460 }
6461 }
6462 // PG divides by the FULL input size (NULL rows included);
6463 // `n` counts only the non-NULL values `items` holds.
6464 let nn = st.within_group_rows.max(n) as f64;
6465 match name {
6466 "rank" => Value::BigInt((before.len() + 1) as i64),
6467 "dense_rank" => {
6468 // Count distinct sort-key tuples among the strictly-
6469 // before rows (items arrive unsorted relative to
6470 // item_keys in the multi-key form, so sort + dedup).
6471 let tuple_cmp = |&x: &usize, &y: &usize| -> core::cmp::Ordering {
6472 if multi {
6473 cmp_order_keys(
6474 order_by,
6475 &[],
6476 order_collations,
6477 &st.item_keys[x * kw..(x + 1) * kw],
6478 &st.item_keys[y * kw..(y + 1) * kw],
6479 mysql,
6480 )
6481 } else {
6482 value_cmp(&items[x], &items[y])
6483 }
6484 };
6485 let mut sorted = before.clone();
6486 sorted.sort_by(tuple_cmp);
6487 let mut distinct = 0usize;
6488 for (k, &i) in sorted.iter().enumerate() {
6489 if k == 0 || tuple_cmp(&sorted[k - 1], &i) != core::cmp::Ordering::Equal {
6490 distinct += 1;
6491 }
6492 }
6493 Value::BigInt((distinct + 1) as i64)
6494 }
6495 "percent_rank" => Value::Float(before.len() as f64 / nn),
6496 "cume_dist" => Value::Float((before_or_eq as f64 + 1.0) / (nn + 1.0)),
6497 _ => unreachable!(),
6498 }
6499 }
6500 // Most frequent value; equal values are adjacent in the sorted
6501 // run, and a frequency tie resolves to the earliest run (the
6502 // smallest value under an ascending sort), matching PG.
6503 "mode" => {
6504 let (mut best_i, mut best_cnt) = (0usize, 1usize);
6505 let (mut run_i, mut run_cnt) = (0usize, 1usize);
6506 for i in 1..n {
6507 if value_cmp(&items[i], &items[run_i]) == core::cmp::Ordering::Equal {
6508 run_cnt += 1;
6509 } else {
6510 run_i = i;
6511 run_cnt = 1;
6512 }
6513 if run_cnt > best_cnt {
6514 best_cnt = run_cnt;
6515 best_i = run_i;
6516 }
6517 }
6518 items[best_i].clone()
6519 }
6520 // The first value whose cumulative fraction reaches `f`. PG accepts
6521 // both a scalar fraction (→ the element) and an array of fractions (→
6522 // an array of the ordered-column element type, with NULL fractions
6523 // yielding NULL elements).
6524 "percentile_disc" => {
6525 let idx_at = |f: f64| -> usize {
6526 if f <= 0.0 {
6527 0
6528 } else {
6529 (crate::eval::f64_ceil(f * n as f64) as usize)
6530 .saturating_sub(1)
6531 .min(n - 1)
6532 }
6533 };
6534 if let Some(fracs) = percentile_fraction_array(fraction) {
6535 let picked: Vec<Value> = fracs
6536 .iter()
6537 .map(|f| f.map_or(Value::Null, |f| items[idx_at(f)].clone()))
6538 .collect();
6539 return Ok(values_to_array(&picked));
6540 }
6541 let f = scalar_fraction.transpose()?.unwrap_or(0.0);
6542 items[idx_at(f)].clone()
6543 }
6544 // Linear interpolation between the two bracketing values. PG accepts
6545 // both a scalar fraction (→ float) and an array of fractions (→ a
6546 // float array, one interpolated value per requested percentile).
6547 "percentile_cont" => {
6548 // v7.39 (read01 orderedsetaggs.c) — the INTERVAL overload
6549 // interpolates component-wise with PG's month→day→time
6550 // remainder spill (a month is 30 days, a day 86400 s).
6551 if items.iter().all(|v| matches!(v, Value::Interval { .. })) {
6552 let iv = |i: usize| -> (f64, f64, f64) {
6553 match &items[i] {
6554 Value::Interval {
6555 months,
6556 days,
6557 micros,
6558 } => (f64::from(*months), f64::from(*days), *micros as f64),
6559 _ => unreachable!(),
6560 }
6561 };
6562 let at = |f: f64| -> Value<'static> {
6563 if n == 1 {
6564 return items[0].clone();
6565 }
6566 let rank = f * (n as f64 - 1.0);
6567 let lo = crate::eval::f64_floor(rank) as usize;
6568 let hi = crate::eval::f64_ceil(rank) as usize;
6569 let frac = rank - lo as f64;
6570 let (lm, ld, lu) = iv(lo);
6571 let (hm, hd, hu) = iv(hi);
6572 let dm = (hm - lm) * frac;
6573 let m_i = dm as i64; // trunc toward zero
6574 let rem_days = (dm - m_i as f64) * 30.0 + (hd - ld) * frac;
6575 let d_i = rem_days as i64;
6576 let us = (rem_days - d_i as f64) * 86_400_000_000.0 + (hu - lu) * frac;
6577 Value::Interval {
6578 months: (lm as i64 + m_i) as i32,
6579 days: (ld as i64 + d_i) as i32,
6580 micros: lu as i64 + libm::round(us) as i64,
6581 }
6582 };
6583 if let Some(fracs) = percentile_fraction_array(fraction) {
6584 let picked: Vec<Value> =
6585 fracs.iter().map(|f| f.map_or(Value::Null, at)).collect();
6586 return Ok(values_to_array(&picked));
6587 }
6588 let f = scalar_fraction.transpose()?.unwrap_or(0.0);
6589 return Ok(at(f));
6590 }
6591 let Some(nums) = items
6592 .iter()
6593 .map(agg_value_to_f64)
6594 .collect::<Option<Vec<f64>>>()
6595 else {
6596 return Ok(Value::Null); // non-numeric ordered set
6597 };
6598 let at = |f: f64| -> f64 {
6599 if n == 1 {
6600 return nums[0];
6601 }
6602 let rank = f * (n as f64 - 1.0);
6603 let lo = crate::eval::f64_floor(rank) as usize;
6604 let hi = crate::eval::f64_ceil(rank) as usize;
6605 let frac = rank - lo as f64;
6606 nums[lo] + (nums[hi] - nums[lo]) * frac
6607 };
6608 if let Some(fracs) = percentile_fraction_array(fraction) {
6609 return Ok(Value::FloatArray(fracs.iter().map(|f| f.map(at)).collect()));
6610 }
6611 let f = scalar_fraction.transpose()?.unwrap_or(0.0);
6612 Value::Float(at(f))
6613 }
6614 _ => unreachable!(),
6615 })
6616}
6617
6618fn infer_agg_type(spec: &AggSpec, schema_cols: &[ColumnSchema]) -> DataType {
6619 // v7.26 (round-20 C) — the argument's statically-derived shape
6620 // types MIN/MAX/SUM/array_agg properly; RowDescription used to
6621 // report TEXT for these, breaking every sqlx typed decode.
6622 let arg_ty = spec
6623 .arg
6624 .as_ref()
6625 .and_then(|a| crate::describe::describe_expr(a, schema_cols))
6626 .map(|shape| shape.ty);
6627 // v7.33 (array_agg argmax) — `(array_agg(x ORDER BY y))[1]` yields the
6628 // ELEMENT type (x), not the array type.
6629 if spec.first_ordered {
6630 return arg_ty.unwrap_or(DataType::Text);
6631 }
6632 match spec.name.as_str() {
6633 "count" | "count_star" => DataType::BigInt,
6634 // v7.38 (read01, T4) — sum(int) → bigint, sum(bigint) → numeric (PG
6635 // widens to numeric to defend against i64 overflow), sum(float) → float.
6636 "sum" => match arg_ty {
6637 Some(DataType::Float) => DataType::Float,
6638 Some(DataType::BigInt) => DataType::Numeric {
6639 precision: 0,
6640 scale: 0,
6641 },
6642 _ => DataType::BigInt,
6643 },
6644 // v7.38 (read01, T4) — avg over any integer / numeric input is NUMERIC
6645 // (PG); only avg(float8) stays double precision.
6646 "avg" => match arg_ty {
6647 Some(DataType::Float) => DataType::Float,
6648 _ => DataType::Numeric {
6649 precision: 0,
6650 scale: 0,
6651 },
6652 },
6653 // v7.17.0 — string_agg always returns TEXT.
6654 "string_agg" | "group_concat" | "xmlagg" => DataType::Text,
6655 // v7.39 (read01 round 73) — the STATIC type follows the same rule the
6656 // finalize does, so `pg_typeof(array_agg(b))` is `boolean[]`.
6657 "array_agg" => match arg_ty {
6658 Some(DataType::Int | DataType::SmallInt) => DataType::IntArray,
6659 Some(DataType::BigInt) => DataType::BigIntArray,
6660 Some(DataType::Bool) => DataType::BoolArray,
6661 Some(DataType::Date) => DataType::DateArray,
6662 Some(DataType::Timestamp) => DataType::TimestampArray,
6663 Some(DataType::Timestamptz) => DataType::TimestamptzArray,
6664 Some(DataType::Uuid) => DataType::UuidArray,
6665 Some(DataType::Float) => DataType::FloatArray,
6666 Some(DataType::Numeric { .. }) => DataType::NumericArray,
6667 Some(DataType::Bytes) => DataType::BytesArray,
6668 _ => DataType::TextArray,
6669 },
6670 // v7.17.0 — boolean aggregates always return BOOL (nullable
6671 // — empty / all-NULL group → NULL).
6672 "bool_and" | "bool_or" => DataType::Bool,
6673 // v7.32 (round-29) — variance / stddev are floating point;
6674 // percentile_cont interpolates to float; the regression family
6675 // (except regr_count) is floating point.
6676 // v7.38 (read01, T4.3) — PG stddev / variance return NUMERIC.
6677 "stddev" | "stddev_samp" | "stddev_pop" | "variance" | "var_samp" | "var_pop" => {
6678 DataType::Numeric {
6679 precision: 0,
6680 scale: 0,
6681 }
6682 }
6683 "percentile_cont" | "covar_pop" | "covar_samp" | "corr" | "regr_avgx" | "regr_avgy"
6684 | "regr_slope" | "regr_intercept" | "regr_r2" | "regr_sxx" | "regr_syy" | "regr_sxy" => {
6685 DataType::Float
6686 }
6687 // v7.32 (round-29) — bitwise aggregates, regr_count, and the
6688 // integer hypothetical-set ranks return an integer.
6689 // v7.38 (read01, T4.4) — bit_and/or/xor return the INPUT integer type
6690 // (PG: bit_and(int) → integer, bit_and(bigint) → bigint).
6691 "bit_and" | "bit_or" | "bit_xor" => match arg_ty {
6692 Some(DataType::SmallInt) => DataType::SmallInt,
6693 Some(DataType::BigInt) => DataType::BigInt,
6694 _ => DataType::Int,
6695 },
6696 "regr_count" | "rank" | "dense_rank" => DataType::BigInt,
6697 // v7.32 (round-29) — hypothetical-set distribution functions.
6698 "percent_rank" | "cume_dist" => DataType::Float,
6699 // v7.32 (round-29) — JSON aggregates return JSON.
6700 "json_agg" | "jsonb_agg" | "json_object_agg" | "jsonb_object_agg" | "json_arrayagg"
6701 | "json_objectagg" => DataType::Json,
6702 // min/max, percentile_disc, mode, and anything pass-through:
6703 // the argument's shape (for ordered-set aggs `spec.arg` is the
6704 // WITHIN GROUP value expression).
6705 _ => arg_ty.unwrap_or(DataType::Text),
6706 }
6707}
6708
6709fn agg_or_group_type(e: &Expr, synth: &[ColumnSchema]) -> DataType {
6710 if let Expr::Column(c) = e
6711 && let Some(s) = synth.iter().find(|s| s.name == c.name)
6712 {
6713 return s.ty;
6714 }
6715 // v7.26 (round-20 C) — compound expressions over aggregates
6716 // (COALESCE(BOOL_OR(…), false), (array_agg(…))[1], CASE …)
6717 // derive their shape statically against the synth schema; the
6718 // old Text fallback broke sqlx typed decodes of exactly these
6719 // columns.
6720 crate::describe::describe_expr(e, synth)
6721 .map(|shape| shape.ty)
6722 .unwrap_or(DataType::Text)
6723}
6724
6725/// v7.39 (round 620) — PG's strict GROUP BY rule, and the diagnosis it earns.
6726///
6727/// `SELECT id, count(*) FROM dc` answered `column "id" does not exist`. The
6728/// column plainly exists; what it is not is grouped. The message came out that
6729/// way because there was no rule at all — the grouped row carries only the
6730/// grouping keys and the aggregates, so the reference simply failed to resolve
6731/// at evaluation time, and the resolver said the only thing it knew. A user
6732/// reading it goes looking for a typo or a missing table.
6733///
6734/// Returns the first bare column reference that is a real input column, is not
6735/// covered by a grouping expression, and is not inside an aggregate. Variants
6736/// this walker does not descend into are left alone, so an uncovered nesting
6737/// keeps the old behaviour rather than inventing an error: under-reporting is
6738/// the status quo, over-reporting would break queries that run today.
6739fn first_ungrouped_column<'a>(
6740 e: &'a Expr,
6741 group_exprs: &[Expr],
6742 columns: &[ColumnSchema],
6743 licensed: &[alloc::string::String],
6744) -> Option<&'a spg_sql::ast::ColumnName> {
6745 if group_exprs.iter().any(|g| g == e) {
6746 return None;
6747 }
6748 let rec = |x: &'a Expr| first_ungrouped_column(x, group_exprs, columns, licensed);
6749 match e {
6750 Expr::Column(c) => {
6751 (column_ref_is_input(c, columns) && !column_is_key_determined(c, licensed)).then_some(c)
6752 }
6753 // An aggregate's arguments are exactly what does not need grouping.
6754 Expr::FunctionCall { name, .. } if is_aggregate_name(&name.to_ascii_lowercase()) => None,
6755 Expr::AggregateOrdered { .. } => None,
6756 // A subquery carries its own scope and its own rules.
6757 Expr::ScalarSubquery(_) | Expr::Exists { .. } | Expr::InSubquery { .. } => None,
6758 Expr::FunctionCall { args, .. } => args.iter().find_map(rec),
6759 Expr::Binary { lhs, rhs, .. } => rec(lhs).or_else(|| rec(rhs)),
6760 Expr::Unary { expr, .. }
6761 | Expr::Cast { expr, .. }
6762 | Expr::IsNull { expr, .. }
6763 | Expr::BoolTest { expr, .. } => rec(expr),
6764 Expr::Like { expr, pattern, .. } => rec(expr).or_else(|| rec(pattern)),
6765 Expr::InList { expr, list, .. } => rec(expr).or_else(|| list.iter().find_map(rec)),
6766 Expr::Case {
6767 operand,
6768 branches,
6769 else_branch,
6770 } => operand
6771 .as_deref()
6772 .and_then(rec)
6773 .or_else(|| branches.iter().find_map(|(w, t)| rec(w).or_else(|| rec(t))))
6774 .or_else(|| else_branch.as_deref().and_then(rec)),
6775 _ => None,
6776 }
6777}
6778
6779/// v7.39 (round 620) — does this column reference name an INPUT column?
6780///
6781/// A joined schema names its columns `a.s`; a single-table one names them `s`
6782/// and answers to the active alias. Matching only the bare name — which the
6783/// first cut of round 620 did — makes every qualified reference in a join
6784/// invisible to both the check and the rewrite below, which is how they
6785/// reached evaluation and came back `missing FROM-clause entry for table "a"`.
6786fn column_ref_is_input(c: &spg_sql::ast::ColumnName, columns: &[ColumnSchema]) -> bool {
6787 if let Some(q) = &c.qualifier {
6788 let composite = alloc::format!("{q}.{}", c.name);
6789 if columns
6790 .iter()
6791 .any(|col| col.name.eq_ignore_ascii_case(&composite))
6792 {
6793 return true;
6794 }
6795 }
6796 columns
6797 .iter()
6798 .any(|col| col.name.eq_ignore_ascii_case(&c.name))
6799}
6800
6801/// v7.39 (round 620) — the qualifiers whose PRIMARY KEY is wholly present in
6802/// the GROUP BY list, which licenses every OTHER column of those tables.
6803///
6804/// `SELECT s, count(*) FROM dc GROUP BY id` where `id` is the primary key is
6805/// answered by PG and was REFUSED here — a query that runs on PG and fails on
6806/// SPG, which is worse than any wording. One row per `id` means `s` has
6807/// exactly one value in the group, so there is nothing ambiguous to resolve;
6808/// the rule is the SQL standard's functional dependency, and PG applies it for
6809/// a base table's primary key.
6810///
6811/// Every FROM entry is considered separately, so a join licenses the side
6812/// whose key is grouped and not the other: `SELECT a.s, b.t … JOIN … GROUP BY
6813/// a.id` answers `a.s` and still refuses `b.t`, which is what PG does.
6814///
6815/// The empty string stands for the unqualified single-table case.
6816fn qualifiers_grouped_by_primary_key(
6817 stmt: &SelectStatement,
6818 group_exprs: &[Expr],
6819 columns: &[ColumnSchema],
6820 catalog: Option<&spg_storage::Catalog>,
6821) -> Vec<alloc::string::String> {
6822 let (Some(from), Some(cat)) = (stmt.from.as_ref(), catalog) else {
6823 return Vec::new();
6824 };
6825 let mut out = Vec::new();
6826 let refs = core::iter::once(&from.primary).chain(from.joins.iter().map(|j| &j.table));
6827 let single = from.joins.is_empty();
6828 for tr in refs {
6829 if tr.unnest_expr.is_some() {
6830 continue;
6831 }
6832 let Some(table) = cat.get(&tr.name) else {
6833 continue;
6834 };
6835 let schema = table.schema();
6836 let Some(pk) = schema
6837 .uniqueness_constraints
6838 .iter()
6839 .find(|u| u.is_primary_key && !u.columns.is_empty())
6840 else {
6841 continue;
6842 };
6843 let qual = tr.alias.as_deref().unwrap_or(tr.name.as_str());
6844 let all_keys_grouped = pk.columns.iter().all(|&pos| {
6845 let Some(name) = schema.columns.get(pos).map(|c| &c.name) else {
6846 return false;
6847 };
6848 // The key column has to be grouped by AS ITSELF, and as this
6849 // table's: an unqualified spelling only counts when there is one
6850 // table for it to mean.
6851 group_exprs.iter().any(|g| match g {
6852 Expr::Column(c) if c.name.eq_ignore_ascii_case(name) => {
6853 let belongs = match &c.qualifier {
6854 Some(q) => q.eq_ignore_ascii_case(qual),
6855 None => single,
6856 };
6857 belongs && column_ref_is_input(c, columns)
6858 }
6859 _ => false,
6860 })
6861 });
6862 if all_keys_grouped {
6863 out.push(alloc::string::String::from(qual));
6864 if single {
6865 out.push(alloc::string::String::new());
6866 }
6867 }
6868 }
6869 out
6870}
6871
6872/// True when this column reference is licensed by one of those keys.
6873fn column_is_key_determined(
6874 c: &spg_sql::ast::ColumnName,
6875 licensed: &[alloc::string::String],
6876) -> bool {
6877 let q = c.qualifier.as_deref().unwrap_or("");
6878 licensed.iter().any(|l| l.eq_ignore_ascii_case(q))
6879}
6880
6881/// v7.39 (round 405) — MySQL's loose GROUP BY: a non-aggregated column
6882/// that is not in GROUP BY is allowed and reads any (the first-seen) row's
6883/// value in the group. PG (and SPG until now) rejects it. Wrapping such a
6884/// bare column in `any_value(col)` reuses the existing aggregate machinery.
6885/// A whole grouping expression stays as-is; an aggregate call is not
6886/// descended into (its inner columns are already fine); a non-aggregate
6887/// function's argument columns are wrapped individually
6888/// (`UPPER(name)` → `UPPER(any_value(name))`).
6889fn wrap_loose_group_columns(
6890 e: Expr,
6891 group_exprs: &[Expr],
6892 columns: &[ColumnSchema],
6893 // v7.39 (round 620) — `None` wraps every ungrouped column, which is what
6894 // MySQL's loose GROUP BY means. `Some(quals)` wraps only the columns a
6895 // grouped primary key determines, so a join licenses the side whose key is
6896 // grouped and leaves the other to be refused.
6897 licensed: Option<&[alloc::string::String]>,
6898) -> Expr {
6899 if group_exprs.iter().any(|g| *g == e) {
6900 return e;
6901 }
6902 let wrap = |x: Expr| wrap_loose_group_columns(x, group_exprs, columns, licensed);
6903 match e {
6904 Expr::Column(c) => {
6905 let claimed = column_ref_is_input(&c, columns)
6906 && licensed.is_none_or(|l| column_is_key_determined(&c, l));
6907 if claimed {
6908 Expr::FunctionCall {
6909 name: String::from("any_value"),
6910 args: alloc::vec![Expr::Column(c)],
6911 }
6912 } else {
6913 Expr::Column(c)
6914 }
6915 }
6916 Expr::FunctionCall { name, args } if is_aggregate_name(&name.to_ascii_lowercase()) => {
6917 Expr::FunctionCall { name, args }
6918 }
6919 Expr::AggregateOrdered { .. } => e,
6920 Expr::FunctionCall { name, args } => Expr::FunctionCall {
6921 name,
6922 args: args.into_iter().map(wrap).collect(),
6923 },
6924 Expr::Binary { op, lhs, rhs } => Expr::Binary {
6925 op,
6926 lhs: Box::new(wrap(*lhs)),
6927 rhs: Box::new(wrap(*rhs)),
6928 },
6929 Expr::Unary { op, expr } => Expr::Unary {
6930 op,
6931 expr: Box::new(wrap(*expr)),
6932 },
6933 Expr::Cast { expr, target } => Expr::Cast {
6934 expr: Box::new(wrap(*expr)),
6935 target,
6936 },
6937 Expr::IsNull { expr, negated } => Expr::IsNull {
6938 expr: Box::new(wrap(*expr)),
6939 negated,
6940 },
6941 Expr::BoolTest {
6942 expr,
6943 value,
6944 negated,
6945 } => Expr::BoolTest {
6946 expr: Box::new(wrap(*expr)),
6947 value,
6948 negated,
6949 },
6950 Expr::Like {
6951 expr,
6952 pattern,
6953 negated,
6954 case_insensitive,
6955 } => Expr::Like {
6956 expr: Box::new(wrap(*expr)),
6957 pattern: Box::new(wrap(*pattern)),
6958 negated,
6959 case_insensitive,
6960 },
6961 Expr::InList {
6962 expr,
6963 list,
6964 negated,
6965 } => Expr::InList {
6966 expr: Box::new(wrap(*expr)),
6967 list: list.into_iter().map(wrap).collect(),
6968 negated,
6969 },
6970 Expr::Case {
6971 operand,
6972 branches,
6973 else_branch,
6974 } => Expr::Case {
6975 operand: operand.map(|o| Box::new(wrap(*o))),
6976 branches: branches
6977 .into_iter()
6978 .map(|(w, t)| (wrap(w), wrap(t)))
6979 .collect(),
6980 else_branch: else_branch.map(|b| Box::new(wrap(*b))),
6981 },
6982 other => other,
6983 }
6984}
6985
6986/// v7.39 (round 404) — MySQL lets HAVING (and ORDER BY) reference a
6987/// SELECT-list alias (`SELECT g, SUM(v) AS sv … HAVING sv > 30`); PG does
6988/// not. Before the aggregate rewrite, replace a bare `Column(alias)` with
6989/// the SELECT expression it names, so the aggregate rewrite then maps it to
6990/// its synthetic column. A nesting this walker does not cover simply leaves
6991/// the column unresolved (the pre-existing "column does not exist" error),
6992/// never a wrong result.
6993fn substitute_having_aliases(e: Expr, aliases: &[(String, Expr)]) -> Expr {
6994 use spg_sql::ast::ColumnName;
6995 let sub = |x: Expr| substitute_having_aliases(x, aliases);
6996 match e {
6997 Expr::Column(ColumnName {
6998 qualifier: None,
6999 name,
7000 }) => aliases
7001 .iter()
7002 .find(|(a, _)| a.eq_ignore_ascii_case(&name))
7003 .map_or_else(
7004 || {
7005 Expr::Column(ColumnName {
7006 qualifier: None,
7007 name,
7008 })
7009 },
7010 |(_, expr)| expr.clone(),
7011 ),
7012 Expr::Binary { op, lhs, rhs } => Expr::Binary {
7013 op,
7014 lhs: Box::new(sub(*lhs)),
7015 rhs: Box::new(sub(*rhs)),
7016 },
7017 Expr::Unary { op, expr } => Expr::Unary {
7018 op,
7019 expr: Box::new(sub(*expr)),
7020 },
7021 Expr::FunctionCall { name, args } => Expr::FunctionCall {
7022 name,
7023 args: args.into_iter().map(sub).collect(),
7024 },
7025 Expr::IsNull { expr, negated } => Expr::IsNull {
7026 expr: Box::new(sub(*expr)),
7027 negated,
7028 },
7029 Expr::BoolTest {
7030 expr,
7031 value,
7032 negated,
7033 } => Expr::BoolTest {
7034 expr: Box::new(sub(*expr)),
7035 value,
7036 negated,
7037 },
7038 Expr::Like {
7039 expr,
7040 pattern,
7041 negated,
7042 case_insensitive,
7043 } => Expr::Like {
7044 expr: Box::new(sub(*expr)),
7045 pattern: Box::new(sub(*pattern)),
7046 negated,
7047 case_insensitive,
7048 },
7049 Expr::InList {
7050 expr,
7051 list,
7052 negated,
7053 } => Expr::InList {
7054 expr: Box::new(sub(*expr)),
7055 list: list.into_iter().map(sub).collect(),
7056 negated,
7057 },
7058 Expr::Case {
7059 operand,
7060 branches,
7061 else_branch,
7062 } => Expr::Case {
7063 operand: operand.map(|o| Box::new(sub(*o))),
7064 branches: branches
7065 .into_iter()
7066 .map(|(w, t)| (sub(w), sub(t)))
7067 .collect(),
7068 else_branch: else_branch.map(|b| Box::new(sub(*b))),
7069 },
7070 Expr::Cast { expr, target } => Expr::Cast {
7071 expr: Box::new(sub(*expr)),
7072 target,
7073 },
7074 other => other,
7075 }
7076}
7077
7078fn rewrite_expr(e: &Expr, group_exprs: &[Expr], aggs: &[AggSpec]) -> Expr {
7079 // v7.33 (array_agg argmax) — `(array_agg(x ORDER BY y))[1]` rewrites
7080 // to its first_ordered synth column, consuming the subscript. Checked
7081 // before the AggregateOrdered/recursion arms (which would otherwise
7082 // rewrite the inner array_agg and leave the subscript). Same matcher
7083 // as collect_aggregates, so the spec it finds is the one collected.
7084 if let Some((arg, order_by, filter)) = first_ordered_array_agg(e) {
7085 let arg_owned = Some(arg.clone());
7086 let filter_owned = filter.cloned();
7087 for (i, spec) in aggs.iter().enumerate() {
7088 if spec.first_ordered
7089 && spec.name == "array_agg"
7090 && spec.arg == arg_owned
7091 && spec.order_by == *order_by
7092 && spec.filter == filter_owned
7093 {
7094 return Expr::Column(spg_sql::ast::ColumnName {
7095 qualifier: None,
7096 name: format!("__agg_{i}"),
7097 });
7098 }
7099 }
7100 }
7101 // v7.24 (round-16 A) — ordered aggregate: match on the inner
7102 // call PLUS the ordering keys.
7103 if let Expr::AggregateOrdered {
7104 call,
7105 order_by,
7106 distinct,
7107 filter,
7108 } = e
7109 && let Expr::FunctionCall { name, args } = call.as_ref()
7110 {
7111 let lower = name.to_ascii_lowercase();
7112 if is_aggregate_name(&lower) {
7113 let canonical: &str = if lower == "every" { "bool_and" } else { &lower };
7114 // Mirror collect_aggregates: ordered-set aggregates take the
7115 // value from the sort spec and the in-parens arg as direct.
7116 let (arg, direct_arg) = if is_within_group_name(canonical) {
7117 (
7118 order_by.first().map(|o| o.expr.clone()),
7119 args.first().cloned(),
7120 )
7121 } else {
7122 (args.first().cloned(), None)
7123 };
7124 let arg2 = if agg_uses_second_arg(canonical) {
7125 args.get(1).cloned()
7126 } else {
7127 None
7128 };
7129 let filter_owned = filter.as_deref().cloned();
7130 for (i, spec) in aggs.iter().enumerate() {
7131 if spec.name == canonical
7132 && spec.arg == arg
7133 && spec.arg2 == arg2
7134 && spec.distinct == *distinct
7135 && spec.order_by == *order_by
7136 && spec.filter == filter_owned
7137 && spec.direct_arg == direct_arg
7138 {
7139 return Expr::Column(spg_sql::ast::ColumnName {
7140 qualifier: None,
7141 name: format!("__agg_{i}"),
7142 });
7143 }
7144 }
7145 }
7146 }
7147 // Match aggregate FunctionCalls first — they sit outside group_by.
7148 if let Expr::FunctionCall { name, args } = e {
7149 let lower = name.to_ascii_lowercase();
7150 if is_aggregate_name(&lower) {
7151 let arg = if lower == "count_star" {
7152 None
7153 } else {
7154 args.first().cloned()
7155 };
7156 // v7.17.0 — match the spec we registered for
7157 // string_agg(value, separator) on the full pair; v7.32 also
7158 // the regression family and json_object_agg.
7159 let arg2 = if agg_uses_second_arg(&lower) {
7160 args.get(1).cloned()
7161 } else {
7162 None
7163 };
7164 // v7.17.0 — `every` collapses into `bool_and` at
7165 // collection; mirror that here so the rewrite finds
7166 // the matching synth column.
7167 let canonical: &str = if lower == "every" {
7168 "bool_and"
7169 } else {
7170 lower.as_str()
7171 };
7172 for (i, spec) in aggs.iter().enumerate() {
7173 if spec.name == canonical
7174 && spec.arg == arg
7175 && spec.arg2 == arg2
7176 && !spec.distinct
7177 && spec.order_by.is_empty()
7178 {
7179 return Expr::Column(spg_sql::ast::ColumnName {
7180 qualifier: None,
7181 name: format!("__agg_{i}"),
7182 });
7183 }
7184 }
7185 }
7186 }
7187 // Match a group_by expression by AST equality.
7188 for (i, g) in group_exprs.iter().enumerate() {
7189 if g == e {
7190 return Expr::Column(spg_sql::ast::ColumnName {
7191 qualifier: None,
7192 name: format!("__grp_{i}"),
7193 });
7194 }
7195 }
7196 // Recurse into children.
7197 match e {
7198 Expr::NamedArg { name, expr } => Expr::NamedArg {
7199 name: name.clone(),
7200 expr: alloc::boxed::Box::new(rewrite_expr(expr, group_exprs, aggs)),
7201 },
7202 Expr::Variadic(expr) => Expr::Variadic(alloc::boxed::Box::new(rewrite_expr(
7203 expr,
7204 group_exprs,
7205 aggs,
7206 ))),
7207 Expr::AggregateOrdered {
7208 call,
7209 order_by,
7210 distinct,
7211 filter,
7212 } => Expr::AggregateOrdered {
7213 call: Box::new(rewrite_expr(call, group_exprs, aggs)),
7214 distinct: *distinct,
7215 order_by: order_by
7216 .iter()
7217 .map(|o| spg_sql::ast::OrderBy {
7218 expr: rewrite_expr(&o.expr, group_exprs, aggs),
7219 desc: o.desc,
7220 nulls_first: o.nulls_first,
7221 collation: o.collation.clone(),
7222 })
7223 .collect(),
7224 // The filter is evaluated against SOURCE rows during
7225 // accumulation, never against synth rows — keep it as-is.
7226 filter: filter.clone(),
7227 },
7228 Expr::Binary { lhs, op, rhs } => Expr::Binary {
7229 lhs: Box::new(rewrite_expr(lhs, group_exprs, aggs)),
7230 op: *op,
7231 rhs: Box::new(rewrite_expr(rhs, group_exprs, aggs)),
7232 },
7233 Expr::Unary { op, expr } => Expr::Unary {
7234 op: *op,
7235 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7236 },
7237 Expr::Cast { expr, target } => Expr::Cast {
7238 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7239 target: target.clone(),
7240 },
7241 Expr::FieldAccess { base, field } => Expr::FieldAccess {
7242 base: Box::new(rewrite_expr(base, group_exprs, aggs)),
7243 field: field.clone(),
7244 },
7245 Expr::IsNull { expr, negated } => Expr::IsNull {
7246 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7247 negated: *negated,
7248 },
7249 Expr::BoolTest {
7250 expr,
7251 value,
7252 negated,
7253 } => Expr::BoolTest {
7254 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7255 value: *value,
7256 negated: *negated,
7257 },
7258 Expr::FunctionCall { name, args } => Expr::FunctionCall {
7259 name: name.clone(),
7260 args: args
7261 .iter()
7262 .map(|a| rewrite_expr(a, group_exprs, aggs))
7263 .collect(),
7264 },
7265 Expr::Like {
7266 expr,
7267 pattern,
7268 negated,
7269 case_insensitive,
7270 } => Expr::Like {
7271 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7272 pattern: Box::new(rewrite_expr(pattern, group_exprs, aggs)),
7273 negated: *negated,
7274 case_insensitive: *case_insensitive,
7275 },
7276 Expr::Extract { field, source } => Expr::Extract {
7277 field: field.clone(),
7278 source: Box::new(rewrite_expr(source, group_exprs, aggs)),
7279 },
7280 // v7.25.2 (round-19 A) — subquery nodes: rewrite group-key
7281 // references INSIDE the body to `__grp_N` so the correlated
7282 // resolver can substitute them against the synthesised group
7283 // row (aggs are NOT matched inside the body — a COUNT in the
7284 // subquery is the subquery's own aggregate).
7285 Expr::ScalarSubquery(s) => {
7286 Expr::ScalarSubquery(Box::new(rewrite_group_keys_in_select(s, group_exprs)))
7287 }
7288 Expr::Exists { subquery, negated } => Expr::Exists {
7289 subquery: Box::new(rewrite_group_keys_in_select(subquery, group_exprs)),
7290 negated: *negated,
7291 },
7292 Expr::InSubquery {
7293 expr,
7294 subquery,
7295 negated,
7296 } => Expr::InSubquery {
7297 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7298 subquery: Box::new(rewrite_group_keys_in_select(subquery, group_exprs)),
7299 negated: *negated,
7300 },
7301 Expr::RowInSubquery {
7302 row,
7303 subquery,
7304 negated,
7305 } => Expr::RowInSubquery {
7306 row: row
7307 .iter()
7308 .map(|el| rewrite_expr(el, group_exprs, aggs))
7309 .collect(),
7310 subquery: Box::new(rewrite_group_keys_in_select(subquery, group_exprs)),
7311 negated: *negated,
7312 },
7313 Expr::RowCmpSubquery { row, op, subquery } => Expr::RowCmpSubquery {
7314 row: row
7315 .iter()
7316 .map(|el| rewrite_expr(el, group_exprs, aggs))
7317 .collect(),
7318 op: *op,
7319 subquery: Box::new(rewrite_group_keys_in_select(subquery, group_exprs)),
7320 },
7321 // v4.12 window / Literal / Column — clone-pass (these don't
7322 // participate in aggregate rewrite).
7323 Expr::WindowFunction { .. } | Expr::Literal(_) | Expr::Placeholder(_) | Expr::Column(_) => {
7324 e.clone()
7325 }
7326 // v7.10.10 — recurse children for array nodes.
7327 Expr::Array(items) => Expr::Array(
7328 items
7329 .iter()
7330 .map(|elem| rewrite_expr(elem, group_exprs, aggs))
7331 .collect(),
7332 ),
7333 Expr::ArraySubscript { target, index } => Expr::ArraySubscript {
7334 target: Box::new(rewrite_expr(target, group_exprs, aggs)),
7335 index: Box::new(rewrite_expr(index, group_exprs, aggs)),
7336 },
7337 Expr::ArraySlice { target, lo, hi } => Expr::ArraySlice {
7338 target: Box::new(rewrite_expr(target, group_exprs, aggs)),
7339 lo: lo
7340 .as_ref()
7341 .map(|b| Box::new(rewrite_expr(b, group_exprs, aggs))),
7342 hi: hi
7343 .as_ref()
7344 .map(|b| Box::new(rewrite_expr(b, group_exprs, aggs))),
7345 },
7346 Expr::AnyAll {
7347 expr,
7348 op,
7349 array,
7350 is_any,
7351 } => Expr::AnyAll {
7352 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7353 op: *op,
7354 array: Box::new(rewrite_expr(array, group_exprs, aggs)),
7355 is_any: *is_any,
7356 },
7357 Expr::InList {
7358 expr,
7359 list,
7360 negated,
7361 } => Expr::InList {
7362 expr: Box::new(rewrite_expr(expr, group_exprs, aggs)),
7363 list: list
7364 .iter()
7365 .map(|item| rewrite_expr(item, group_exprs, aggs))
7366 .collect(),
7367 negated: *negated,
7368 },
7369 Expr::Case {
7370 operand,
7371 branches,
7372 else_branch,
7373 } => Expr::Case {
7374 operand: operand
7375 .as_deref()
7376 .map(|o| Box::new(rewrite_expr(o, group_exprs, aggs))),
7377 branches: branches
7378 .iter()
7379 .map(|(w, t)| {
7380 (
7381 rewrite_expr(w, group_exprs, aggs),
7382 rewrite_expr(t, group_exprs, aggs),
7383 )
7384 })
7385 .collect(),
7386 else_branch: else_branch
7387 .as_deref()
7388 .map(|e| Box::new(rewrite_expr(e, group_exprs, aggs))),
7389 },
7390 }
7391}
7392
7393/// v7.25.2 (round-19 A) — rewrite group-key references inside a
7394/// subquery body to `__grp_N` synthetic columns (aggregates are
7395/// not touched: empty spec list). Runs through the canonical
7396/// Select walker so every expression slot is covered.
7397fn rewrite_group_keys_in_select(
7398 s: &spg_sql::ast::SelectStatement,
7399 group_exprs: &[Expr],
7400) -> spg_sql::ast::SelectStatement {
7401 let mut out = s.clone();
7402 let _ = crate::walk_select_exprs_mut(&mut out, &mut |e| {
7403 *e = rewrite_expr(e, group_exprs, &[]);
7404 Ok(())
7405 });
7406 out
7407}
7408
7409/// Canonical string key for a tuple of group values. Used as map key.
7410/// Per-value group-key encoding (shared by owned and borrowed paths).
7411fn encode_one(out: &mut String, v: &Value) {
7412 encode_one_in(out, v, false);
7413}
7414
7415/// v7.39 (round 364, M4 P2) — key encoder with the session dialect. On a
7416/// MySQL session a text group / distinct key is FOLDED (accent- and
7417/// case-insensitive) so `Foo`/`foo`/`FOO` share one group and `bar`/`Bär`
7418/// merge — while the group's OUTPUT value stays the first row's original,
7419/// because only the key is folded, not the stored value.
7420fn encode_one_in(out: &mut String, v: &Value, mysql: bool) {
7421 use core::fmt::Write;
7422 if mysql {
7423 if let Value::Text(s) | Value::Json(s) = v {
7424 let _ = write!(out, "S{}|", spg_storage::mysql_compare_fold(s));
7425 return;
7426 }
7427 if let Value::BpChar(s) = v {
7428 let folded = spg_storage::mysql_compare_fold_char(s);
7429 let _ = write!(out, "S{folded}|");
7430 return;
7431 }
7432 }
7433 encode_one_raw(out, v);
7434}
7435
7436fn encode_one_raw(out: &mut String, v: &Value) {
7437 use core::fmt::Write;
7438 match v {
7439 Value::Null => out.push_str("N|"),
7440 // v7.36 (perf — mailrs Phase 1) — switch the integer / float
7441 // encoders to `write!`. `n.to_string()` allocates a fresh
7442 // `String` per cell just to push its bytes into the
7443 // (already-cleared) reuse buffer — for the 25 k-row JOIN
7444 // probe in `count_messages` that's 25 k heap allocs per
7445 // query. `write!(&mut String, ...)` formats straight into
7446 // the buffer; no intermediate alloc.
7447 Value::SmallInt(n) => {
7448 let _ = write!(out, "s{n}|");
7449 }
7450 Value::Int(n) => {
7451 let _ = write!(out, "I{n}|");
7452 }
7453 Value::BigInt(n) => {
7454 let _ = write!(out, "B{n}|");
7455 }
7456 Value::Float(x) => {
7457 // v7.37.16 — fold -0.0 into 0.0: PG's float8 equality (hash and
7458 // btree opclasses) treats them as one value, so GROUP BY /
7459 // DISTINCT must key them together (count(DISTINCT) differential).
7460 // NaN needs no fold — every NaN renders "NaN" here already.
7461 let x = if *x == 0.0 { 0.0 } else { *x };
7462 let _ = write!(out, "F{x}|");
7463 }
7464 Value::Real(x) => {
7465 let x = if *x == 0.0 { 0.0 } else { *x };
7466 let _ = write!(out, "R{x}|");
7467 }
7468 Value::Bool(b) => {
7469 out.push(if *b { 'T' } else { 'f' });
7470 out.push('|');
7471 }
7472 Value::Text(s) => {
7473 out.push('S');
7474 out.push_str(s);
7475 out.push('|');
7476 }
7477 // v7.38 (read01, T11/R3) — bpchar groups / dedups blank-insensitively,
7478 // and shares the text key so `'ab'::char(4)` and `'ab'` co-group.
7479 Value::BpChar(s) => {
7480 out.push('S');
7481 out.push_str(s.trim_end_matches(' '));
7482 out.push('|');
7483 }
7484 Value::Vector(v) => {
7485 out.push('V');
7486 for x in v.iter() {
7487 out.push_str(&x.to_string());
7488 out.push(',');
7489 }
7490 out.push('|');
7491 }
7492 // v6.0.1: GROUP BY on a `VECTOR(N) USING SQ8` column.
7493 // Two cells with byte-identical `(min, max, bytes)`
7494 // share the same group; equivalence is byte-equality
7495 // (same as f32 grouping today — neither path tries to
7496 // normalise nan/-0).
7497 Value::Sq8Vector(q) => {
7498 out.push('Q');
7499 out.push_str(&q.min.to_string());
7500 out.push('@');
7501 out.push_str(&q.max.to_string());
7502 out.push(':');
7503 for b in &q.bytes {
7504 out.push_str(&b.to_string());
7505 out.push(',');
7506 }
7507 out.push('|');
7508 }
7509 // v6.0.3: GROUP BY on a `VECTOR(N) USING HALF` column.
7510 // Byte-equality over the raw u16 bits; matches the SQ8
7511 // path's byte-key model.
7512 Value::HalfVector(h) => {
7513 out.push('H');
7514 for b in &h.bytes {
7515 out.push_str(&b.to_string());
7516 out.push(',');
7517 }
7518 out.push('|');
7519 }
7520 Value::Numeric { scaled, scale, .. } => {
7521 // v7.38 (read01) — DISTINCT keys numerically-equal decimals as one
7522 // regardless of scale (1.0 = 1.00), so strip trailing fractional
7523 // zeros before encoding, matching PG (and set-op / GROUP BY dedup).
7524 let (mut s, mut sc) = (*scaled, *scale);
7525 while sc > 0 && s % 10 == 0 {
7526 s /= 10;
7527 sc -= 1;
7528 }
7529 out.push('D');
7530 out.push_str(&s.to_string());
7531 out.push('@');
7532 out.push_str(&sc.to_string());
7533 out.push('|');
7534 }
7535 Value::Date(d) => {
7536 out.push('d');
7537 out.push_str(&d.to_string());
7538 out.push('|');
7539 }
7540 Value::Timestamp(t) => {
7541 out.push('t');
7542 out.push_str(&t.to_string());
7543 out.push('|');
7544 }
7545 Value::Interval {
7546 months,
7547 days,
7548 micros,
7549 } => {
7550 out.push('i');
7551 out.push_str(&months.to_string());
7552 out.push('m');
7553 out.push_str(&days.to_string());
7554 out.push('d');
7555 out.push_str(µs.to_string());
7556 out.push('|');
7557 }
7558 Value::Json(s) => {
7559 out.push('j');
7560 out.push_str(s);
7561 out.push('|');
7562 }
7563 // v7.5.0 — Value is #[non_exhaustive] for downstream
7564 // forward-compat. Any future variant lacking explicit
7565 // handling here will share a debug-derived group key,
7566 // which is observably wrong but won't crash.
7567 _ => {
7568 out.push('?');
7569 out.push_str(&format!("{v:?}"));
7570 out.push('|');
7571 }
7572 }
7573}
7574
7575/// v7.30 (perf campaign) - encode from borrowed cells without
7576/// materialising an owned Vec<Value<'static>> first.
7577pub(crate) fn encode_key_refs(vals: &[&Value]) -> String {
7578 let mut out = String::new();
7579 for v in vals {
7580 encode_one(&mut out, v);
7581 }
7582 out
7583}
7584
7585/// v7.31 (perf 3e) — encode into a caller-owned scratch buffer.
7586/// The per-row key paths (group hash, DISTINCT set, join build/
7587/// probe) ran 24k+ String allocations per query through the
7588/// allocator just to LOOK UP a map; the scratch form allocates
7589/// only when a map actually has to take ownership (vacant insert).
7590/// v7.39 (round 590) — append ONE value's encoding, for the join key that
7591/// mixes stored cells with computed ones and so cannot clear as it goes.
7592/// v7.39 (round 590, moved here round 593+) — one component of a key with a COMPUTED side.
7593///
7594/// The whole requirement is that two values SQL calls equal encode the same,
7595/// or the join silently loses rows. Across the numeric family that is not
7596/// free: `5` as INT, `5` as BIGINT, `5.0` as double and `5.00` as NUMERIC all
7597/// compare equal and would otherwise carry four different tags, so they are
7598/// all rendered as one canonical decimal. A non-integral value can never
7599/// equal an integer, so it simply renders as itself; NaN equals nothing and
7600/// any encoding will do. Everything outside the numeric family keeps the
7601/// encoder the column-to-column path already uses.
7602pub(crate) fn push_canonical_key(out: &mut String, v: &Value) {
7603 use core::fmt::Write;
7604 match v {
7605 Value::SmallInt(n) => {
7606 let _ = write!(out, "n{n}|");
7607 }
7608 Value::Int(n) => {
7609 let _ = write!(out, "n{n}|");
7610 }
7611 Value::BigInt(n) => {
7612 let _ = write!(out, "n{n}|");
7613 }
7614 // `-0.0` prints with its sign but equals `0`.
7615 Value::Float(f) if *f == 0.0 => out.push_str("n0|"),
7616 Value::Float(f) => {
7617 let _ = write!(out, "n{f}|");
7618 }
7619 Value::Numeric { .. } => {
7620 let t = crate::eval::value_to_text(v);
7621 let t = if t.contains('.') {
7622 t.trim_end_matches('0').trim_end_matches('.')
7623 } else {
7624 t.as_str()
7625 };
7626 let _ = write!(out, "n{t}|");
7627 }
7628 _ => encode_one_into(out, v),
7629 }
7630}
7631
7632/// v7.39 (round 596) — a whole key encoded the canonical way, for the two
7633/// sides of a decorrelated EXISTS: the set is built from the inner column's
7634/// values and probed with the outer EXPRESSION's, and those need not share a
7635/// numeric width for `=` to call them equal.
7636pub(crate) fn encode_canonical_key(vals: &[Value<'_>]) -> String {
7637 let mut out = String::new();
7638 for v in vals {
7639 push_canonical_key(&mut out, v);
7640 }
7641 out
7642}
7643
7644pub(crate) fn encode_one_into(out: &mut String, v: &Value) {
7645 encode_one_raw(out, v);
7646}
7647
7648pub(crate) fn encode_key_refs_into(vals: &[&Value], out: &mut String) {
7649 encode_key_refs_into_in(vals, out, false);
7650}
7651
7652/// v7.38.14 — key encode with a per-POSITION fold decision.
7653///
7654/// `encode_key_refs_into_in` takes one bool for the whole key, which
7655/// cannot express the case a join actually presents: one key column
7656/// declared `COLLATE utf8mb4_bin` beside another that folds. `folds` is
7657/// resolved once per join from the key columns' collations; a short or
7658/// missing entry means "do not fold", which is what every existing
7659/// caller wants.
7660pub(crate) fn encode_key_refs_folded(vals: &[&Value], out: &mut String, folds: &[bool]) {
7661 out.clear();
7662 for (i, v) in vals.iter().enumerate() {
7663 encode_one_in(out, v, folds.get(i).copied().unwrap_or(false));
7664 }
7665}
7666
7667/// v7.39 (round 364, M4 P2) — key encode with the session dialect.
7668pub(crate) fn encode_key_refs_into_in(vals: &[&Value], out: &mut String, mysql: bool) {
7669 out.clear();
7670 for v in vals {
7671 encode_one_in(out, v, mysql);
7672 }
7673}
7674
7675pub(crate) fn encode_key(vals: &[Value<'static>]) -> String {
7676 let mut out = String::new();
7677 for v in vals {
7678 encode_one(&mut out, v);
7679 }
7680 out
7681}
7682
7683#[allow(clippy::cast_precision_loss)]
7684/// v7.37.17 (17.6 siblings) — intersect two ranges (same kind).
7685/// The greater lower bound wins (tie keeps inclusivity only when
7686/// both are inclusive); the smaller upper bound mirrors it; an
7687/// unbounded side loses to a bounded one. lower > upper — or a
7688/// touch that isn't inclusive on both ends — collapses to empty,
7689/// and any empty input pins the fold at empty.
7690fn range_intersect(a: &Value<'static>, b: &Value<'static>) -> Value<'static> {
7691 let (
7692 Value::Range {
7693 kind,
7694 lower: la,
7695 upper: ua,
7696 lower_inc: lia,
7697 upper_inc: uia,
7698 empty: ea,
7699 },
7700 Value::Range {
7701 lower: lb,
7702 upper: ub,
7703 lower_inc: lib_,
7704 upper_inc: uib,
7705 empty: eb,
7706 ..
7707 },
7708 ) = (a, b)
7709 else {
7710 return Value::Null;
7711 };
7712 let kind = *kind;
7713 let empty_range = Value::Range {
7714 kind,
7715 lower: None,
7716 upper: None,
7717 lower_inc: false,
7718 upper_inc: false,
7719 empty: true,
7720 };
7721 if *ea || *eb {
7722 return empty_range;
7723 }
7724 // Greater lower bound (None = -infinity loses to any bound).
7725 let (lower, lower_inc) = match (la, lb) {
7726 (None, None) => (None, false),
7727 (Some(x), None) => (Some(x.clone()), *lia),
7728 (None, Some(y)) => (Some(y.clone()), *lib_),
7729 (Some(x), Some(y)) => match value_cmp(x, y) {
7730 core::cmp::Ordering::Greater => (Some(x.clone()), *lia),
7731 core::cmp::Ordering::Less => (Some(y.clone()), *lib_),
7732 core::cmp::Ordering::Equal => (Some(x.clone()), *lia && *lib_),
7733 },
7734 };
7735 // Smaller upper bound (None = +infinity loses to any bound).
7736 let (upper, upper_inc) = match (ua, ub) {
7737 (None, None) => (None, false),
7738 (Some(x), None) => (Some(x.clone()), *uia),
7739 (None, Some(y)) => (Some(y.clone()), *uib),
7740 (Some(x), Some(y)) => match value_cmp(x, y) {
7741 core::cmp::Ordering::Less => (Some(x.clone()), *uia),
7742 core::cmp::Ordering::Greater => (Some(y.clone()), *uib),
7743 core::cmp::Ordering::Equal => (Some(x.clone()), *uia && *uib),
7744 },
7745 };
7746 if let (Some(lo), Some(up)) = (&lower, &upper) {
7747 match value_cmp(lo, up) {
7748 core::cmp::Ordering::Greater => return empty_range,
7749 core::cmp::Ordering::Equal if !(lower_inc && upper_inc) => {
7750 return empty_range;
7751 }
7752 _ => {}
7753 }
7754 }
7755 Value::Range {
7756 kind,
7757 lower,
7758 upper,
7759 lower_inc,
7760 upper_inc,
7761 empty: false,
7762 }
7763}
7764
7765/// v7.38 (read01, T6.P3) — fold a NUMERIC input's kind into a running sum's kind:
7766/// NaN wins; ±Inf + finite → that Inf; +Inf + -Inf → NaN; else unchanged.
7767fn fold_sum_kind(
7768 acc: spg_storage::NumericKind,
7769 incoming: spg_storage::NumericKind,
7770) -> spg_storage::NumericKind {
7771 use spg_storage::NumericKind as NK;
7772 match (acc, incoming) {
7773 (NK::NaN, _) | (_, NK::NaN) => NK::NaN,
7774 (NK::Finite, k) | (k, NK::Finite) => k,
7775 (a, b) if a == b => a,
7776 _ => NK::NaN,
7777 }
7778}
7779
7780/// v7.39 (enum order knife) — min/max extreme comparison: member order when
7781/// the spec's argument is enum-typed, the generic value order otherwise.
7782fn extreme_cmp(
7783 enum_labels: Option<&[String]>,
7784 a: &Value,
7785 b: &Value,
7786 mysql: bool,
7787) -> core::cmp::Ordering {
7788 extreme_cmp_in(enum_labels, None, a, b, mysql)
7789}
7790
7791/// v7.39 (round 690) — `extreme_cmp` with the argument column's collation.
7792///
7793/// `min`/`max` over a column declared `COLLATE "en_US.utf8"` answered
7794/// `Banana` and `Ápple` where PG18 gives `apple` and `Zebra`. The collation
7795/// rides beside `enum_labels`, which is already exactly this: per-aggregate
7796/// metadata about the argument, resolved once where the spec is built.
7797///
7798/// No derivation needed here — `min(loc)`'s argument is the column itself.
7799/// An expression argument gets None and keeps byte order, which is the same
7800/// limit `ORDER BY upper(loc)` has.
7801fn extreme_cmp_in(
7802 enum_labels: Option<&[String]>,
7803 collation: Option<&str>,
7804 a: &Value,
7805 b: &Value,
7806 mysql: bool,
7807) -> core::cmp::Ordering {
7808 if let Some(labels) = enum_labels
7809 && let Some(ord) = crate::eval::enum_ord_cmp(labels, a, b)
7810 {
7811 return ord;
7812 }
7813 if let (Value::Text(x), Value::Text(y), Some(c)) = (a, b, collation)
7814 && let Some(ord) = crate::collate::compare(c, x, y)
7815 {
7816 return ord;
7817 }
7818 // v7.39 (round 412) — MIN / MAX over text under the MySQL default
7819 // collation compares by the folded form (case- and accent-insensitive,
7820 // PAD SPACE), matching ORDER BY (round 411).
7821 if mysql {
7822 // v7.38.18 — each side on its own type; see `mysql_fold_value`.
7823 if let (Some(x), Some(y)) = (
7824 spg_storage::mysql_fold_value(a),
7825 spg_storage::mysql_fold_value(b),
7826 ) {
7827 return x.cmp(&y);
7828 }
7829 }
7830 value_cmp(a, b)
7831}
7832
7833/// Compare two values for `min` / `max`.
7834///
7835/// v7.39 (round 674) — the 228 lines that used to live here were a SECOND
7836/// comparison matrix, written independently of `orderby::value_cmp`. A
7837/// census of which `Value` variants each named found them diverged rather
7838/// than duplicated, and two silent wrongs fell out of the gap: `ORDER BY
7839/// time_col` did not sort (round 672) and `min`/`max` over `CHAR(n)`
7840/// returned the first row (round 672). Round 673 found four more on the
7841/// orderby side, where a canonical-text fallback had `ORDER BY money`
7842/// putting $100 before $9.
7843///
7844/// What stays here is the ONLY thing the two legitimately disagreed about:
7845/// where NULL sorts. This one puts NULLs last so `min`/`max` skip them;
7846/// `orderby::value_cmp` puts them first and the ORDER BY layer above it
7847/// applies NULLS FIRST / NULLS LAST. Both were correct in context, which is
7848/// why merging the matrices wholesale would have flipped one of them —
7849/// verified before collapsing, not after, and the eight NULL shapes are
7850/// pinned.
7851fn value_cmp(a: &Value, b: &Value) -> core::cmp::Ordering {
7852 use core::cmp::Ordering;
7853 match (a, b) {
7854 (Value::Null, Value::Null) => Ordering::Equal,
7855 // NULLs last, so a NULL never wins a min() or a max().
7856 (Value::Null, _) => Ordering::Greater,
7857 (_, Value::Null) => Ordering::Less,
7858 _ => crate::orderby::value_cmp(a, b),
7859 }
7860}
7861
7862/// v7.37.9 Phase 0 diagnostic counters — see
7863/// `.claude/notes/v7.37.9-class-a-c-cascade-closure-plan.md`. These
7864/// are read-only telemetry, do not gate any code path. Used by
7865/// `xtests/dogfood_replay/src/bin/counter_dump.rs` to verify
7866/// whether the DISTA A-3 + array_agg-ordered fast paths actually
7867/// fire on the mailrs Class A SQL shape.
7868pub static DISTA_LITERAL_ARG2_CACHE_FIRE: core::sync::atomic::AtomicU64 =
7869 core::sync::atomic::AtomicU64::new(0);
7870pub static AGGREGATE_ARRAY_AGG_ORDER_BY_FIRE: core::sync::atomic::AtomicU64 =
7871 core::sync::atomic::AtomicU64::new(0);
7872
7873/// v7.37.9 Phase 1A-ext — per-row spec dispatch branches in
7874/// `accumulate_groups`'s hot loop. Verifies the Phase 1A
7875/// decomposition agent's S06 assumption ("14 specs × eval_expr per
7876/// row"). Sum should equal `n_specs × n_input_rows`. Branch
7877/// distribution tells which attack target ROI is highest:
7878/// FAST_POS many = baseline OK; COMPILED_MISS many = Step-VM is
7879/// hot path; EVAL_FALLBACK > 0 = uncompilable specs walking the
7880/// eval_expr tree per row × Cow row materialise.
7881pub static AGG_PER_ROW_FAST_POS: core::sync::atomic::AtomicU64 =
7882 core::sync::atomic::AtomicU64::new(0);
7883pub static AGG_PER_ROW_COMPILED_HIT: core::sync::atomic::AtomicU64 =
7884 core::sync::atomic::AtomicU64::new(0);
7885pub static AGG_PER_ROW_COMPILED_MISS: core::sync::atomic::AtomicU64 =
7886 core::sync::atomic::AtomicU64::new(0);
7887pub static AGG_PER_ROW_EVAL_FALLBACK: core::sync::atomic::AtomicU64 =
7888 core::sync::atomic::AtomicU64::new(0);
7889pub static AGG_PER_ROW_COUNT_STAR_SENTINEL: core::sync::atomic::AtomicU64 =
7890 core::sync::atomic::AtomicU64::new(0);
7891
7892#[cfg(test)]
7893mod value_cmp_mixed_numeric_tests {
7894 //! v7.37.16 Slice A — direct coverage of the mixed NUMERIC↔int/float
7895 //! arms in the aggregate-local `value_cmp` (drives min / max / argmin
7896 //! / argmax / mode / ordered-set aggregates). These pairs previously
7897 //! hit `_ => Equal`, which made `min`/`max` over a mixed NUMERIC/int
7898 //! key keep whichever row arrived first. Semantics now mirror
7899 //! binop.rs: int→NUMERIC exact promotion, NUMERIC→f64 demotion vs a
7900 //! float.
7901 use super::value_cmp;
7902 use core::cmp::Ordering;
7903 use spg_storage::Value;
7904
7905 fn num(scaled: i128, scale: u16) -> Value<'static> {
7906 Value::Numeric {
7907 scaled,
7908 scale,
7909 kind: spg_storage::NumericKind::Finite,
7910 }
7911 }
7912
7913 #[test]
7914 fn numeric_vs_integer_and_float() {
7915 assert_eq!(value_cmp(&num(250, 2), &Value::Int(5)), Ordering::Less);
7916 assert_eq!(value_cmp(&Value::Int(5), &num(250, 2)), Ordering::Greater);
7917 // debug-string/Equal fallback bug: 1000 vs 9 must be Greater.
7918 assert_eq!(
7919 value_cmp(&num(1000, 0), &Value::SmallInt(9)),
7920 Ordering::Greater
7921 );
7922 assert_eq!(value_cmp(&num(20, 1), &Value::BigInt(2)), Ordering::Equal);
7923 assert_eq!(value_cmp(&Value::BigInt(2), &num(20, 1)), Ordering::Equal);
7924 // NUMERIC↔float demotion.
7925 assert_eq!(value_cmp(&num(35, 1), &Value::Float(3.5)), Ordering::Equal);
7926 assert_eq!(
7927 value_cmp(&num(35, 1), &Value::Float(3.0)),
7928 Ordering::Greater
7929 );
7930 assert_eq!(value_cmp(&Value::Float(1.0), &num(25, 1)), Ordering::Less);
7931 }
7932
7933 /// v7.39 (round 231) — `is_aggregate_name` admits a name and
7934 /// `classify_agg_name` panics on anything it doesn't know, so the two
7935 /// lists drifting apart turns into a SQL-reachable abort. That is how
7936 /// `every(x) OVER (…)` crashed the query in round 230. Walk the whole
7937 /// admitted set and classify each one.
7938 #[test]
7939 fn every_aggregate_name_classifies() {
7940 const NAMES: &[&str] = &[
7941 "count",
7942 "count_star",
7943 "sum",
7944 "min",
7945 "max",
7946 "avg",
7947 "any_value",
7948 "range_agg",
7949 "range_intersect_agg",
7950 "string_agg",
7951 "group_concat",
7952 "xmlagg",
7953 "array_agg",
7954 "bool_and",
7955 "bool_or",
7956 "every",
7957 "stddev",
7958 "stddev_samp",
7959 "stddev_pop",
7960 "variance",
7961 "var_samp",
7962 "var_pop",
7963 "bit_and",
7964 "bit_or",
7965 "bit_xor",
7966 "json_agg",
7967 "jsonb_agg",
7968 "json_object_agg",
7969 "jsonb_object_agg",
7970 ];
7971 for n in NAMES {
7972 assert!(
7973 super::is_aggregate_name(n),
7974 "{n} should be an aggregate name"
7975 );
7976 // Panics if the classifier doesn't know it.
7977 let _ = super::classify_agg_name(super::canonical_agg_name(n));
7978 }
7979 // Anything `is_aggregate_name` admits must classify, so a name added
7980 // to one list and not the other fails here rather than at runtime.
7981 for n in NAMES {
7982 assert!(
7983 super::is_aggregate_name(&n.to_ascii_uppercase()),
7984 "{n} should be case-insensitive"
7985 );
7986 }
7987 }
7988}