spg_engine/select.rs
1//! SELECT execution — the window / meta-view / CTE variants and the
2//! subquery-resolution pre-pass. Lifted out of `lib.rs` (v7.32 engine
3//! modularisation). These `impl Engine` methods are dispatched from the
4//! bare-SELECT entry points and drive the non-trivial SELECT shapes.
5
6use alloc::borrow::Cow;
7use alloc::string::{String, ToString};
8use alloc::vec::Vec;
9
10use spg_sql::ast::{
11 ColumnName, Expr, FromClause, SelectItem, SelectStatement, Statement, TableRef, UnionKind,
12};
13use spg_storage::{
14 Catalog, ColumnSchema, DataType, Row, StorageError, TableSchema, Value, VecEncoding,
15};
16
17use crate::describe;
18use crate::eval::{EvalContext, EvalError};
19use crate::join::RowRef;
20use crate::system_catalog::collect_view_refs;
21use crate::{
22 ByteBudget, CancelToken, Engine, EngineError, OrderKey, QueryResult, aggregate,
23 apply_offset_and_limit, apply_offset_and_limit_tagged, approx_row_bytes, build_order_keys,
24 collect_meta_view_names, collect_qualified_refs, collect_scalar_subqueries,
25 collect_window_nodes, compute_window_partition, eval, expr_tree_has_subquery,
26 materialise_in_order, materialise_meta_view, memoize, order_by_value_cmp_in, partition_key_cmp,
27 rewrite_window_to_columns, select_has_window, select_references_meta_view, select_refers_to,
28 sort_by_keys, synth_info_key_column_usage, synth_info_referential_constraints,
29 synth_info_routines, synth_info_statistics, synth_information_schema_columns,
30 synth_information_schema_tables, synth_mysql_db, synth_mysql_user, synth_pg_attribute,
31 synth_pg_class, synth_pg_constraint, synth_pg_database, synth_pg_extension, synth_pg_index_raw,
32 synth_pg_indexes, synth_pg_namespace, synth_pg_operator, synth_pg_proc, synth_pg_roles,
33 synth_pg_sequence, synth_pg_settings, synth_pg_timezone_abbrevs, synth_pg_timezone_names,
34 synth_pg_trigger, synth_pg_type, synth_pg_views, topk_trim, try_gin_jsonb_seek, try_gin_seek,
35 try_index_seek, try_nsw_knn, try_pk_walk_top_n, try_trgm_seek, value_is_bigint,
36 value_is_integer, value_to_i64,
37};
38
39/// v7.39 (round 618) — a recursive term that can be run over the working set
40/// directly, instead of through a whole query execution per round.
41///
42/// PG plans the recursive term ONCE and re-scans a worktable each iteration.
43/// SPG emptied and refilled a real table and then called `exec_select_cancel`
44/// — FROM resolution, schema build, predicate compilation, projection build
45/// and result materialisation — for every round. Measured with the counting
46/// allocator on `WITH RECURSIVE r(n) AS (SELECT 1 UNION ALL SELECT n+1 FROM r
47/// WHERE n < N)`: about 40 allocations and 99 kB PER ROUND while the working
48/// set is one row, or 1.98 GB at N = 20000.
49///
50/// This is the shape that covers the ordinary recursive term: read the CTE,
51/// filter it, project it. Anything else — a join, an aggregate, a window, a
52/// subquery, DISTINCT, GROUP BY, ORDER BY, LIMIT, a locking clause, a
53/// non-table source — returns `None` and keeps the general path, so the
54/// answers it gives are the ones that path gave.
55struct RecursiveTermPlan<'t> {
56 items: Vec<&'t Expr>,
57 where_: Option<&'t Expr>,
58 alias: String,
59}
60
61fn plan_recursive_term<'t>(
62 t: &'t SelectStatement,
63 cte_name: &str,
64 ncols: usize,
65) -> Option<RecursiveTermPlan<'t>> {
66 if !t.unions.is_empty()
67 || !t.ctes.is_empty()
68 || t.distinct
69 || !t.distinct_on.is_empty()
70 || t.group_by.is_some()
71 || t.group_by_all
72 || t.having.is_some()
73 || !t.order_by.is_empty()
74 || t.limit.is_some()
75 || t.offset.is_some()
76 || t.limit_with_ties
77 || t.locking.is_some()
78 {
79 return None;
80 }
81 let from = t.from.as_ref()?;
82 if !from.joins.is_empty() {
83 return None;
84 }
85 let p = &from.primary;
86 if !p.name.eq_ignore_ascii_case(cte_name)
87 || p.as_of_segment.is_some()
88 || p.unnest_expr.is_some()
89 || !p.unnest_column_aliases.is_empty()
90 || p.with_ordinality
91 || p.generate_series_args.is_some()
92 || p.lateral_subquery.is_some()
93 || p.jsonb_each_text_arg.is_some()
94 || p.table_fn_call.is_some()
95 {
96 return None;
97 }
98 let unsupported = |e: &Expr| {
99 crate::aggregate::contains_aggregate(e)
100 || crate::subquery::expr_has_subquery(e)
101 || crate::window::expr_has_window_pub(e)
102 };
103 let mut items: Vec<&Expr> = Vec::with_capacity(t.items.len());
104 for it in &t.items {
105 match it {
106 SelectItem::Expr { expr, .. } => {
107 if unsupported(expr) {
108 return None;
109 }
110 items.push(expr);
111 }
112 // `*` would have to be expanded against the CTE's own schema;
113 // the general path already does that, so leave it there.
114 _ => return None,
115 }
116 }
117 if items.len() != ncols {
118 return None;
119 }
120 if let Some(w) = &t.where_
121 && unsupported(w)
122 {
123 return None;
124 }
125 Some(RecursiveTermPlan {
126 items,
127 where_: t.where_.as_ref(),
128 alias: p.alias.clone().unwrap_or_else(|| p.name.clone()),
129 })
130}
131
132impl Engine {
133 /// v4.12 window executor. Implements `ROW_NUMBER` / `RANK` /
134 /// `DENSE_RANK` and the partition-aware aggregates `SUM` /
135 /// `AVG` / `COUNT` / `MIN` / `MAX`. The plan is:
136 /// 1. Apply the WHERE filter.
137 /// 2. For each unique `WindowFunction` node in the projection,
138 /// partition + sort, compute the per-row value.
139 /// 3. Append the window values as synthetic columns (`__win_N`)
140 /// to the row schema.
141 /// 4. Rewrite the projection to read those columns.
142 /// 5. Hand off to the regular project / ORDER BY / LIMIT pipe.
143 #[allow(
144 clippy::too_many_lines,
145 clippy::type_complexity,
146 clippy::needless_range_loop
147 )] // window-eval is one cohesive pipe; splitting fragments
148 pub(crate) fn exec_select_with_window(
149 &self,
150 stmt: &SelectStatement,
151 cancel: CancelToken<'_>,
152 ) -> Result<QueryResult, EngineError> {
153 let from = stmt.from.as_ref().ok_or_else(|| {
154 EngineError::Unsupported("window functions require a FROM clause".into())
155 })?;
156 // v7.17.0 Phase 3.P0-43 — JOIN + window functions. Phase
157 // 3.6 rejected this combination outright ("queued for
158 // v5.x"); P0-43 materialises the join + WHERE through the
159 // existing nested-loop helper and runs the window pipeline
160 // on the joined row set with the combined `alias.col`
161 // schema. The window expressions resolve through the
162 // qualifier-aware column resolver same as the aggregate /
163 // projection paths on JOIN.
164 let (schema_cols_owned, alias_opt): (Vec<ColumnSchema>, Option<&str>);
165 // v7.39 (round 976) — rows this walk OWNS. A derived FROM item and
166 // a JOIN both produce rows that exist nowhere else, so they land
167 // here; a plain stored table does not, and borrows instead.
168 //
169 // It used to clone every row out of the table, on the reasoning
170 // that "the clone is cheap relative to the window computation that
171 // follows". Measured on 400k rows, `row_number() OVER ()` cost
172 // 31.881 ms against 46.520 with a 200-byte column added — so the
173 // clone tracks row width at about 36 ns per row per 200 bytes, and
174 // the window computation it was being compared against is a
175 // counter increment per row. Nothing downstream needs the rows
176 // owned: the very next statement used to be
177 // `filtered.iter().collect()` into the `&Row` slice the window
178 // pipeline actually reads.
179 let mut owned_rows: Vec<Row<'static>> = Vec::new();
180 // What the pipeline reads. Borrows `owned_rows` or the table.
181 let mut filtered: Vec<&Row<'static>> = Vec::new();
182 // Set by the branches that fill `owned_rows`, because "empty" is
183 // an answer a query can legitimately have and so cannot be the
184 // signal for which of the two holds the rows.
185 let mut rows_are_owned = false;
186 if from.joins.is_empty() {
187 let primary = &from.primary;
188 // v7.37 D.13 — window functions over a derived table (subquery /
189 // VALUES / unnest / generate_series). The catalog-by-name lookup
190 // below only finds real tables, so a derived primary threw
191 // TableNotFound. Materialise the derived rows + schema through the
192 // same helper the non-window FROM-primary path uses, then WHERE-
193 // filter and feed the identical window pipeline.
194 let is_derived = primary.lateral_subquery.is_some()
195 || primary.unnest_expr.is_some()
196 || primary.generate_series_args.is_some()
197 || primary.jsonb_each_text_arg.is_some()
198 || primary.table_fn_call.is_some();
199 if is_derived {
200 let (drows, dcols) = self.materialise_table_ref(primary)?;
201 schema_cols_owned = dcols;
202 alias_opt = primary.alias.as_deref();
203 let ctx = self.ev_ctx(&schema_cols_owned, alias_opt);
204 let mut owned: Vec<Row<'static>> = Vec::new();
205 for (i, row) in drows.into_iter().enumerate() {
206 if i.is_multiple_of(256) {
207 cancel.check()?;
208 }
209 if let Some(w) = &stmt.where_ {
210 let cond = eval::eval_expr(w, &row, &ctx)?;
211 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
212 continue;
213 }
214 }
215 owned.push(row);
216 }
217 owned_rows = owned;
218 rows_are_owned = true;
219 } else {
220 let table = self.active_catalog().get(&primary.name).ok_or_else(|| {
221 StorageError::TableNotFound {
222 name: primary.name.clone(),
223 }
224 })?;
225 let alias = primary.alias.as_deref().unwrap_or(primary.name.as_str());
226 schema_cols_owned = table.schema().columns.clone();
227 alias_opt = Some(alias);
228 let ctx = self.ev_ctx(&schema_cols_owned, alias_opt);
229 // The WHERE test, in ONE place, for all four ways a row can
230 // reach this walk. It deliberately does not touch the row
231 // collections: a closure that pushed into them would tie
232 // its argument to the closure body and no borrowed row
233 // could escape it, which is what forced the clone-shaped
234 // version of this loop in the first place.
235 let passes = |row: &Row<'static>| -> Result<bool, EngineError> {
236 if let Some(w) = &stmt.where_ {
237 let cond = eval::eval_expr(w, row, &ctx)?;
238 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
239 return Ok(false);
240 }
241 }
242 Ok(true)
243 };
244 // v7.37.15 Phase B — scan_visible filters rows by the
245 // engine's current snapshot. Phase B's `current_snapshot()`
246 // returns `Snapshot::unbounded()` so every row is visible,
247 // matching pre-v7.37.15 byte-for-byte. Phase C will wire
248 // real per-tx snapshots through this same callsite — no
249 // code change needed here when that lands.
250 let snap = self.current_snapshot();
251 if table.has_cold_rows_fast() {
252 // v7.36 (cold-tier coverage) — a cold segment's rows
253 // are produced on demand and live in a temporary this
254 // walk cannot borrow from, so a table carrying any owns
255 // its rows. Hot iter then cold iter, both through the
256 // same WHERE, as before.
257 let mut owned: Vec<Row<'static>> = Vec::new();
258 for (i, row) in table.scan_visible(&snap) {
259 if i.is_multiple_of(256) {
260 cancel.check()?;
261 }
262 if passes(row)? {
263 owned.push(row.clone());
264 }
265 }
266 let hot_len = table.row_count();
267 for (offset, row) in self.iter_cold_rows_of_table(table).iter().enumerate() {
268 let i = hot_len + offset;
269 if i.is_multiple_of(256) {
270 cancel.check()?;
271 }
272 if passes(row)? {
273 owned.push(row.clone());
274 }
275 }
276 owned_rows = owned;
277 rows_are_owned = true;
278 } else {
279 // v7.39 (round 975) — ask the indices first, the way
280 // the streaming walk has since round 970. This walk had
281 // the same hole and it is reached by any statement
282 // carrying a window function, so a WHERE that names an
283 // indexed column read the whole table: measured on 400k
284 // rows, `row_number() OVER () … WHERE id = 500` — a
285 // ONE-row answer on a primary key — took 13.762 ms
286 // against PG18.4's 0.151, while the same predicate
287 // without the window took 0.091. The cost was
288 // independent of how many rows survived (999 survivors
289 // cost 13.312 ms) and of row width (13.312 narrow vs
290 // 13.327 wide), which is what a full table walk looks
291 // like and what a result-shaped cost does not.
292 //
293 // The seek only NARROWS — `passes` still applies the
294 // whole WHERE — so no answer can change. Positions
295 // arrive visibility-filtered by the same predicate the
296 // scan applies and capped at a quarter of the table,
297 // and `None` walks the table exactly as before.
298 let seek_positions: Option<Vec<usize>> = stmt.where_.as_ref().and_then(|w| {
299 crate::index_access::try_index_seek_positions(
300 w,
301 &schema_cols_owned,
302 table,
303 alias,
304 &snap,
305 )
306 });
307 match seek_positions {
308 Some(mut positions) => {
309 // Table order, which is the order the scan
310 // would have produced.
311 positions.sort_unstable();
312 for (n, pos) in positions.into_iter().enumerate() {
313 if n.is_multiple_of(256) {
314 cancel.check()?;
315 }
316 let Some(row) = table.rows().get(pos) else {
317 continue;
318 };
319 if passes(row)? {
320 filtered.push(row);
321 }
322 }
323 }
324 None => {
325 for (i, row) in table.scan_visible(&snap) {
326 if i.is_multiple_of(256) {
327 cancel.check()?;
328 }
329 if passes(row)? {
330 filtered.push(row);
331 }
332 }
333 }
334 }
335 }
336 }
337 } else {
338 let deferred = self.build_joined_filtered_rows(
339 from,
340 stmt.where_.as_ref(),
341 cancel,
342 None,
343 &mut ByteBudget::new(self.max_query_bytes),
344 )?;
345 // A join's survivors are row-index tuples over its sources, so
346 // there is no single row to borrow — this branch owns them.
347 owned_rows = deferred.materialise();
348 rows_are_owned = true;
349 schema_cols_owned = deferred.combined_schema;
350 alias_opt = None;
351 }
352 if rows_are_owned {
353 filtered = owned_rows.iter().collect();
354 }
355 let schema_cols = &schema_cols_owned;
356 let ctx = self.ev_ctx(schema_cols, alias_opt);
357 let alias = alias_opt.unwrap_or("");
358 let n_rows = filtered.len();
359 // The window pipeline reads `&[&Row<'static>]`, and `filtered`
360 // already is one whichever branch produced it — the separate
361 // `filtered_refs` this used to build was the collect that made
362 // owning the rows look necessary.
363
364 // 2) Collect unique window function nodes from projection.
365 let mut window_nodes: Vec<Expr> = Vec::new();
366 for item in &stmt.items {
367 if let SelectItem::Expr { expr, .. } = item {
368 collect_window_nodes(expr, &mut window_nodes);
369 }
370 }
371 // v7.39 (round 592) — and from ORDER BY, which may name a window the
372 // select list never mentions. The order-key builder below rewrites
373 // window calls to `__win_N` columns, and a call that was never
374 // collected has no column to become.
375 for o in &stmt.order_by {
376 collect_window_nodes(&o.expr, &mut window_nodes);
377 }
378
379 // 3) For each window, compute per-row value.
380 // Index: same order as window_nodes; for row i, win_vals[w][i].
381 let mut win_vals: Vec<Vec<Value<'static>>> = Vec::with_capacity(window_nodes.len());
382 for wnode in &window_nodes {
383 let Expr::WindowFunction {
384 name,
385 args,
386 partition_by,
387 order_by,
388 frame,
389 null_treatment,
390 filter,
391 } = wnode
392 else {
393 unreachable!("collect_window_nodes pushes only WindowFunction");
394 };
395 // Compute (partition_key, order_key, original_index) for each row.
396 // v7.39 (round 593) — a key that is a plain column sits at the same
397 // position in every row, but was resolved BY NAME for each one. A
398 // per-library profile of `lag(id) OVER (ORDER BY id)` put
399 // `resolve_column` at 5.8% of the query on its own, with
400 // `rehydrate_cell` and the `eval_expr` dispatch behind it. Resolve
401 // once; anything that is not a plain column keeps the resolver.
402 let p_bound: Vec<Option<usize>> = partition_by
403 .iter()
404 .map(|e| crate::orderby::bound_column_position(e, schema_cols, alias_opt))
405 .collect();
406 let o_bound: Vec<Option<usize>> = order_by
407 .iter()
408 .map(|(e, _, _)| crate::orderby::bound_column_position(e, schema_cols, alias_opt))
409 .collect();
410 let arg_bound = args
411 .first()
412 .and_then(|a| crate::orderby::bound_column_position(a, schema_cols, alias_opt));
413 // v7.39 (round 690) — a window's ORDER BY over a column that
414 // declares a collation sorts by it, the same as a top-level
415 // ORDER BY. Resolved from the bound position, so only a bare
416 // column gets one; an expression produces a new value and the
417 // derivation that would give IT a collation is unbuilt.
418 let o_colls: Vec<Option<alloc::string::String>> = o_bound
419 .iter()
420 .map(|p| {
421 p.and_then(|pos| schema_cols.get(pos))
422 .and_then(|sc| sc.collation_name.clone())
423 .filter(|n| crate::collate::is_supported(n))
424 })
425 .collect();
426 let mut indexed: Vec<(Vec<Value<'static>>, Vec<(Value, bool, Option<bool>)>, usize)> =
427 Vec::with_capacity(n_rows);
428 // v7.39 (round 731) — single bound INT partition key, no window
429 // ORDER BY: group on the i64 directly. The generic build paid
430 // two heap Vecs per row (pkey + empty okey) plus a canonical
431 // string encode per row just to bucket 500k rows into 100
432 // groups; the whole per-row key apparatus disappears here.
433 // Neither key Vec is read downstream on this path: the hash
434 // grouping replaces partition_key_cmp, and okey is empty by
435 // construction.
436 let int_pkey_fast = order_by.is_empty()
437 && partition_by.len() == 1
438 && p_bound[0].is_some_and(|pos| {
439 matches!(
440 schema_cols.get(pos).map(|c| c.ty),
441 Some(
442 spg_storage::DataType::Int
443 | spg_storage::DataType::BigInt
444 | spg_storage::DataType::SmallInt
445 )
446 )
447 });
448 // v7.39 (round 979) — the same idea for a single bound INT
449 // window ORDER BY: sort on the i64 instead of on a heap vector
450 // per row.
451 //
452 // Measured at 400k rows (round 978, ablation, answer checked
453 // byte-for-byte against the general path on a key column that
454 // is a permutation): `row_number() OVER (ORDER BY k)` went
455 // 157.057-157.868 ms to 31.253-31.679, which is 79.8% and puts
456 // it on top of the `OVER ()` baseline — the sort essentially
457 // disappears. Round 977 had already shown the cost was
458 // key-shaped rather than row-shaped: the sort's share was
459 // 132.0 ms on a three-integer table and 132.5 with a 200-byte
460 // column added, and a per-row COPY does scale with width
461 // (round 976 measured that at +36 ns/row/200 bytes).
462 //
463 // Gated to ROW_NUMBER, which is the one function that reads
464 // neither key vector — it numbers the order it is handed.
465 // `rank` and `dense_rank` compare adjacent entries' order keys
466 // in `compute_window_partition`, so leaving those vectors
467 // empty would silently give every row rank 1. A wider version
468 // would carry the i64 in the entry and teach those two to use
469 // it; this one is the part that can be shown correct by
470 // construction.
471 let int_okey_fast = partition_by.is_empty()
472 && order_by.len() == 1
473 && frame.is_none()
474 && filter.is_none()
475 && matches!(null_treatment, spg_sql::ast::NullTreatment::Respect)
476 && name.eq_ignore_ascii_case("row_number")
477 && o_bound[0].is_some_and(|pos| {
478 matches!(
479 schema_cols.get(pos).map(|c| c.ty),
480 Some(
481 spg_storage::DataType::Int
482 | spg_storage::DataType::BigInt
483 | spg_storage::DataType::SmallInt
484 )
485 )
486 });
487 // Set when a cell in that column turns out not to be an
488 // integer after all. The declared type says it should be, but
489 // "should" is not a thing to sort 400k rows on, so the general
490 // path takes over and this build is discarded.
491 let mut int_okey_bailed = false;
492 if int_okey_fast {
493 let pos = o_bound[0].expect("gated bound");
494 let desc = order_by[0].1;
495 // PG orders NULLs last ascending and first descending
496 // unless the query says otherwise.
497 let nulls_first = order_by[0].2.unwrap_or(desc);
498 let mut keyed: Vec<(bool, i64, usize)> = Vec::with_capacity(n_rows);
499 for (i, row) in filtered.iter().enumerate() {
500 match row.values.get(pos) {
501 Some(Value::Int(n)) => keyed.push((false, i64::from(*n), i)),
502 Some(Value::BigInt(n)) => keyed.push((false, *n, i)),
503 Some(Value::SmallInt(n)) => keyed.push((false, i64::from(*n), i)),
504 Some(Value::Null) | None => keyed.push((true, 0, i)),
505 Some(_) => {
506 int_okey_bailed = true;
507 break;
508 }
509 }
510 }
511 if !int_okey_bailed {
512 // `null_rank` puts NULLs on the side the query asked
513 // for; the row's original index breaks every tie, so
514 // equal keys keep the order the scan produced — what
515 // the stable sort below would have given them.
516 let null_rank = |is_null: bool| -> u8 { u8::from(is_null != nulls_first) };
517 keyed.sort_unstable_by(|a, b| {
518 null_rank(a.0)
519 .cmp(&null_rank(b.0))
520 .then_with(|| {
521 if a.0 {
522 core::cmp::Ordering::Equal
523 } else if desc {
524 b.1.cmp(&a.1)
525 } else {
526 a.1.cmp(&b.1)
527 }
528 })
529 .then_with(|| a.2.cmp(&b.2))
530 });
531 for (_, _, i) in keyed {
532 indexed.push((Vec::new(), Vec::new(), i));
533 }
534 } else {
535 indexed.clear();
536 }
537 }
538 if int_okey_fast && !int_okey_bailed {
539 // Ordered above; nothing else to build.
540 } else if int_pkey_fast {
541 let pos = p_bound[0].expect("gated bound");
542 let mut slot: hashbrown::HashMap<Option<i64>, usize> = hashbrown::HashMap::new();
543 let mut groups: Vec<Vec<usize>> = Vec::new();
544 for (i, row) in filtered.iter().enumerate() {
545 let k: Option<i64> = match row.values.get(pos) {
546 Some(Value::BigInt(n)) => Some(*n),
547 Some(Value::Int(n)) => Some(i64::from(*n)),
548 Some(Value::SmallInt(n)) => Some(i64::from(*n)),
549 _ => None,
550 };
551 match slot.get(&k) {
552 Some(&gi) => groups[gi].push(i),
553 None => {
554 slot.insert(k, groups.len());
555 groups.push(alloc::vec![i]);
556 }
557 }
558 }
559 // The downstream partition-boundary scan compares pkeys
560 // of ADJACENT entries, so the key must ride along — one
561 // single-element Vec per row (half the generic build's
562 // allocations, no string encode).
563 for g in groups {
564 for i in g {
565 let k: Value<'static> = match filtered[i].values.get(pos) {
566 Some(v) => v.clone(),
567 None => Value::Null,
568 };
569 indexed.push((alloc::vec![k], Vec::new(), i));
570 }
571 }
572 } else {
573 for (i, row) in filtered.iter().enumerate() {
574 let pkey: Vec<Value<'static>> = partition_by
575 .iter()
576 .enumerate()
577 .map(
578 |(k, p)| match p_bound[k].and_then(|pos| row.values.get(pos)) {
579 Some(v) => Ok(v.clone()),
580 None => eval::eval_expr(p, row, &ctx),
581 },
582 )
583 .collect::<Result<_, _>>()?;
584 // v7.39 (read01 round 54) — a window's ORDER BY over an enum
585 // column must sort by MEMBER order (enumsortorder), not the
586 // label's text. Enum values are Text at runtime, so the raw
587 // value key sorted alphabetically — `row_number() OVER (ORDER
588 // BY mood)` numbered the rows happy,ok,sad. Substitute the
589 // member ordinal, the same key the top-level ORDER BY uses.
590 // (Closes the enum-order knife's recorded window residual.)
591 let okey: Vec<(Value, bool, Option<bool>)> = order_by
592 .iter()
593 .enumerate()
594 .map(|(k, (e, desc, nf))| -> Result<_, EngineError> {
595 let v = match o_bound[k].and_then(|pos| row.values.get(pos)) {
596 Some(v) => v.clone(),
597 None => eval::eval_expr(e, row, &ctx)?,
598 };
599 let v = match crate::orderby::enum_order_ordinal(e, &v, &ctx) {
600 Some(ord) => Value::Float(ord),
601 None => v,
602 };
603 Ok((v, *desc, *nf))
604 })
605 .collect::<Result<_, _>>()?;
606 indexed.push((pkey, okey, i));
607 }
608 }
609 // Sort by (partition_key, order_key). Partition key uses
610 // a stable encoded form; order key respects ASC/DESC.
611 // v7.39 (round 731) — with NO window ORDER BY the sort's only
612 // job was putting same-partition rows next to each other, and a
613 // 500k-row comparison sort is a spectacular way to hash-group:
614 // the panel's `sum(id) OVER (PARTITION BY g)` spent ~100 ms
615 // here. Group by encoded key instead, preserving row order
616 // inside each group — exactly what the stable sort preserved,
617 // so every function (row_number included) answers the same.
618 if int_okey_fast && !int_okey_bailed {
619 // Already ordered by the i64 key above.
620 } else if int_pkey_fast {
621 // Already grouped above; same-partition rows are adjacent
622 // in original row order.
623 } else if order_by.is_empty() && !partition_by.is_empty() {
624 let mut slot: hashbrown::HashMap<String, usize> = hashbrown::HashMap::new();
625 let mut groups: Vec<
626 Vec<(Vec<Value<'static>>, Vec<(Value, bool, Option<bool>)>, usize)>,
627 > = Vec::new();
628 let mut keybuf = String::new();
629 for entry in indexed.drain(..) {
630 keybuf.clear();
631 for v in &entry.0 {
632 crate::aggregate::push_canonical_key(&mut keybuf, v);
633 }
634 match slot.get(keybuf.as_str()) {
635 Some(&gi) => groups[gi].push(entry),
636 None => {
637 slot.insert(keybuf.clone(), groups.len());
638 groups.push(alloc::vec![entry]);
639 }
640 }
641 }
642 for g in groups {
643 indexed.extend(g);
644 }
645 } else {
646 indexed.sort_by(|a, b| {
647 let p_cmp = partition_key_cmp(&a.0, &b.0);
648 if p_cmp != core::cmp::Ordering::Equal {
649 return p_cmp;
650 }
651 crate::window::order_key_cmp_in(&a.1, &b.1, &o_colls)
652 });
653 }
654 // Per-partition compute.
655 let mut out_vals: Vec<Value<'static>> = alloc::vec![Value::Null; n_rows];
656 let mut p_start = 0;
657 while p_start < indexed.len() {
658 let mut p_end = p_start + 1;
659 while p_end < indexed.len()
660 && partition_key_cmp(&indexed[p_start].0, &indexed[p_end].0)
661 == core::cmp::Ordering::Equal
662 {
663 p_end += 1;
664 }
665 // Compute the function within this partition slice.
666 compute_window_partition(
667 name,
668 args,
669 arg_bound,
670 !order_by.is_empty(),
671 frame.as_ref(),
672 *null_treatment,
673 filter.as_deref(),
674 &indexed[p_start..p_end],
675 &filtered,
676 &ctx,
677 &mut out_vals,
678 )?;
679 p_start = p_end;
680 }
681 win_vals.push(out_vals);
682 }
683
684 // 4) Build extended schema: original columns + synthetic.
685 let mut ext_cols = schema_cols.clone();
686 for i in 0..window_nodes.len() {
687 ext_cols.push(ColumnSchema::new(
688 alloc::format!("__win_{i}"),
689 DataType::Text, // type doesn't matter for projection eval
690 true,
691 ));
692 }
693 // 6) Rewrite the projection: WindowFunction nodes → Column(__win_N).
694 let mut rewritten_items: Vec<SelectItem> = Vec::with_capacity(stmt.items.len());
695 for item in &stmt.items {
696 let new_item = match item {
697 SelectItem::Wildcard => SelectItem::Wildcard,
698 SelectItem::QualifiedWildcard(q) => SelectItem::QualifiedWildcard(q.clone()),
699 SelectItem::Expr { expr, alias } => {
700 let mut e = expr.clone();
701 rewrite_window_to_columns(&mut e, &window_nodes);
702 // The rewrite swaps the window call for a synthetic
703 // `__win_N` column, and the projection then reported
704 // THAT as the column name — `SELECT count(*) OVER ()`
705 // answered `__win_0`, an internal name, where PG18
706 // answers `count`. Pin the name while the call the
707 // column is named for is still in hand.
708 let alias = if alias.is_none() && e != *expr {
709 Some(default_output_name(expr, self.backslash_escapes))
710 } else {
711 alias.clone()
712 };
713 SelectItem::Expr { expr: e, alias }
714 }
715 };
716 rewritten_items.push(new_item);
717 }
718
719 // 7) Project into final rows. JOIN case uses None so the
720 // qualifier check in `resolve_column` falls through to the
721 // composite `alias.col` schema lookup; single-table case
722 // keeps the bare alias so `bare_col` resolution still
723 // works for the projection's per-row column references.
724 // v7.39 (read01 round 54) — build through `ev_ctx`, the canonical
725 // constructor: it threads the catalog (plus render style / tz / GUCs)
726 // that a bare `EvalContext::new` drops. Without the catalog the OUTER
727 // `ORDER BY <enum col>` of a windowed query sorted by TEXT — the
728 // window values were right, the row order silently was not.
729 let ext_ctx = self.ev_ctx(&ext_cols, alias_opt);
730 let projection = build_projection_hiding_tail(
731 &rewritten_items,
732 &ext_cols,
733 alias,
734 self.backslash_escapes,
735 window_nodes.len(),
736 )?;
737 let mut tagged: Vec<(Vec<OrderKey>, Row)> = Vec::with_capacity(n_rows);
738 // v7.39 (round 592) — the extended row (input columns plus the window
739 // values) used to be materialised for EVERY input row and kept until
740 // the projection had run: the input values cloned into a fresh Vec,
741 // then grown once to take the window columns. A counting allocator put
742 // the window path at 4 allocations a row where a plain derived table
743 // takes 1, and named all four — the input row, the clone, the growth,
744 // and the projected row. Only the last has to exist afterwards, so the
745 // extended row is one buffer refilled per row.
746 let mut ext_row: Row<'static> =
747 Row::new(Vec::with_capacity(schema_cols.len() + window_nodes.len()));
748 for i in 0..n_rows {
749 if i.is_multiple_of(256) {
750 cancel.check()?;
751 }
752 ext_row.values.clear();
753 ext_row.values.extend(filtered[i].values.iter().cloned());
754 for w in 0..window_nodes.len() {
755 ext_row.values.push(win_vals[w][i].clone());
756 }
757 let row = &ext_row;
758 let mut values = Vec::with_capacity(projection.len());
759 for p in &projection {
760 values.push(eval::eval_expr(&p.expr, row, &ext_ctx)?);
761 }
762 let order_keys = if stmt.order_by.is_empty() {
763 Vec::new()
764 } else {
765 let mut keys = Vec::with_capacity(stmt.order_by.len());
766 for o in &stmt.order_by {
767 let mut e = o.expr.clone();
768 rewrite_window_to_columns(&mut e, &window_nodes);
769 let key = eval::eval_expr(&e, row, &ext_ctx)?;
770 // v7.39 (read01 round 54) — this path builds its order keys
771 // itself instead of going through `build_order_keys`, so it
772 // skipped the enum-ordinal substitution: the OUTER
773 // `ORDER BY <enum col>` of a windowed query sorted by the
774 // label's TEXT, not by member order. The window values were
775 // right and only the row order was wrong — silently.
776 match crate::orderby::enum_order_ordinal(&e, &key, &ext_ctx) {
777 Some(ord) => keys.push(value_to_order_key(&Value::Float(ord))?),
778 None => keys.push(value_to_order_key(&key)?),
779 }
780 }
781 keys
782 };
783 tagged.push((order_keys, Row::new(values)));
784 }
785 // ORDER BY + LIMIT/OFFSET on the projected rows.
786 if !stmt.order_by.is_empty() {
787 let descs: Vec<bool> = stmt.order_by.iter().map(|o| o.desc).collect();
788 sort_by_keys(&mut tagged, &descs);
789 }
790 let mut out_rows: Vec<Row<'static>> = tagged.into_iter().map(|(_, r)| r).collect();
791 // v7.37 D.41 — `SELECT DISTINCT` over a window projection: the window
792 // pipeline builds one output row per input row, so DISTINCT must dedup the
793 // projected rows (PG evaluates window functions before DISTINCT). Applied
794 // after ORDER BY (duplicate rows share sort keys, so order is preserved)
795 // and before LIMIT.
796 if stmt.distinct {
797 out_rows = dedup_rows(out_rows, self.backslash_escapes);
798 }
799 apply_offset_and_limit(&mut out_rows, stmt.offset_literal(), stmt.limit_literal());
800 let final_cols: Vec<ColumnSchema> = projection
801 .into_iter()
802 .map(|p| {
803 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
804 c.user_enum_type = p.user_enum_type;
805 c.collation_name = p.collation_name;
806 c.mysql_fsp = p.mysql_fsp;
807 c
808 })
809 .collect();
810 Ok(QueryResult::Rows {
811 columns: final_cols,
812 rows: out_rows,
813 })
814 }
815
816 /// v4.11: materialise each CTE into a temp table inside a
817 /// cloned catalog, then run the body SELECT against a fresh
818 /// engine instance that owns the enriched catalog. The clone
819 /// is moderately expensive — only paid by CTE-bearing queries.
820 /// Subqueries inside CTE bodies / the main body resolve as
821 /// usual; `clock_fn` is propagated so `NOW()` lines up.
822 /// v7.16.2 — mailrs round-10 A.3. Materialise the
823 /// `information_schema.*` / `pg_catalog.*` virtual views
824 /// the SELECT references, then re-execute the SELECT
825 /// against an enriched catalog where those views are real
826 /// tables. Same pattern as `exec_with_ctes`. The temp
827 /// engine carries `meta_views_materialised = true` so its
828 /// own meta-dispatch short-circuits — without that we'd
829 /// infinite-recurse since the temp catalog's view name
830 /// still starts with `__spg_info_` and re-triggers the
831 /// check.
832 pub(crate) fn exec_select_with_meta_views(
833 &self,
834 stmt: &SelectStatement,
835 cancel: CancelToken<'_>,
836 ) -> Result<QueryResult, EngineError> {
837 let catalog = self.meta_view_catalog(stmt)?;
838 let mut temp = Engine::restore(catalog);
839 if let Some(c) = self.clock {
840 temp = temp.with_clock(c);
841 }
842 if let Some(f) = self.salt_fn {
843 temp = temp.with_salt_fn(f);
844 }
845 // v7.39 (round 522) — the temp engine holds the materialised
846 // catalog and, until now, nothing of the SESSION. So every
847 // session-scoped answer changed the moment a system view
848 // appeared in the FROM clause: `SELECT current_user` said
849 // `unmei` and `SELECT current_user FROM pg_class` said `admin`;
850 // `current_setting('work_mem')` fell back to the boot default
851 // after a SET; `application_name` read empty. A privilege check
852 // written against a catalog join was reading a different
853 // identity than the same check written without one.
854 //
855 // Carry what a session can be observed through — its parameters
856 // (which is also where the session user lives), the role store
857 // the privilege builtins read, the dialect, and the rendering
858 // settings a timestamp is spelled with.
859 temp.session_params.clone_from(&self.session_params);
860 temp.users.clone_from(&self.users);
861 temp.backslash_escapes = self.backslash_escapes;
862 temp.mysql_strict = self.mysql_strict;
863 temp.render_style = self.render_style;
864 temp.tz_offset_fn = self.tz_offset_fn;
865 temp.tz_localize_fn = self.tz_localize_fn;
866 temp.tz_abbrev_fn = self.tz_abbrev_fn;
867 temp.meta_views_materialised = true;
868 temp.exec_select_cancel(stmt, cancel)
869 }
870
871 /// v7.39 (round 462) — the catalog a meta-view SELECT resolves
872 /// against: this engine's catalog with every `__spg_*` view the
873 /// statement references materialised into it.
874 ///
875 /// Split out of `exec_select_with_meta_views` so Describe can reach
876 /// the same shapes execution reaches. Describe used to look the FROM
877 /// relation up in the plain catalog, where a system view does not
878 /// exist, and reported "no columns" for every one of them — so an
879 /// extended-protocol client reading `pg_stat_user_tables` got rows
880 /// with no column metadata. Sharing the materialisation means a
881 /// view added here is described correctly the day it is added.
882 pub(crate) fn meta_view_catalog(&self, stmt: &SelectStatement) -> Result<Catalog, EngineError> {
883 let mut needed: alloc::collections::BTreeSet<String> = alloc::collections::BTreeSet::new();
884 collect_meta_view_names(stmt, &mut needed);
885 let mut catalog = self.active_catalog().clone();
886 for view in &needed {
887 if catalog.get(view).is_some() {
888 continue;
889 }
890 match view.as_str() {
891 "__spg_info_columns" => {
892 let (schema, rows) = synth_information_schema_columns(
893 self.active_catalog(),
894 self.backslash_escapes,
895 );
896 materialise_meta_view(&mut catalog, view, schema, rows)?;
897 }
898 "__spg_info_tables" => {
899 let (schema, rows) = synth_information_schema_tables(self.active_catalog());
900 materialise_meta_view(&mut catalog, view, schema, rows)?;
901 }
902 "__spg_pg_class" => {
903 let (schema, rows) = synth_pg_class(
904 self.active_catalog(),
905 i64::try_from(self.vacuum_oldest_active()).unwrap_or(i64::MAX),
906 );
907 materialise_meta_view(&mut catalog, view, schema, rows)?;
908 }
909 "__spg_pg_attribute" => {
910 let (schema, rows) = synth_pg_attribute(self.active_catalog());
911 materialise_meta_view(&mut catalog, view, schema, rows)?;
912 }
913 // v7.17.0 Phase 3.P0-50 — pg_catalog.pg_type for
914 // sqlx / SQLAlchemy / Diesel / pgAdmin lookups.
915 "__spg_pg_type" => {
916 let (schema, rows) = synth_pg_type(self.active_catalog());
917 materialise_meta_view(&mut catalog, view, schema, rows)?;
918 }
919 // v7.39 (round 621) — pg_catalog.pg_operator, which did not
920 // exist at all.
921 "__spg_pg_operator" => {
922 let (schema, rows) = synth_pg_operator(self.active_catalog());
923 materialise_meta_view(&mut catalog, view, schema, rows)?;
924 }
925 // v7.17.0 Phase 3.P0-51 — pg_catalog.pg_proc for
926 // function-name introspection (ORM / pgAdmin).
927 "__spg_pg_proc" => {
928 let (schema, rows) = synth_pg_proc(self.active_catalog());
929 materialise_meta_view(&mut catalog, view, schema, rows)?;
930 }
931 // v7.24 (round-16 D) — pg_catalog.pg_trigger. The
932 // round-16 "why doesn't prod fire the trigger"
933 // question was unanswerable because triggers had NO
934 // introspection surface; tgname/tgenabled plus the
935 // pragmatic relname/timing/events/function columns
936 // make "is it registered and enabled" a one-liner.
937 "__spg_pg_trigger" => {
938 let (schema, rows) = synth_pg_trigger(self.active_catalog());
939 materialise_meta_view(&mut catalog, view, schema, rows)?;
940 }
941 // v7.17.0 Phase 3.P0-52 — pg_catalog.pg_namespace
942 // (schema list for admin tools' tree views).
943 "__spg_pg_namespace" => {
944 let (schema, rows) = synth_pg_namespace(self.active_catalog());
945 materialise_meta_view(&mut catalog, view, schema, rows)?;
946 }
947 // v7.39 — pg_tables convenience view (was a pgwire
948 // canned response that ignored projections).
949 "__spg_pg_tables" => {
950 let (schema, rows) =
951 crate::system_catalog::synth_pg_tables(self.active_catalog());
952 materialise_meta_view(&mut catalog, view, schema, rows)?;
953 }
954 // v7.37.24 (24.1) — pg_catalog.pg_enum (label list
955 // for ENUM types; sqlx / ORM enum codecs read this).
956 "__spg_pg_enum" => {
957 let (schema, rows) =
958 crate::system_catalog::synth_pg_enum(self.active_catalog());
959 materialise_meta_view(&mut catalog, view, schema, rows)?;
960 }
961 // v7.37.21 (21.13) — pg_catalog.pg_replication_slots
962 // (shape-stable empty until 21.12 persists slot state).
963 // v7.39 (round 277) — session-scoped prepared statements.
964 "__spg_pg_prepared_statements" => {
965 let (schema, rows) = crate::system_catalog::synth_pg_prepared_statements(
966 &self.prepared_statements,
967 );
968 materialise_meta_view(&mut catalog, view, schema, rows)?;
969 }
970 "__spg_pg_replication_slots" => {
971 let (schema, rows) =
972 crate::system_catalog::synth_pg_replication_slots(self.active_catalog());
973 materialise_meta_view(&mut catalog, view, schema, rows)?;
974 }
975 // v7.37.21 (21.13-b) — pg_catalog.pg_publication
976 // (one row per CREATE PUBLICATION).
977 "__spg_pg_publication" => {
978 let (schema, rows) = crate::system_catalog::synth_pg_publication(self);
979 materialise_meta_view(&mut catalog, view, schema, rows)?;
980 }
981 // v7.37.21 (21.13-c) — pg_catalog.pg_subscription
982 // (one row per CREATE SUBSCRIPTION; subconninfo
983 // redacted so dashboards can't leak credentials).
984 "__spg_pg_subscription" => {
985 let (schema, rows) = crate::system_catalog::synth_pg_subscription(self);
986 materialise_meta_view(&mut catalog, view, schema, rows)?;
987 }
988 // v7.37.22 (22.x-stat-db) — pg_catalog.pg_stat_database
989 // (one row for SPG's single database; counters are
990 // shape-stable 0 until wiring lands).
991 "__spg_pg_stat_database" => {
992 let (schema, rows) = crate::system_catalog::synth_pg_stat_database(
993 self,
994 self.stat_tup_inserted,
995 self.stat_tup_updated,
996 self.stat_tup_deleted,
997 );
998 materialise_meta_view(&mut catalog, view, schema, rows)?;
999 }
1000 // v7.37.22 (22.14) — pg_catalog.pg_stat_user_tables
1001 // (per-table churn counters; live_tup = row count).
1002 "__spg_pg_stat_user_tables" => {
1003 // r192 — DML counters come from the engine-side
1004 // non-transactional map, not the (tx-shadowed)
1005 // catalog tables.
1006 let (schema, rows) = crate::system_catalog::synth_pg_stat_user_tables(
1007 self.active_catalog(),
1008 &self.table_write_stats,
1009 );
1010 materialise_meta_view(&mut catalog, view, schema, rows)?;
1011 }
1012 // v7.37.22 (22.15) — pg_catalog.pg_stat_user_indexes
1013 // (per-index usage counters; flag unused indexes).
1014 "__spg_pg_stat_user_indexes" => {
1015 let (schema, rows) =
1016 crate::system_catalog::synth_pg_stat_user_indexes(self.active_catalog());
1017 materialise_meta_view(&mut catalog, view, schema, rows)?;
1018 }
1019 // v7.37.22 (22.16) — pg_catalog.pg_stat_bgwriter.
1020 "__spg_pg_stat_bgwriter" => {
1021 let (schema, rows) =
1022 crate::system_catalog::synth_pg_stat_bgwriter(self.active_catalog());
1023 materialise_meta_view(&mut catalog, view, schema, rows)?;
1024 }
1025 // v7.38 (read01 P3.14) — pg_catalog.pg_stat_checkpointer /
1026 // pg_stat_wal shell views (shape-stable, counters pending).
1027 "__spg_pg_stat_checkpointer" => {
1028 let (schema, rows) =
1029 crate::system_catalog::synth_pg_stat_checkpointer(self.active_catalog());
1030 materialise_meta_view(&mut catalog, view, schema, rows)?;
1031 }
1032 "__spg_pg_stat_wal" => {
1033 let (schema, rows) =
1034 crate::system_catalog::synth_pg_stat_wal(self.active_catalog());
1035 materialise_meta_view(&mut catalog, view, schema, rows)?;
1036 }
1037 // v7.38 (read01 P3.15) — pg_catalog.pg_stat_slru /
1038 // pg_stat_subscription_stats shell views.
1039 "__spg_pg_stat_slru" => {
1040 let (schema, rows) =
1041 crate::system_catalog::synth_pg_stat_slru(self.active_catalog());
1042 materialise_meta_view(&mut catalog, view, schema, rows)?;
1043 }
1044 "__spg_pg_stat_subscription_stats" => {
1045 let (schema, rows) = crate::system_catalog::synth_pg_stat_subscription_stats(
1046 self.active_catalog(),
1047 );
1048 materialise_meta_view(&mut catalog, view, schema, rows)?;
1049 }
1050 // v7.37.22 (22.17) — pg_catalog.pg_stat_archiver.
1051 "__spg_pg_stat_archiver" => {
1052 let (schema, rows) =
1053 crate::system_catalog::synth_pg_stat_archiver(self.active_catalog());
1054 materialise_meta_view(&mut catalog, view, schema, rows)?;
1055 }
1056 // v7.37.21 (21.13-d) — pg_catalog.pg_stat_replication.
1057 "__spg_pg_stat_replication" => {
1058 let (schema, rows) =
1059 crate::system_catalog::synth_pg_stat_replication(self.active_catalog());
1060 materialise_meta_view(&mut catalog, view, schema, rows)?;
1061 }
1062 // v7.37.24 (24.13) — pg_catalog.pg_am.
1063 "__spg_pg_am" => {
1064 let (schema, rows) = crate::system_catalog::synth_pg_am(self.active_catalog());
1065 materialise_meta_view(&mut catalog, view, schema, rows)?;
1066 }
1067 // v7.37.22 (22.18) — pg_catalog.pg_stat_io (PG 16+).
1068 "__spg_pg_stat_io" => {
1069 let (schema, rows) =
1070 crate::system_catalog::synth_pg_stat_io(self.active_catalog());
1071 materialise_meta_view(&mut catalog, view, schema, rows)?;
1072 }
1073 // v7.37.22 (22.19) — pg_catalog.pg_stat_user_functions.
1074 "__spg_pg_stat_user_functions" => {
1075 let (schema, rows) =
1076 crate::system_catalog::synth_pg_stat_user_functions(self.active_catalog());
1077 materialise_meta_view(&mut catalog, view, schema, rows)?;
1078 }
1079 // v7.39 (round 287) — pg_catalog.pg_largeobject{,_metadata}.
1080 "__spg_pg_largeobject" => {
1081 let (schema, rows) =
1082 crate::system_catalog::synth_pg_largeobject(self.active_catalog());
1083 materialise_meta_view(&mut catalog, view, schema, rows)?;
1084 }
1085 "__spg_pg_largeobject_metadata" => {
1086 let (schema, rows) =
1087 crate::system_catalog::synth_pg_largeobject_metadata(self.active_catalog());
1088 materialise_meta_view(&mut catalog, view, schema, rows)?;
1089 }
1090 // v7.37.23 (23.7-a) — pg_catalog.pg_statistic_ext.
1091 "__spg_pg_statistic_ext" => {
1092 let (schema, rows) =
1093 crate::system_catalog::synth_pg_statistic_ext(self.active_catalog());
1094 materialise_meta_view(&mut catalog, view, schema, rows)?;
1095 }
1096 // v7.37.24 (24.15) — pg_catalog.pg_statistic.
1097 "__spg_pg_statistic" => {
1098 let (schema, rows) =
1099 crate::system_catalog::synth_pg_statistic(self.active_catalog());
1100 materialise_meta_view(&mut catalog, view, schema, rows)?;
1101 }
1102 // v7.37.22 (22.20) — pg_catalog.pg_stat_progress_vacuum.
1103 "__spg_pg_stat_progress_vacuum" => {
1104 let (schema, rows) =
1105 crate::system_catalog::synth_pg_stat_progress_vacuum(self.active_catalog());
1106 materialise_meta_view(&mut catalog, view, schema, rows)?;
1107 }
1108 // v7.37.22 (22.21) — pg_catalog.pg_stat_progress_create_index.
1109 "__spg_pg_stat_progress_create_index" => {
1110 let (schema, rows) = crate::system_catalog::synth_pg_stat_progress_create_index(
1111 self.active_catalog(),
1112 );
1113 materialise_meta_view(&mut catalog, view, schema, rows)?;
1114 }
1115 // v7.37.22 (22.22) — pg_catalog.pg_stat_progress_analyze.
1116 "__spg_pg_stat_progress_analyze" => {
1117 let (schema, rows) = crate::system_catalog::synth_pg_stat_progress_analyze(
1118 self.active_catalog(),
1119 );
1120 materialise_meta_view(&mut catalog, view, schema, rows)?;
1121 }
1122 // v7.37.24 (24.16) — pg_catalog.pg_inherits
1123 // (partition parent → child OID mapping).
1124 "__spg_pg_inherits" => {
1125 let (schema, rows) =
1126 crate::system_catalog::synth_pg_inherits(self.active_catalog());
1127 materialise_meta_view(&mut catalog, view, schema, rows)?;
1128 }
1129 // v7.39 (round 650) — the text-search catalogs, filled
1130 // with what SPG actually has rather than PG's thirty.
1131 "__spg_pg_ts_config_map" => {
1132 let (schema, rows) =
1133 crate::system_catalog::synth_pg_ts_config_map(self.active_catalog());
1134 materialise_meta_view(&mut catalog, view, schema, rows)?;
1135 }
1136 "__spg_pg_ts_config" => {
1137 let (schema, rows) =
1138 crate::system_catalog::synth_pg_ts_config(self.active_catalog());
1139 materialise_meta_view(&mut catalog, view, schema, rows)?;
1140 }
1141 "__spg_pg_ts_dict" => {
1142 let (schema, rows) =
1143 crate::system_catalog::synth_pg_ts_dict(self.active_catalog());
1144 materialise_meta_view(&mut catalog, view, schema, rows)?;
1145 }
1146 "__spg_pg_ts_parser" => {
1147 let (schema, rows) =
1148 crate::system_catalog::synth_pg_ts_parser(self.active_catalog());
1149 materialise_meta_view(&mut catalog, view, schema, rows)?;
1150 }
1151 "__spg_pg_ts_template" => {
1152 let (schema, rows) =
1153 crate::system_catalog::synth_pg_ts_template(self.active_catalog());
1154 materialise_meta_view(&mut catalog, view, schema, rows)?;
1155 }
1156 // v7.37.24 (24.17) — pg_catalog.pg_depend
1157 // (dependency graph; shape-stable empty since
1158 // SPG's drop enforcement is per-kind, not per-object).
1159 "__spg_pg_depend" => {
1160 let (schema, rows) =
1161 crate::system_catalog::synth_pg_depend(self.active_catalog());
1162 materialise_meta_view(&mut catalog, view, schema, rows)?;
1163 }
1164 // 7.38.1 S5.1 — pg_catalog.pg_opclass (pg_dump wall #1).
1165 "__spg_pg_opclass" => {
1166 let (schema, rows) =
1167 crate::system_catalog::synth_pg_opclass(self.active_catalog());
1168 materialise_meta_view(&mut catalog, view, schema, rows)?;
1169 }
1170 "__spg_pg_opfamily" => {
1171 let (schema, rows) =
1172 crate::system_catalog::synth_pg_opfamily(self.active_catalog());
1173 materialise_meta_view(&mut catalog, view, schema, rows)?;
1174 }
1175 "__spg_pg_amop" => {
1176 let (schema, rows) =
1177 crate::system_catalog::synth_pg_amop(self.active_catalog());
1178 materialise_meta_view(&mut catalog, view, schema, rows)?;
1179 }
1180 "__spg_pg_amproc" => {
1181 let (schema, rows) =
1182 crate::system_catalog::synth_pg_amproc(self.active_catalog());
1183 materialise_meta_view(&mut catalog, view, schema, rows)?;
1184 }
1185 // v7.38 (read01) — pg_catalog.pg_attrdef (column defaults;
1186 // ORM reflection + pg_dump read the deparsed default text).
1187 "__spg_pg_attrdef" => {
1188 let (schema, rows) =
1189 crate::system_catalog::synth_pg_attrdef(self.active_catalog());
1190 materialise_meta_view(&mut catalog, view, schema, rows)?;
1191 }
1192 // v7.39 (RLS) — pg_catalog.pg_policy (raw) + pg_policies (view).
1193 "__spg_pg_policy" => {
1194 let (schema, rows) =
1195 crate::system_catalog::synth_pg_policy(self.active_catalog());
1196 materialise_meta_view(&mut catalog, view, schema, rows)?;
1197 }
1198 "__spg_pg_policies" => {
1199 let (schema, rows) =
1200 crate::system_catalog::synth_pg_policies(self.active_catalog());
1201 materialise_meta_view(&mut catalog, view, schema, rows)?;
1202 }
1203 // v7.37.24 (24.14) — pg_catalog.pg_collation.
1204 "__spg_pg_collation" => {
1205 let (schema, rows) =
1206 crate::system_catalog::synth_pg_collation(self.active_catalog());
1207 materialise_meta_view(&mut catalog, view, schema, rows)?;
1208 }
1209 // v7.37.23 (23.6-b) — pg_catalog.pg_tablespace.
1210 "__spg_pg_tablespace" => {
1211 let (schema, rows) =
1212 crate::system_catalog::synth_pg_tablespace(self.active_catalog());
1213 materialise_meta_view(&mut catalog, view, schema, rows)?;
1214 }
1215 // v7.17.0 Phase 3.P0-53 — pg_catalog.pg_indexes view
1216 // for pgAdmin / DataGrip "indexes per table" listings.
1217 "__spg_pg_indexes" => {
1218 let (schema, rows) = synth_pg_indexes(self.active_catalog());
1219 materialise_meta_view(&mut catalog, view, schema, rows)?;
1220 }
1221 // v7.39 (read01 round 50) — pg_catalog.pg_description, backing
1222 // psql's \d+ comment column and pg_dump's COMMENT ON emission.
1223 "__spg_pg_description" => {
1224 let (schema, rows) =
1225 crate::system_catalog::synth_pg_description(self.active_catalog());
1226 materialise_meta_view(&mut catalog, view, schema, rows)?;
1227 }
1228 // v7.17.0 Phase 3.P0-53 — pg_catalog.pg_index (raw)
1229 // for index introspection by ORM compilers.
1230 "__spg_pg_index" => {
1231 let (schema, rows) = synth_pg_index_raw(self.active_catalog());
1232 materialise_meta_view(&mut catalog, view, schema, rows)?;
1233 }
1234 // v7.17.0 Phase 3.P0-54 — pg_catalog.pg_constraint
1235 // for FK / UNIQUE / PK / CHECK introspection.
1236 "__spg_pg_constraint" => {
1237 let (schema, rows) = synth_pg_constraint(self.active_catalog());
1238 materialise_meta_view(&mut catalog, view, schema, rows)?;
1239 }
1240 // v7.37 U11 — pg_catalog.pg_sequence, one row per CREATE
1241 // SEQUENCE (psql \d <seq> + ORM sequence introspection).
1242 "__spg_pg_sequence" => {
1243 let (schema, rows) = synth_pg_sequence(self.active_catalog());
1244 materialise_meta_view(&mut catalog, view, schema, rows)?;
1245 }
1246 // v7.17.0 Phase 3.P0-55 — pg_catalog.pg_database /
1247 // pg_roles / pg_user. SPG is single-database so
1248 // pg_database surfaces just `postgres`; pg_roles
1249 // / pg_user walk the engine's UserStore.
1250 "__spg_pg_database" => {
1251 let (schema, rows) = synth_pg_database(self);
1252 materialise_meta_view(&mut catalog, view, schema, rows)?;
1253 }
1254 "__spg_pg_roles" => {
1255 let (schema, rows) = synth_pg_roles(self);
1256 materialise_meta_view(&mut catalog, view, schema, rows)?;
1257 }
1258 // v7.39 (round 542) — pg_user is a DIFFERENT view over the
1259 // same roles, with PG's own `use*` column names. It used to
1260 // publish pg_roles' columns under this name.
1261 "__spg_pg_user" => {
1262 let (schema, rows) = crate::system_catalog::synth_pg_user(self);
1263 materialise_meta_view(&mut catalog, view, schema, rows)?;
1264 }
1265 // v7.39 (read01 round 58) — role membership.
1266 "__spg_pg_auth_members" => {
1267 let (schema, rows) = crate::system_catalog::synth_pg_auth_members(self);
1268 materialise_meta_view(&mut catalog, view, schema, rows)?;
1269 }
1270 // v7.17.0 Phase 3.P0-56 — pg_catalog.pg_views. PG's
1271 // pg_views surfaces every CREATE VIEW result; SPG
1272 // ships one row per declared view from the catalog.
1273 "__spg_pg_views" => {
1274 let (schema, rows) = synth_pg_views(self.active_catalog());
1275 materialise_meta_view(&mut catalog, view, schema, rows)?;
1276 }
1277 // v7.39 (round 143) — pg_catalog.pg_rules: one row per
1278 // catalogued query-rewrite RULE.
1279 "__spg_pg_rules" => {
1280 let (schema, rows) =
1281 crate::system_catalog::synth_pg_rules(self.active_catalog());
1282 materialise_meta_view(&mut catalog, view, schema, rows)?;
1283 }
1284 // v7.39 (round 312) — pg_catalog.pg_rewrite: the rule
1285 // catalogue `pg_get_ruledef(oid)` resolves against.
1286 "__spg_pg_rewrite" => {
1287 let (schema, rows) =
1288 crate::system_catalog::synth_pg_rewrite(self.active_catalog());
1289 materialise_meta_view(&mut catalog, view, schema, rows)?;
1290 }
1291 // v7.39 (round 542) — pg_catalog.pg_matviews, with rows
1292 // and PG's own column names.
1293 "__spg_pg_matviews" => {
1294 let (schema, rows) =
1295 crate::system_catalog::synth_pg_matviews(self.active_catalog());
1296 materialise_meta_view(&mut catalog, view, schema, rows)?;
1297 }
1298 // pg_catalog.pg_extension — native capability list
1299 // (mailrs embed round-12).
1300 // v7.39 (round 546) — the catalogs SPG has real content
1301 // for, from the facts it already holds.
1302 "__spg_pg_db_role_setting" => {
1303 let (schema, rows) = crate::system_catalog::synth_pg_db_role_setting(self);
1304 materialise_meta_view(&mut catalog, view, schema, rows)?;
1305 }
1306 "__spg_pg_language" => {
1307 let (schema, rows) = crate::system_catalog::synth_pg_language();
1308 materialise_meta_view(&mut catalog, view, schema, rows)?;
1309 }
1310 "__spg_pg_sequences" => {
1311 let (schema, rows) =
1312 crate::system_catalog::synth_pg_sequences(self.active_catalog());
1313 materialise_meta_view(&mut catalog, view, schema, rows)?;
1314 }
1315 "__spg_pg_range" => {
1316 let (schema, rows) = crate::system_catalog::synth_pg_range();
1317 materialise_meta_view(&mut catalog, view, schema, rows)?;
1318 }
1319 "__spg_pg_partitioned_table" => {
1320 let (schema, rows) =
1321 crate::system_catalog::synth_pg_partitioned_table(self.active_catalog());
1322 materialise_meta_view(&mut catalog, view, schema, rows)?;
1323 }
1324 "__spg_pg_authid" => {
1325 let (schema, rows) = crate::system_catalog::synth_pg_authid(self);
1326 materialise_meta_view(&mut catalog, view, schema, rows)?;
1327 }
1328 "__spg_pg_group" => {
1329 let (schema, rows) = crate::system_catalog::synth_pg_group(self);
1330 materialise_meta_view(&mut catalog, view, schema, rows)?;
1331 }
1332 "__spg_pg_shadow" => {
1333 let (schema, rows) = crate::system_catalog::synth_pg_shadow(self);
1334 materialise_meta_view(&mut catalog, view, schema, rows)?;
1335 }
1336 // v7.39 (round 544) — pg_cast, probed from the real
1337 // cast implementation.
1338 "__spg_pg_cast" => {
1339 let (schema, rows) = crate::system_catalog::synth_pg_cast();
1340 materialise_meta_view(&mut catalog, view, schema, rows)?;
1341 }
1342 // v7.39 (round 541) — an empty catalog that exists.
1343 "__spg_pg_foreign_table" => {
1344 let (schema, rows) = crate::system_catalog::synth_pg_foreign_table();
1345 materialise_meta_view(&mut catalog, view, schema, rows)?;
1346 }
1347 "__spg_pg_extension" => {
1348 let (schema, rows) = synth_pg_extension();
1349 materialise_meta_view(&mut catalog, view, schema, rows)?;
1350 }
1351 // v7.39 (round 502) — the timezone catalogues.
1352 "__spg_pg_timezone_names" => {
1353 let (schema, rows) = synth_pg_timezone_names(self);
1354 materialise_meta_view(&mut catalog, view, schema, rows)?;
1355 }
1356 "__spg_pg_timezone_abbrevs" => {
1357 let (schema, rows) = synth_pg_timezone_abbrevs(self);
1358 materialise_meta_view(&mut catalog, view, schema, rows)?;
1359 }
1360 // v7.17.0 Phase 3.P0-57 — pg_catalog.pg_settings.
1361 "__spg_pg_settings" => {
1362 let (schema, rows) = synth_pg_settings(self);
1363 materialise_meta_view(&mut catalog, view, schema, rows)?;
1364 }
1365 // v7.17.0 Phase 3.P0-63 — information_schema.KEY_COLUMN_USAGE.
1366 // v7.39 (read01 round 51) — information_schema.role_table_grants
1367 // and .table_privileges. Both report the owner's seven implicit
1368 // table privileges; SPG's single role owns everything.
1369 // v7.39 (read01 round 59) — information_schema.column_privileges.
1370 "__spg_info_column_privileges" => {
1371 let (schema, rows) =
1372 crate::system_catalog::synth_info_column_privileges(self.active_catalog());
1373 materialise_meta_view(&mut catalog, view, schema, rows)?;
1374 }
1375 "__spg_info_role_table_grants" | "__spg_info_table_privileges" => {
1376 let grantee = self.current_role().to_string();
1377 let (schema, rows) = crate::system_catalog::synth_info_role_table_grants(
1378 self.active_catalog(),
1379 &grantee,
1380 );
1381 materialise_meta_view(&mut catalog, view, schema, rows)?;
1382 }
1383 "__spg_info_key_column_usage" => {
1384 let (schema, rows) = synth_info_key_column_usage(self.active_catalog());
1385 materialise_meta_view(&mut catalog, view, schema, rows)?;
1386 }
1387 // v7.17.0 Phase 3.P0-64 — information_schema.REFERENTIAL_CONSTRAINTS.
1388 "__spg_info_referential_constraints" => {
1389 let (schema, rows) = synth_info_referential_constraints(self.active_catalog());
1390 materialise_meta_view(&mut catalog, view, schema, rows)?;
1391 }
1392 // v7.17.0 Phase 3.P0-64 — information_schema.STATISTICS.
1393 "__spg_info_statistics" => {
1394 let (schema, rows) = synth_info_statistics(self.active_catalog());
1395 materialise_meta_view(&mut catalog, view, schema, rows)?;
1396 }
1397 // v7.17.0 Phase 3.P0-64 — information_schema.ROUTINES.
1398 "__spg_info_routines" => {
1399 let (schema, rows) = synth_info_routines();
1400 materialise_meta_view(&mut catalog, view, schema, rows)?;
1401 }
1402 // v7.37.24 (24.3) — information_schema.attributes.
1403 "__spg_info_attributes" => {
1404 let (schema, rows) = crate::system_catalog::synth_information_schema_attributes(
1405 self.active_catalog(),
1406 );
1407 materialise_meta_view(&mut catalog, view, schema, rows)?;
1408 }
1409 // v7.37.24 (24.2) — information_schema.domains.
1410 "__spg_info_domains" => {
1411 let (schema, rows) = crate::system_catalog::synth_information_schema_domains(
1412 self.active_catalog(),
1413 );
1414 materialise_meta_view(&mut catalog, view, schema, rows)?;
1415 }
1416 // v7.37.24 (24.9) — information_schema.schemata.
1417 "__spg_info_schemata" => {
1418 let (schema, rows) = crate::system_catalog::synth_information_schema_schemata(
1419 self.active_catalog(),
1420 );
1421 materialise_meta_view(&mut catalog, view, schema, rows)?;
1422 }
1423 // v7.37.24 (24.9) — information_schema.views.
1424 "__spg_info_views" => {
1425 let (schema, rows) = crate::system_catalog::synth_information_schema_views(
1426 self.active_catalog(),
1427 );
1428 materialise_meta_view(&mut catalog, view, schema, rows)?;
1429 }
1430 // v7.37.24 (24.9) — information_schema.table_constraints.
1431 "__spg_info_table_constraints" => {
1432 let (schema, rows) =
1433 crate::system_catalog::synth_information_schema_table_constraints(
1434 self.active_catalog(),
1435 );
1436 materialise_meta_view(&mut catalog, view, schema, rows)?;
1437 }
1438 // v7.37.17 — information_schema.constraint_column_usage.
1439 "__spg_info_constraint_column_usage" => {
1440 let (schema, rows) = crate::system_catalog::synth_info_constraint_column_usage(
1441 self.active_catalog(),
1442 );
1443 materialise_meta_view(&mut catalog, view, schema, rows)?;
1444 }
1445 // v7.37.17 — information_schema.triggers.
1446 "__spg_info_triggers" => {
1447 let (schema, rows) =
1448 crate::system_catalog::synth_info_triggers(self.active_catalog());
1449 materialise_meta_view(&mut catalog, view, schema, rows)?;
1450 }
1451 // v7.37.17 — information_schema.check_constraints.
1452 "__spg_info_check_constraints" => {
1453 let (schema, rows) =
1454 crate::system_catalog::synth_info_check_constraints(self.active_catalog());
1455 materialise_meta_view(&mut catalog, view, schema, rows)?;
1456 }
1457 // v7.37.17 — information_schema.sequences.
1458 "__spg_info_sequences" => {
1459 let (schema, rows) =
1460 crate::system_catalog::synth_info_sequences(self.active_catalog());
1461 materialise_meta_view(&mut catalog, view, schema, rows)?;
1462 }
1463 // v7.17.0 Phase 3.P0-65 — mysql.user / mysql.db.
1464 "__spg_mysql_user" => {
1465 let (schema, rows) = synth_mysql_user(self);
1466 materialise_meta_view(&mut catalog, view, schema, rows)?;
1467 }
1468 "__spg_mysql_db" => {
1469 let (schema, rows) = synth_mysql_db();
1470 materialise_meta_view(&mut catalog, view, schema, rows)?;
1471 }
1472 // v7.39 (round 541) — the catalogs PG has that SPG is
1473 // genuinely empty of. Table-driven; see EMPTY_PG_CATALOGS.
1474 other if crate::system_catalog::synth_empty_pg_catalog(other).is_some() => {
1475 let (schema, rows) =
1476 crate::system_catalog::synth_empty_pg_catalog(other).expect("just checked");
1477 materialise_meta_view(&mut catalog, view, schema, rows)?;
1478 }
1479 _ => {
1480 return Err(EngineError::Unsupported(alloc::format!(
1481 "meta view {view:?} is not yet materialisable; \
1482 v7.16.2 covers information_schema.columns / .tables \
1483 and pg_catalog.pg_class / pg_attribute; \
1484 v7.17.0 P0-50..P0-57 add pg_type / pg_proc / pg_namespace / \
1485 pg_indexes / pg_index / pg_constraint / pg_database / pg_roles / \
1486 pg_user / pg_views / pg_matviews / pg_settings"
1487 )));
1488 }
1489 }
1490 }
1491 Ok(catalog)
1492 }
1493
1494 pub(crate) fn exec_with_ctes(
1495 &self,
1496 stmt: &SelectStatement,
1497 cancel: CancelToken<'_>,
1498 ) -> Result<QueryResult, EngineError> {
1499 cancel.check()?;
1500 // v7.37.43-T4.4 — `&self` SELECT path: only read-only CTE
1501 // bodies are supported here. Writable CTEs on a SELECT
1502 // outer require `&mut self` and route through the
1503 // top-level `exec_select_cancel_mut` entry; sentori
1504 // 0065's WITH-INSERT-INSERT shape comes in as a top-level
1505 // INSERT, not a SELECT, so this restriction is harmless
1506 // in practice.
1507 if stmt.ctes.iter().any(|c| c.body.is_modifying()) {
1508 // v7.39 (read01 round 81) — PG's wording. A data-modifying CTE
1509 // (`WITH d AS (DELETE … RETURNING …) …`) is only legal at the top
1510 // of a statement, not nested inside a subquery; this path is
1511 // reached exactly when one is nested. The old text described SPG's
1512 // own executor plumbing ("the top-level mutable entry"), which
1513 // means nothing to a client.
1514 return Err(EngineError::Unsupported(
1515 "WITH clause containing a data-modifying statement must be at the top level".into(),
1516 ));
1517 }
1518 let catalog = self.materialise_ctes_readonly(&stmt.ctes, cancel)?;
1519 // Strip CTEs from the body before running on the temp engine
1520 // so we don't recurse forever.
1521 let mut body = stmt.clone();
1522 body.ctes = Vec::new();
1523 let mut temp = Engine::restore(catalog);
1524 if let Some(c) = self.clock {
1525 temp = temp.with_clock(c);
1526 }
1527 if let Some(f) = self.salt_fn {
1528 temp = temp.with_salt_fn(f);
1529 }
1530 temp.exec_select_cancel(&body, cancel)
1531 }
1532
1533 /// v7.37.43-T4.4 — read-only CTE materialiser used by the
1534 /// `&self` SELECT path. Caller guarantees no modifying CTE
1535 /// bodies are present.
1536 pub(crate) fn materialise_ctes_readonly(
1537 &self,
1538 ctes: &[spg_sql::ast::Cte],
1539 cancel: CancelToken<'_>,
1540 ) -> Result<crate::Catalog, EngineError> {
1541 cancel.check()?;
1542 let mut catalog = self.active_catalog().clone();
1543 for cte in ctes {
1544 let body_select = cte.body.as_select().ok_or_else(|| {
1545 EngineError::Unsupported(alloc::format!(
1546 "data-modifying CTE not supported on this SELECT entry"
1547 ))
1548 })?;
1549 // v7.39 (round 156) — a CTE may SHADOW a same-named real table
1550 // (PG scoping: the WITH name wins for the outer query and later
1551 // CTEs, while THIS body still sees the real table — a
1552 // non-recursive body's self-name is the table, probe P2). This
1553 // materialiser works on a CLONE, so the shadow is simply: run
1554 // the body against the untouched clone, then drop the real
1555 // table from the clone before installing the CTE's temp. A
1556 // RECURSIVE self-reference is the CTE itself (P6), so there the
1557 // drop happens before the iterating materialiser runs.
1558 let (columns, rows) = if cte.recursive && select_refers_to(body_select, &cte.name) {
1559 let synthetic = spg_sql::ast::Cte {
1560 name: cte.name.clone(),
1561 body: spg_sql::ast::CteBody::Select(body_select.clone()),
1562 recursive: true,
1563 column_overrides: cte.column_overrides.clone(),
1564 search: None,
1565 cycle: None,
1566 };
1567 if catalog.get(&cte.name).is_some() {
1568 let _ = catalog.drop_table(&cte.name);
1569 }
1570 self.materialise_recursive_cte(&synthetic, &catalog, cancel)?
1571 } else {
1572 let mut cte_engine = Engine::restore(catalog.clone());
1573 if let Some(c) = self.clock {
1574 cte_engine = cte_engine.with_clock(c);
1575 }
1576 if let Some(f) = self.salt_fn {
1577 cte_engine = cte_engine.with_salt_fn(f);
1578 }
1579 let body_result = cte_engine.exec_select_cancel(body_select, cancel)?;
1580 let QueryResult::Rows { columns, rows } = body_result else {
1581 return Err(EngineError::Unsupported(alloc::format!(
1582 "CTE {:?} body did not return rows",
1583 cte.name
1584 )));
1585 };
1586 (columns, rows)
1587 };
1588 let inferred = infer_column_types(&columns, &rows);
1589 let mut columns = inferred;
1590 if !cte.column_overrides.is_empty() {
1591 if cte.column_overrides.len() != columns.len() {
1592 return Err(EngineError::Unsupported(alloc::format!(
1593 "CTE {:?} column list has {} names but body returns {} columns",
1594 cte.name,
1595 cte.column_overrides.len(),
1596 columns.len()
1597 )));
1598 }
1599 for (col, name) in columns.iter_mut().zip(cte.column_overrides.iter()) {
1600 col.name.clone_from(name);
1601 }
1602 }
1603 let schema = TableSchema::new(cte.name.clone(), columns);
1604 // v7.39 (round 156) — the body ran against the untouched clone;
1605 // from here on the CTE name resolves to the temp (PG scoping).
1606 if catalog.get(&cte.name).is_some() {
1607 let _ = catalog.drop_table(&cte.name);
1608 }
1609 catalog.create_table(schema).map_err(EngineError::Storage)?;
1610 let table = catalog
1611 .get_mut(&cte.name)
1612 .expect("just-created CTE table must exist");
1613 for row in rows {
1614 table.insert(row).map_err(EngineError::Storage)?;
1615 }
1616 }
1617 Ok(catalog)
1618 }
1619
1620 /// v7.37.43-T4.4 — shared CTE materialiser (mutable variant).
1621 /// Retained for non-DML callers; the DML path (writable CTE on
1622 /// INSERT/UPDATE/DELETE outer) uses `run_with_cte_temps` in
1623 /// `dml.rs` which installs the CTE temps directly on the
1624 /// active catalog so the outer statement's writes hit real
1625 /// tables.
1626 #[allow(dead_code)]
1627 pub(crate) fn materialise_ctes(
1628 &mut self,
1629 ctes: &[spg_sql::ast::Cte],
1630 cancel: CancelToken<'_>,
1631 ) -> Result<crate::Catalog, EngineError> {
1632 cancel.check()?;
1633 // v7.37.43-T4.4 — modifying CTEs need to write through the
1634 // SAME catalog as the outer statement, not a clone (PG's
1635 // writable CTE puts all modifications in one transaction).
1636 // For the read-only case the original logic cloned, but
1637 // since the outer statement also goes through the cloned
1638 // engine and ALL writes must converge, we now drive the
1639 // accumulator off `self.active_catalog().clone()` and
1640 // commit the modifying writes directly to `self`'s active
1641 // catalog so the surface is consistent.
1642 let mut catalog = self.active_catalog().clone();
1643 // v7.39 (round 149) — a modifying CTE body's target must be a
1644 // real relation, never a sibling CTE (PG: relation does not
1645 // exist); checked before any alias lands in the accumulator.
1646 for cte in ctes {
1647 let body_target = match &cte.body {
1648 spg_sql::ast::CteBody::Select(_) => None,
1649 spg_sql::ast::CteBody::Insert(i) => Some(i.table.as_str()),
1650 spg_sql::ast::CteBody::Update(u) => Some(u.table.as_str()),
1651 spg_sql::ast::CteBody::Delete(d) => Some(d.table.as_str()),
1652 spg_sql::ast::CteBody::Merge(m) => Some(m.target.as_str()),
1653 };
1654 if let Some(t) = body_target
1655 && ctes.iter().any(|c| c.name.eq_ignore_ascii_case(t))
1656 && catalog.get(t).is_none()
1657 {
1658 return Err(EngineError::Storage(
1659 spg_storage::StorageError::TableNotFound { name: t.into() },
1660 ));
1661 }
1662 }
1663 for cte in ctes {
1664 if catalog.get(&cte.name).is_some() {
1665 return Err(EngineError::Unsupported(alloc::format!(
1666 "CTE name {:?} shadows an existing table; rename the CTE",
1667 cte.name
1668 )));
1669 }
1670 let (columns, rows) = match &cte.body {
1671 // v7.39 (round 145) — see the sibling site: only a body that
1672 // truly self-references takes the iterating materialiser.
1673 spg_sql::ast::CteBody::Select(body)
1674 if cte.recursive && select_refers_to(body, &cte.name) =>
1675 {
1676 // Recursive CTE — the existing helper takes a
1677 // SELECT body and the snapshot catalog.
1678 let synthetic = spg_sql::ast::Cte {
1679 name: cte.name.clone(),
1680 body: spg_sql::ast::CteBody::Select(body.clone()),
1681 recursive: true,
1682 column_overrides: cte.column_overrides.clone(),
1683 search: None,
1684 cycle: None,
1685 };
1686 self.materialise_recursive_cte(&synthetic, &catalog, cancel)?
1687 }
1688 spg_sql::ast::CteBody::Select(body) => {
1689 // v7.25 (round-17) — run against the accumulated
1690 // catalog so later CTEs can reference earlier
1691 // ones in the same WITH clause.
1692 let mut cte_engine = Engine::restore(catalog.clone());
1693 if let Some(c) = self.clock {
1694 cte_engine = cte_engine.with_clock(c);
1695 }
1696 if let Some(f) = self.salt_fn {
1697 cte_engine = cte_engine.with_salt_fn(f);
1698 }
1699 let body_result = cte_engine.exec_select_cancel(body, cancel)?;
1700 let QueryResult::Rows { columns, rows } = body_result else {
1701 return Err(EngineError::Unsupported(alloc::format!(
1702 "CTE {:?} body did not return rows",
1703 cte.name
1704 )));
1705 };
1706 (columns, rows)
1707 }
1708 spg_sql::ast::CteBody::Insert(body) => {
1709 self.exec_modifying_cte_insert(&cte.name, body, cancel)?
1710 }
1711 spg_sql::ast::CteBody::Update(body) => {
1712 self.exec_modifying_cte_update(&cte.name, body, cancel)?
1713 }
1714 spg_sql::ast::CteBody::Delete(body) => {
1715 self.exec_modifying_cte_delete(&cte.name, body, cancel)?
1716 }
1717 spg_sql::ast::CteBody::Merge(body) => {
1718 self.exec_modifying_cte_merge(&cte.name, body, cancel)?
1719 }
1720 };
1721 // v4.22: the projection builder labels any non-column
1722 // expression as Text — including literal SELECT 1.
1723 // Promote each column's type to whatever the rows
1724 // actually carry so the CTE storage table accepts them.
1725 let inferred = infer_column_types(&columns, &rows);
1726 let mut columns = inferred;
1727 if !cte.column_overrides.is_empty() {
1728 if cte.column_overrides.len() != columns.len() {
1729 return Err(EngineError::Unsupported(alloc::format!(
1730 "CTE {:?} column list has {} names but body returns {} columns",
1731 cte.name,
1732 cte.column_overrides.len(),
1733 columns.len()
1734 )));
1735 }
1736 for (col, name) in columns.iter_mut().zip(cte.column_overrides.iter()) {
1737 col.name.clone_from(name);
1738 }
1739 }
1740 let schema = TableSchema::new(cte.name.clone(), columns);
1741 catalog.create_table(schema).map_err(EngineError::Storage)?;
1742 let table = catalog
1743 .get_mut(&cte.name)
1744 .expect("just-created CTE table must exist");
1745 for row in rows {
1746 table.insert(row).map_err(EngineError::Storage)?;
1747 }
1748 }
1749 Ok(catalog)
1750 }
1751
1752 /// v7.37.43-T4.4 — execute an INSERT CTE body. Runs the INSERT
1753 /// against `self` (so the mutation lands in the active catalog
1754 /// inside the current transaction) and captures the RETURNING
1755 /// projection — column schema + rows — to materialise as the
1756 /// CTE alias's table. An INSERT without RETURNING produces a
1757 /// 0-row table with a synthetic single-column placeholder
1758 /// (matches PG: the CTE alias is still defined, but referencing
1759 /// it from the outer query without RETURNING raises a
1760 /// column-resolution error at scan time).
1761 fn exec_modifying_cte_insert(
1762 &mut self,
1763 cte_name: &str,
1764 body: &spg_sql::ast::InsertStatement,
1765 _cancel: CancelToken<'_>,
1766 ) -> Result<
1767 (
1768 Vec<spg_storage::ColumnSchema>,
1769 Vec<spg_storage::Row<'static>>,
1770 ),
1771 EngineError,
1772 > {
1773 // round 151 — a WITH-headed body keeps its own ctes; the body
1774 // statement routes through its writable-CTE entry (outer CTEs
1775 // are never copied into bodies, so no recursion risk).
1776 let body = body.clone();
1777 let result = self.exec_insert(body)?;
1778 match result {
1779 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1780 QueryResult::CommandOk { .. } => {
1781 // No RETURNING — emit a sentinel single-column
1782 // schema with zero rows so the alias is defined.
1783 let placeholder = spg_storage::ColumnSchema::new(
1784 alloc::format!("{cte_name}_returning_absent"),
1785 spg_storage::DataType::Text,
1786 true,
1787 );
1788 Ok((alloc::vec![placeholder], Vec::new()))
1789 }
1790 }
1791 }
1792
1793 /// v7.37.43-T4.4 — execute an UPDATE CTE body, same semantics
1794 /// as INSERT above.
1795 fn exec_modifying_cte_update(
1796 &mut self,
1797 cte_name: &str,
1798 body: &spg_sql::ast::UpdateStatement,
1799 cancel: CancelToken<'_>,
1800 ) -> Result<
1801 (
1802 Vec<spg_storage::ColumnSchema>,
1803 Vec<spg_storage::Row<'static>>,
1804 ),
1805 EngineError,
1806 > {
1807 let body = body.clone();
1808 let result = self.exec_update_cancel(&body, cancel)?;
1809 match result {
1810 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1811 QueryResult::CommandOk { .. } => {
1812 let placeholder = spg_storage::ColumnSchema::new(
1813 alloc::format!("{cte_name}_returning_absent"),
1814 spg_storage::DataType::Text,
1815 true,
1816 );
1817 Ok((alloc::vec![placeholder], Vec::new()))
1818 }
1819 }
1820 }
1821
1822 /// v7.37.43-T4.4 — execute a DELETE CTE body.
1823 fn exec_modifying_cte_delete(
1824 &mut self,
1825 cte_name: &str,
1826 body: &spg_sql::ast::DeleteStatement,
1827 cancel: CancelToken<'_>,
1828 ) -> Result<
1829 (
1830 Vec<spg_storage::ColumnSchema>,
1831 Vec<spg_storage::Row<'static>>,
1832 ),
1833 EngineError,
1834 > {
1835 let body = body.clone();
1836 let result = self.exec_delete_cancel(&body, cancel)?;
1837 match result {
1838 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1839 QueryResult::CommandOk { .. } => {
1840 let placeholder = spg_storage::ColumnSchema::new(
1841 alloc::format!("{cte_name}_returning_absent"),
1842 spg_storage::DataType::Text,
1843 true,
1844 );
1845 Ok((alloc::vec![placeholder], Vec::new()))
1846 }
1847 }
1848 }
1849
1850 /// v7.39 (round 149) — execute a MERGE CTE body (PG 17).
1851 fn exec_modifying_cte_merge(
1852 &mut self,
1853 cte_name: &str,
1854 body: &spg_sql::ast::MergeStatement,
1855 cancel: CancelToken<'_>,
1856 ) -> Result<
1857 (
1858 Vec<spg_storage::ColumnSchema>,
1859 Vec<spg_storage::Row<'static>>,
1860 ),
1861 EngineError,
1862 > {
1863 let body = body.clone();
1864 let result = self.exec_merge_cancel(&body, cancel)?;
1865 match result {
1866 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1867 QueryResult::CommandOk { .. } => {
1868 let placeholder = spg_storage::ColumnSchema::new(
1869 alloc::format!("{cte_name}_returning_absent"),
1870 spg_storage::DataType::Text,
1871 true,
1872 );
1873 Ok((alloc::vec![placeholder], Vec::new()))
1874 }
1875 }
1876 }
1877
1878 /// v4.22: materialise a WITH RECURSIVE CTE. The body must be a
1879 /// UNION (or UNION ALL) of an anchor that does not reference
1880 /// the CTE name, and one or more recursive terms that do. The
1881 /// anchor runs first; each subsequent iteration runs the
1882 /// recursive term against a temp catalog where the CTE name is
1883 /// bound to the *previous* iteration's output. Iteration stops
1884 /// when the recursive term yields no rows; UNION (DISTINCT)
1885 /// deduplicates against the accumulated result, UNION ALL does
1886 /// not. A hard cap on total rows prevents runaway queries.
1887 #[allow(clippy::too_many_lines)]
1888 pub(crate) fn materialise_recursive_cte(
1889 &self,
1890 cte: &spg_sql::ast::Cte,
1891 base_catalog: &Catalog,
1892 cancel: CancelToken<'_>,
1893 ) -> Result<(Vec<ColumnSchema>, Vec<Row<'static>>), EngineError> {
1894 const MAX_TOTAL_ROWS: usize = 1_000_000;
1895 const MAX_ITERATIONS: usize = 100_000;
1896 cancel.check()?;
1897 // v7.37.43-T4.4 — RECURSIVE only supports SELECT bodies;
1898 // a modifying recursive CTE is parser-rejectable but we
1899 // guard here defensively.
1900 let body_select = cte.body.as_select().ok_or_else(|| {
1901 EngineError::Unsupported(alloc::format!(
1902 "WITH RECURSIVE {:?} body must be a SELECT, not a data-modifying statement",
1903 cte.name
1904 ))
1905 })?;
1906 if body_select.unions.is_empty() {
1907 return Err(EngineError::Unsupported(alloc::format!(
1908 "WITH RECURSIVE {:?} body must be a UNION of an anchor and a recursive term",
1909 cte.name
1910 )));
1911 }
1912 // Anchor: the body's leading SELECT, with unions stripped.
1913 let mut anchor = body_select.clone();
1914 let all_union_terms = core::mem::take(&mut anchor.unions);
1915 anchor.ctes = Vec::new();
1916 // v7.37 D.42 — split the UNION members: those that do NOT reference the
1917 // CTE are additional ANCHOR terms, only the ones that do recurse. A
1918 // multi-row VALUES seed lowers to `SELECT r1 UNION ALL SELECT r2 UNION
1919 // ALL <recursive>`, so the leading SELECT alone is not the whole anchor —
1920 // treating the non-recursive `SELECT r2` as a recursive term made it
1921 // re-emit its constant row every iteration → runaway loop.
1922 let (anchor_terms, union_terms): (Vec<_>, Vec<_>) = all_union_terms
1923 .into_iter()
1924 .partition(|(_, t)| !select_refers_to(t, &cte.name));
1925 let anchor_result = self.exec_select_cancel(&anchor, cancel)?;
1926 let QueryResult::Rows {
1927 columns: anchor_cols,
1928 rows: mut anchor_rows,
1929 } = anchor_result
1930 else {
1931 return Err(EngineError::Unsupported(alloc::format!(
1932 "WITH RECURSIVE {:?}: anchor did not return rows",
1933 cte.name
1934 )));
1935 };
1936 // Append every non-recursive UNION member's rows to the anchor set.
1937 for (_, term) in &anchor_terms {
1938 let mut term = term.clone();
1939 term.ctes = Vec::new();
1940 if let QueryResult::Rows { rows, .. } = self.exec_select_cancel(&term, cancel)? {
1941 anchor_rows.extend(rows);
1942 }
1943 }
1944 // The projection builder labels non-column expressions Text;
1945 // refine column types from the anchor's actual values so the
1946 // intermediate iter-catalog tables accept them.
1947 let mut columns = infer_column_types(&anchor_cols, &anchor_rows);
1948 if !cte.column_overrides.is_empty() {
1949 if cte.column_overrides.len() != columns.len() {
1950 return Err(EngineError::Unsupported(alloc::format!(
1951 "CTE {:?} column list has {} names but anchor returns {} columns",
1952 cte.name,
1953 cte.column_overrides.len(),
1954 columns.len()
1955 )));
1956 }
1957 for (col, name) in columns.iter_mut().zip(cte.column_overrides.iter()) {
1958 col.name.clone_from(name);
1959 }
1960 }
1961 let mut all_rows: Vec<Row<'static>> = anchor_rows.clone();
1962 let mut working_set: Vec<Row<'static>> = anchor_rows;
1963 let mut seen: alloc::collections::BTreeSet<Vec<u8>> = alloc::collections::BTreeSet::new();
1964 // Track at least one "all UNION ALL" flag — if every union
1965 // kind is ALL we skip the dedup step (faster + matches PG).
1966 let all_union_all = union_terms.iter().all(|(k, _)| matches!(k, UnionKind::All));
1967 if !all_union_all {
1968 for r in &all_rows {
1969 seen.insert(encode_row_key(r));
1970 }
1971 }
1972 // v7.39 (round 598) — the engine and its catalog are built ONCE.
1973 // Each iteration used to clone the catalog, create the CTE table,
1974 // and construct a whole `Engine` — which initialises 82 fields — to
1975 // hold that round's working set. A counting allocator put the loop
1976 // at 63 allocations and 104 kB per iteration, or 1 GB for a
1977 // 10,000-row recursive CTE, and none of it varied with how much
1978 // else was in the catalog: the per-round rebuild WAS the cost. The
1979 // table is emptied and refilled instead.
1980 let mut iter_catalog = base_catalog.clone();
1981 let schema = TableSchema::new(cte.name.clone(), columns.clone());
1982 iter_catalog
1983 .create_table(schema)
1984 .map_err(EngineError::Storage)?;
1985 let mut iter_engine = Engine::restore(iter_catalog);
1986 if let Some(c) = self.clock {
1987 iter_engine = iter_engine.with_clock(c);
1988 }
1989 if let Some(f) = self.salt_fn {
1990 iter_engine = iter_engine.with_salt_fn(f);
1991 }
1992 // The recursive terms are cloned once too — the clone stripped the
1993 // CTE list off each of them, per term per iteration.
1994 let recursive_terms: Vec<SelectStatement> = union_terms
1995 .iter()
1996 .map(|(_, t)| {
1997 let mut t = t.clone();
1998 t.ctes = Vec::new();
1999 t
2000 })
2001 .collect();
2002 // v7.39 (round 618) — plan every recursive term once. Taken only if
2003 // ALL of them plan, so a query never runs half on each path.
2004 let term_plans: Option<Vec<RecursiveTermPlan<'_>>> = recursive_terms
2005 .iter()
2006 .map(|t| plan_recursive_term(t, &cte.name, columns.len()))
2007 .collect();
2008 let fast_ctx = term_plans.as_ref().map(|plans| {
2009 let alias = plans[0].alias.clone();
2010 (alias, ())
2011 });
2012 for iter in 0..MAX_ITERATIONS {
2013 cancel.check()?;
2014 if working_set.is_empty() {
2015 break;
2016 }
2017 if let (Some(plans), Some((_, ()))) = (term_plans.as_ref(), fast_ctx.as_ref()) {
2018 // The worktable IS the working set: no table to empty and
2019 // refill, and no query execution per round.
2020 let mut next_set: Vec<Row<'static>> = Vec::new();
2021 for plan in plans {
2022 let ctx = self.ev_ctx(&columns, Some(&plan.alias));
2023 for row in &working_set {
2024 cancel.check()?;
2025 if let Some(w) = plan.where_ {
2026 let v = eval::eval_expr(w, row, &ctx).map_err(EngineError::Eval)?;
2027 if !matches!(v, Value::Bool(true)) {
2028 continue;
2029 }
2030 }
2031 let mut vals: Vec<Value<'static>> = Vec::with_capacity(plan.items.len());
2032 for it in &plan.items {
2033 vals.push(eval::eval_expr(it, row, &ctx).map_err(EngineError::Eval)?);
2034 }
2035 let out = Row::new(vals);
2036 if !all_union_all {
2037 let key = encode_row_key(&out);
2038 if !seen.insert(key) {
2039 continue;
2040 }
2041 }
2042 next_set.push(out);
2043 }
2044 }
2045 if next_set.is_empty() {
2046 break;
2047 }
2048 all_rows.extend(next_set.iter().cloned());
2049 working_set = next_set;
2050 if all_rows.len() > MAX_TOTAL_ROWS {
2051 return Err(EngineError::Unsupported(alloc::format!(
2052 "WITH RECURSIVE {:?}: produced more than {MAX_TOTAL_ROWS} rows — likely runaway recursion",
2053 cte.name
2054 )));
2055 }
2056 if iter + 1 == MAX_ITERATIONS {
2057 return Err(EngineError::Unsupported(alloc::format!(
2058 "WITH RECURSIVE {:?}: exceeded {MAX_ITERATIONS} iterations",
2059 cte.name
2060 )));
2061 }
2062 continue;
2063 }
2064 {
2065 // Truncated rather than dropped and recreated: the table's
2066 // own structure is what dropping it throws away, and it is
2067 // identical every round.
2068 let cat = iter_engine.base_catalog_mut();
2069 let table = cat.get_mut(&cte.name).expect("created above");
2070 table.truncate();
2071 for row in &working_set {
2072 table.insert(row.clone()).map_err(EngineError::Storage)?;
2073 }
2074 }
2075 // Run each recursive term in sequence and collect new rows.
2076 let mut next_set: Vec<Row<'static>> = Vec::new();
2077 for term in &recursive_terms {
2078 let r = iter_engine.exec_select_cancel(term, cancel)?;
2079 let QueryResult::Rows {
2080 columns: rc,
2081 rows: rs,
2082 } = r
2083 else {
2084 return Err(EngineError::Unsupported(alloc::format!(
2085 "WITH RECURSIVE {:?}: recursive term did not return rows",
2086 cte.name
2087 )));
2088 };
2089 if rc.len() != columns.len() {
2090 return Err(EngineError::Unsupported(alloc::format!(
2091 "WITH RECURSIVE {:?}: column count of recursive term ({}) does not match anchor ({})",
2092 cte.name,
2093 rc.len(),
2094 columns.len()
2095 )));
2096 }
2097 for row in rs {
2098 if !all_union_all {
2099 let key = encode_row_key(&row);
2100 if !seen.insert(key) {
2101 continue;
2102 }
2103 }
2104 next_set.push(row);
2105 }
2106 }
2107 if next_set.is_empty() {
2108 break;
2109 }
2110 all_rows.extend(next_set.iter().cloned());
2111 working_set = next_set;
2112 if all_rows.len() > MAX_TOTAL_ROWS {
2113 return Err(EngineError::Unsupported(alloc::format!(
2114 "WITH RECURSIVE {:?}: produced more than {MAX_TOTAL_ROWS} rows — likely runaway recursion",
2115 cte.name
2116 )));
2117 }
2118 if iter + 1 == MAX_ITERATIONS {
2119 return Err(EngineError::Unsupported(alloc::format!(
2120 "WITH RECURSIVE {:?}: exceeded {MAX_ITERATIONS} iterations",
2121 cte.name
2122 )));
2123 }
2124 }
2125 Ok((columns, all_rows))
2126 }
2127
2128 pub(crate) fn resolve_select_subqueries(
2129 &self,
2130 stmt: &mut SelectStatement,
2131 cancel: CancelToken<'_>,
2132 ) -> Result<(), EngineError> {
2133 for item in &mut stmt.items {
2134 if let SelectItem::Expr { expr, alias } = item {
2135 // An UNCORRELATED subquery is replaced by its value right
2136 // here, and the shape the column was named for goes with
2137 // it: by projection time `SELECT EXISTS(SELECT 1)` is a
2138 // boolean literal, so SPG answered `?column?` where PG18
2139 // answers `exists`. Only a subquery at the TOP of the item
2140 // loses its name this way — one nested inside a call still
2141 // reports the call.
2142 if alias.is_none()
2143 && matches!(
2144 expr,
2145 Expr::ScalarSubquery(_)
2146 | Expr::Exists { .. }
2147 | Expr::InSubquery { .. }
2148 | Expr::RowInSubquery { .. }
2149 | Expr::RowCmpSubquery { .. }
2150 )
2151 {
2152 *alias = Some(default_output_name(expr, self.backslash_escapes));
2153 }
2154 self.resolve_expr_subqueries(expr, cancel)?;
2155 }
2156 }
2157 if let Some(w) = &mut stmt.where_ {
2158 self.resolve_expr_subqueries(w, cancel)?;
2159 }
2160 // v7.24.1 — JOIN ON conditions can carry subqueries too;
2161 // they were never walked, so even an UNCORRELATED subquery
2162 // in ON hit "subquery reached row eval".
2163 if let Some(from) = &mut stmt.from {
2164 for j in &mut from.joins {
2165 if let Some(on) = &mut j.on {
2166 self.resolve_expr_subqueries(on, cancel)?;
2167 }
2168 }
2169 }
2170 if let Some(gs) = &mut stmt.group_by {
2171 for g in gs {
2172 self.resolve_expr_subqueries(g, cancel)?;
2173 }
2174 }
2175 if let Some(h) = &mut stmt.having {
2176 self.resolve_expr_subqueries(h, cancel)?;
2177 }
2178 for o in &mut stmt.order_by {
2179 self.resolve_expr_subqueries(&mut o.expr, cancel)?;
2180 }
2181 for (_, peer) in &mut stmt.unions {
2182 self.resolve_select_subqueries(peer, cancel)?;
2183 }
2184 Ok(())
2185 }
2186
2187 #[allow(clippy::only_used_in_recursion)] // engine handle reads aren't really pure
2188 pub(crate) fn resolve_expr_subqueries(
2189 &self,
2190 e: &mut Expr,
2191 cancel: CancelToken<'_>,
2192 ) -> Result<(), EngineError> {
2193 // Replace-on-this-node cases first.
2194 if let Some(replacement) = self.subquery_replacement(e, cancel)? {
2195 *e = replacement;
2196 return Ok(());
2197 }
2198 match e {
2199 Expr::NamedArg { expr, .. } => self.resolve_expr_subqueries(expr, cancel)?,
2200 Expr::Variadic(expr) => self.resolve_expr_subqueries(expr, cancel)?,
2201 Expr::AggregateOrdered { call, order_by, .. } => {
2202 self.resolve_expr_subqueries(call, cancel)?;
2203 for o in order_by.iter_mut() {
2204 self.resolve_expr_subqueries(&mut o.expr, cancel)?;
2205 }
2206 }
2207 Expr::Binary { lhs, rhs, .. } => {
2208 self.resolve_expr_subqueries(lhs, cancel)?;
2209 self.resolve_expr_subqueries(rhs, cancel)?;
2210 }
2211 Expr::Unary { expr, .. }
2212 | Expr::Cast { expr, .. }
2213 | Expr::IsNull { expr, .. }
2214 | Expr::BoolTest { expr, .. }
2215 | Expr::FieldAccess { base: expr, .. } => {
2216 self.resolve_expr_subqueries(expr, cancel)?;
2217 }
2218 Expr::FunctionCall { args, .. } => {
2219 for a in args {
2220 self.resolve_expr_subqueries(a, cancel)?;
2221 }
2222 }
2223 Expr::Like { expr, pattern, .. } => {
2224 self.resolve_expr_subqueries(expr, cancel)?;
2225 self.resolve_expr_subqueries(pattern, cancel)?;
2226 }
2227 Expr::Extract { source, .. } => self.resolve_expr_subqueries(source, cancel)?,
2228 // v4.12 window functions — recurse into args + ORDER BY
2229 // + PARTITION BY in case they carry inner subqueries.
2230 Expr::WindowFunction {
2231 args,
2232 partition_by,
2233 order_by,
2234 ..
2235 } => {
2236 for a in args {
2237 self.resolve_expr_subqueries(a, cancel)?;
2238 }
2239 for p in partition_by {
2240 self.resolve_expr_subqueries(p, cancel)?;
2241 }
2242 for (e, _, _) in order_by {
2243 self.resolve_expr_subqueries(e, cancel)?;
2244 }
2245 }
2246 // Subquery nodes are handled in subquery_replacement
2247 // (which returned None — defensive no-op); Literal /
2248 // Column are leaves.
2249 Expr::ScalarSubquery(_)
2250 | Expr::Exists { .. }
2251 | Expr::InSubquery { .. }
2252 | Expr::RowInSubquery { .. }
2253 | Expr::RowCmpSubquery { .. }
2254 | Expr::Literal(_)
2255 | Expr::Placeholder(_)
2256 | Expr::Column(_) => {}
2257 // v7.30.2 — list elements can carry scalar subqueries
2258 // (`x IN (1, (SELECT …))`).
2259 Expr::InList { expr, list, .. } => {
2260 self.resolve_expr_subqueries(expr, cancel)?;
2261 for item in list {
2262 self.resolve_expr_subqueries(item, cancel)?;
2263 }
2264 }
2265 // v7.10.10 — recurse children.
2266 Expr::Array(items) => {
2267 for elem in items {
2268 self.resolve_expr_subqueries(elem, cancel)?;
2269 }
2270 }
2271 Expr::ArraySubscript { target, index } => {
2272 self.resolve_expr_subqueries(target, cancel)?;
2273 self.resolve_expr_subqueries(index, cancel)?;
2274 }
2275 Expr::ArraySlice { target, lo, hi } => {
2276 self.resolve_expr_subqueries(target, cancel)?;
2277 if let Some(l) = lo {
2278 self.resolve_expr_subqueries(l, cancel)?;
2279 }
2280 if let Some(h) = hi {
2281 self.resolve_expr_subqueries(h, cancel)?;
2282 }
2283 }
2284 Expr::AnyAll { expr, array, .. } => {
2285 self.resolve_expr_subqueries(expr, cancel)?;
2286 // Quantified subquery — an uncorrelated one
2287 // materialises up front; a correlated one stays for
2288 // the per-row resolver.
2289 if let Expr::ScalarSubquery(inner) = array.as_mut() {
2290 if !crate::subquery::select_is_correlated(inner) {
2291 let s = (**inner).clone();
2292 **array = self.materialize_quantified_rows(&s, cancel)?;
2293 }
2294 } else {
2295 self.resolve_expr_subqueries(array, cancel)?;
2296 }
2297 }
2298 Expr::Case {
2299 operand,
2300 branches,
2301 else_branch,
2302 } => {
2303 if let Some(o) = operand {
2304 self.resolve_expr_subqueries(o, cancel)?;
2305 }
2306 for (w, t) in branches {
2307 self.resolve_expr_subqueries(w, cancel)?;
2308 self.resolve_expr_subqueries(t, cancel)?;
2309 }
2310 if let Some(e) = else_branch {
2311 self.resolve_expr_subqueries(e, cancel)?;
2312 }
2313 }
2314 }
2315 Ok(())
2316 }
2317}
2318
2319impl Engine {
2320 /// v6.10.2 — projection for AS OF SEGMENT. Resolves
2321 /// `SelectItem::Wildcard` to all schema columns and
2322 /// `SelectItem::Expr` via the regular eval path.
2323 pub(crate) fn project_row_simple(
2324 &self,
2325 row: &Row<'static>,
2326 items: &[SelectItem],
2327 schema_cols: &[ColumnSchema],
2328 alias: &str,
2329 ) -> Result<Row<'static>, EngineError> {
2330 let ctx = self.ev_ctx(schema_cols, Some(alias));
2331 let cancel = CancelToken::none();
2332 let mut out_vals = Vec::new();
2333 for item in items {
2334 match item {
2335 // In a single-table projection (AS OF SEGMENT / RETURNING) a
2336 // qualified `t.*` covers exactly the same columns as a bare `*`.
2337 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => {
2338 out_vals.extend(row.values.iter().cloned());
2339 }
2340 SelectItem::Expr { expr, .. } => {
2341 let v = self.eval_expr_with_correlated(expr, row, &ctx, cancel, None)?;
2342 out_vals.push(v);
2343 }
2344 }
2345 }
2346 Ok(Row::new(out_vals))
2347 }
2348
2349 /// v6.10.2 — derive the output `ColumnSchema` list for an
2350 /// AS OF SEGMENT projection. Wildcards take the full schema;
2351 /// expressions take the alias if present or a synthetic
2352 /// `?column?` (PG convention) otherwise.
2353 pub(crate) fn derive_output_columns(
2354 &self,
2355 items: &[SelectItem],
2356 schema_cols: &[ColumnSchema],
2357 table_alias: &str,
2358 ) -> Vec<ColumnSchema> {
2359 let mut out = Vec::new();
2360 for item in items {
2361 match item {
2362 // `t.*` / `OLD.*` / `NEW.*` all mirror the full table schema in
2363 // a single-table projection.
2364 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => {
2365 out.extend(schema_cols.iter().cloned());
2366 }
2367 SelectItem::Expr { expr, alias } => {
2368 // Bare column references inherit the schema
2369 // column's name + type — PG names `RETURNING id`
2370 // "id" and types it BIGINT, and the sqlx embed
2371 // path type-checks RowDescription against the
2372 // Rust target (mailrs embed round-12).
2373 if let Expr::Column(col) = expr
2374 && let Some(sc) = schema_cols.iter().find(|c| c.name == col.name)
2375 {
2376 let name = alias.clone().unwrap_or_else(|| sc.name.clone());
2377 let mut c = ColumnSchema::new(name, sc.ty, sc.nullable);
2378 // v7.39 (read01 round 54) — carry the enum identity:
2379 // it lives outside the DataType lattice, so a derived
2380 // table built from this schema otherwise forgets it and
2381 // the OUTER `ORDER BY <enum col>` silently sorts by the
2382 // label's TEXT instead of member order.
2383 c.user_enum_type = sc.user_enum_type.clone();
2384 out.push(c);
2385 continue;
2386 }
2387 let name = alias.clone().unwrap_or_else(|| "?column?".to_string());
2388 // v7.30.4 (mailrs round-27, P0) — type the
2389 // expression with the same inference the SELECT
2390 // list uses (INT−INT=INT, BIGINT+INT=BIGINT…).
2391 // The old Text default broke every typed decode
2392 // of `RETURNING uidnext - 1 AS uid`: four days
2393 // of inbound mail indexed nowhere. Inference
2394 // failure keeps the old Text fallback rather
2395 // than inventing new error paths here.
2396 // v7.39 (round 258) — take the enum identity from the
2397 // same projection build, not just the type: a constant
2398 // SELECT (`SELECT 'ok'::mood AS x`, which is what a
2399 // VALUES row lowers to) is an EXPRESSION, so it landed
2400 // here and the derived table forgot the enum.
2401 let (ty, nullable) = build_projection(
2402 core::slice::from_ref(item),
2403 schema_cols,
2404 table_alias,
2405 self.backslash_escapes,
2406 )
2407 .ok()
2408 .and_then(|p| p.into_iter().next())
2409 .map_or((DataType::Text, true), |p| (p.ty, p.nullable));
2410 out.push(ColumnSchema::new(name, ty, nullable));
2411 }
2412 }
2413 }
2414 out
2415 }
2416
2417 /// v4.5: SELECT with cooperative cancellation. The token is
2418 /// honoured between UNION peers and inside the bare-SELECT row
2419 /// loop; HNSW kNN graph walks and the aggregate executor don't
2420 /// honour it yet (deferred — those paths bound their work
2421 /// internally by `LIMIT k` and `GROUP BY` cardinality).
2422 /// v7.38 (read01 P3.NEW3) — materialise a `spg_*` / `pg_*` meta-view by
2423 /// its (lowercased) name, or None if the name isn't a virtual view.
2424 /// Callers decide whether to return it directly (`SELECT *`) or stage
2425 /// it as a temp table for the full query pipeline.
2426 fn meta_view_result(&self, name: &str) -> Option<QueryResult> {
2427 Some(match name {
2428 "spg_statistic" => self.exec_spg_statistic(),
2429 "spg_stat_replication" => self.exec_spg_stat_replication(),
2430 "spg_stat_segment" => self.exec_spg_stat_segment(),
2431 "spg_memory_stats" => self.exec_spg_memory_stats(),
2432 "spg_stat_query" => self.exec_spg_stat_query(),
2433 "pg_stat_statements" => self.exec_pg_stat_statements(),
2434 "spg_stat_activity" => self.exec_spg_stat_activity(),
2435 "pg_stat_activity" => self.exec_pg_stat_activity(),
2436 "pg_locks" => self.exec_pg_locks(),
2437 "pg_statio_user_tables" => self.exec_pg_statio_user_tables(),
2438 "spg_stat_mvcc" => self.exec_spg_stat_mvcc(),
2439 "spg_partition_health" => self.exec_spg_partition_health(),
2440 "spg_audit_chain" => self.exec_spg_audit_chain(),
2441 "spg_audit_verify" => self.exec_spg_audit_verify(),
2442 "spg_table_ddl" => self.exec_spg_table_ddl(),
2443 "spg_role_ddl" => self.exec_spg_role_ddl(),
2444 "spg_database_ddl" => self.exec_spg_database_ddl(),
2445 _ => return None,
2446 })
2447 }
2448
2449 /// v7.39 (round 462) — the catalog an admin / stat view SELECT
2450 /// describes against: this engine's catalog with the view staged as a
2451 /// table, exactly as `exec_select_cancel_as` stages it for a
2452 /// non-bare query.
2453 ///
2454 /// These views never reach the catalog — each is a fixed row set built
2455 /// inside its own `exec_*` — so Describe reported no columns for all
2456 /// seventeen of them. Rows are deliberately not inserted: Describe
2457 /// only needs the shape, and `infer_column_types` reads the rows we
2458 /// already have in hand.
2459 pub(crate) fn admin_view_catalog(&self, stmt: &SelectStatement) -> Option<Catalog> {
2460 let from = stmt.from.as_ref()?;
2461 if !from.joins.is_empty() || self.active_catalog().get(&from.primary.name).is_some() {
2462 return None;
2463 }
2464 let lower = from.primary.name.to_ascii_lowercase();
2465 let QueryResult::Rows { columns, rows } = self.meta_view_result(&lower)? else {
2466 return None;
2467 };
2468 let mut catalog = self.active_catalog().clone();
2469 let cols = infer_column_types(&columns, &rows);
2470 catalog
2471 .create_table(TableSchema::new(from.primary.name.clone(), cols))
2472 .ok()?;
2473 Some(catalog)
2474 }
2475
2476 pub(crate) fn exec_select_cancel(
2477 &self,
2478 stmt: &SelectStatement,
2479 cancel: CancelToken<'_>,
2480 ) -> Result<QueryResult, EngineError> {
2481 self.exec_select_cancel_as(stmt, cancel, None)
2482 }
2483
2484 /// v7.39 (round 334, V55) — the same read core, authorised as
2485 /// `as_role`. A `SECURITY DEFINER` function's body runs as the
2486 /// function's OWNER: that is the entire point of the form, and without
2487 /// it every definer function failed with "permission denied" on the
2488 /// very table it exists to expose.
2489 /// v7.39 (round 559) — see the call site. `None` for anything but
2490 /// the bare shape, so every other query keeps its old path.
2491 fn try_bare_count_star(
2492 &self,
2493 stmt: &SelectStatement,
2494 as_role: Option<&str>,
2495 ) -> Result<Option<QueryResult>, EngineError> {
2496 use spg_sql::ast::SelectItem;
2497 if as_role.is_some()
2498 || !stmt.ctes.is_empty()
2499 || !stmt.unions.is_empty()
2500 || stmt.where_.is_some()
2501 || stmt.group_by.is_some()
2502 || stmt.having.is_some()
2503 || stmt.distinct
2504 || !stmt.order_by.is_empty()
2505 || stmt.limit.is_some()
2506 || stmt.offset.is_some()
2507 || stmt.items.len() != 1
2508 {
2509 return Ok(None);
2510 }
2511 let Some(from) = &stmt.from else {
2512 return Ok(None);
2513 };
2514 if !from.joins.is_empty()
2515 || stmt.locking.is_some()
2516 || from.primary.lateral_subquery.is_some()
2517 || from.primary.unnest_expr.is_some()
2518 || from.primary.generate_series_args.is_some()
2519 || from.primary.name.is_empty()
2520 || from.primary.name.starts_with("__spg_")
2521 {
2522 return Ok(None);
2523 }
2524 // A partition PARENT holds no rows of its own — they live in the
2525 // children — so its header count is 0 and the ordinary path has
2526 // to fan out. Caught by the partition conformance cases.
2527 //
2528 // v7.39 (round 645) — and an INHERITANCE parent holds only SOME
2529 // of them, which is worse: its header count is a real number,
2530 // just not the answer. `SELECT count(*) FROM par` returned 1
2531 // where PG returns 2, because this shortcut fired before the
2532 // fan-out could. The question is "does anything descend from
2533 // this", not "was it declared a partition parent".
2534 if crate::partition::has_children(self.active_catalog(), &from.primary.name) {
2535 return Ok(None);
2536 }
2537 let SelectItem::Expr { expr, alias } = &stmt.items[0] else {
2538 return Ok(None);
2539 };
2540 let spg_sql::ast::Expr::FunctionCall { name, args } = expr else {
2541 return Ok(None);
2542 };
2543 if !name.eq_ignore_ascii_case("count_star") || !args.is_empty() {
2544 return Ok(None);
2545 }
2546 // A row-security policy filters rows, so the header count is not
2547 // the answer; the ordinary path applies the policy.
2548 let Some(table) = self.active_catalog().get(&from.primary.name) else {
2549 return Ok(None);
2550 };
2551 if table.schema().row_security {
2552 return Ok(None);
2553 }
2554 // Rows frozen to the cold tier are not in `headers`, so the
2555 // header count would miss them. Caught by the cold-tier e2e.
2556 if table.has_cold_rows_fast() {
2557 return Ok(None);
2558 }
2559 let n = table.count_visible(&self.current_snapshot());
2560 let col = alias.clone().unwrap_or_else(|| String::from("count"));
2561 Ok(Some(QueryResult::Rows {
2562 columns: alloc::vec![ColumnSchema::new(col, DataType::BigInt, false)],
2563 rows: alloc::vec![Row::new(alloc::vec![Value::BigInt(
2564 i64::try_from(n).unwrap_or(i64::MAX)
2565 )])],
2566 }))
2567 }
2568
2569 /// v7.39 (round 560) — `SELECT <indexed col> FROM t WHERE <range on
2570 /// that col>` served from the index, never reading a row.
2571 ///
2572 /// Measured over pgwire on a 500k table, a 100k-row range: PG18's
2573 /// Index Only Scan 3.6 ms against SPG's 30 ms, widening with the row
2574 /// count (2x at 1k). PG needs its visibility map for this — a heap
2575 /// tuple carries its own visibility, so an index entry alone cannot
2576 /// say whether the row is live, and PG reads the heap for any page
2577 /// the map does not mark all-visible. SPG keeps a header array
2578 /// beside the rows, so the locator answers it directly and there is
2579 /// no map to be stale.
2580 /// v7.39 (round 564) — the shape test, once, for both the
2581 /// materialising scan and the streaming one.
2582 ///
2583 /// Two callers asking the same question in two places is how a fact
2584 /// starts drifting; the answer here is the single copy. Returns the
2585 /// table, the alias the predicate is written against, the projected
2586 /// column's position, and the name the single output column takes.
2587 pub(crate) fn index_only_shape<'s>(
2588 &'s self,
2589 stmt: &'s SelectStatement,
2590 ) -> Option<(&'s spg_storage::Table, &'s str, usize, String)> {
2591 use spg_sql::ast::SelectItem;
2592 if !stmt.ctes.is_empty()
2593 || !stmt.unions.is_empty()
2594 || stmt.group_by.is_some()
2595 || stmt.having.is_some()
2596 || stmt.distinct
2597 || stmt.locking.is_some()
2598 || !stmt.order_by.is_empty()
2599 || stmt.limit.is_some()
2600 || stmt.offset.is_some()
2601 || stmt.items.len() != 1
2602 {
2603 return None;
2604 }
2605 let (Some(from), Some(_)) = (&stmt.from, &stmt.where_) else {
2606 return None;
2607 };
2608 if !from.joins.is_empty()
2609 || from.primary.lateral_subquery.is_some()
2610 || from.primary.unnest_expr.is_some()
2611 || from.primary.generate_series_args.is_some()
2612 || from.primary.name.is_empty()
2613 || from.primary.name.starts_with("__spg_")
2614 {
2615 return None;
2616 }
2617 // v7.39 (round 645) — see the note on the sibling shortcut above:
2618 // an inheritance parent's own header count is not the answer.
2619 if crate::partition::has_children(self.active_catalog(), &from.primary.name) {
2620 return None;
2621 }
2622 let SelectItem::Expr { expr, alias } = &stmt.items[0] else {
2623 return None;
2624 };
2625 let spg_sql::ast::Expr::Column(c) = expr else {
2626 return None;
2627 };
2628 let alias_name = from.primary.alias.as_deref().unwrap_or(&from.primary.name);
2629 if let Some(q) = c.qualifier.as_deref()
2630 && !q.eq_ignore_ascii_case(alias_name)
2631 {
2632 return None;
2633 }
2634 let table = self.active_catalog().get(&from.primary.name)?;
2635 if table.schema().row_security {
2636 return None;
2637 }
2638 let cols = &table.schema().columns;
2639 let pos = cols
2640 .iter()
2641 .position(|s| s.name.eq_ignore_ascii_case(&c.name))?;
2642 let out = alias.clone().unwrap_or_else(|| cols[pos].name.clone());
2643 Some((table, alias_name, pos, out))
2644 }
2645
2646 /// v7.39 (round 565) — would this statement be answered out of the
2647 /// index alone?
2648 ///
2649 /// EXPLAIN has to name the node the executor will actually run, and
2650 /// the only honest way to know is to ask the same two questions the
2651 /// executor asks: the statement's shape, and everything decidable
2652 /// about the scan before it walks. Neither is re-stated here.
2653 pub(crate) fn stmt_takes_index_only_scan(&self, stmt: &SelectStatement) -> bool {
2654 let Some((table, alias_name, pos, _)) = self.index_only_shape(stmt) else {
2655 return false;
2656 };
2657 let Some(where_) = stmt.where_.as_ref() else {
2658 return false;
2659 };
2660 crate::index_access::index_only_precheck(
2661 where_,
2662 &table.schema().columns,
2663 table,
2664 alias_name,
2665 pos,
2666 )
2667 .is_some()
2668 }
2669
2670 fn try_index_only_scan(
2671 &self,
2672 stmt: &SelectStatement,
2673 ) -> Result<Option<QueryResult>, EngineError> {
2674 let Some((table, alias_name, pos, out_name)) = self.index_only_shape(stmt) else {
2675 return Ok(None);
2676 };
2677 // r1058 — same declines as `try_exec_joined_streaming`: CTEs
2678 // are not materialised here, and a partition parent's own
2679 // heap/indexes are empty (its rows live in the children).
2680 if !stmt.ctes.is_empty() {
2681 return Ok(None);
2682 }
2683 if let Some(from) = &stmt.from
2684 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
2685 {
2686 return Ok(None);
2687 }
2688 let where_ = stmt.where_.as_ref().expect("shape checked it");
2689 let cols = &table.schema().columns;
2690 let Some(values) = crate::index_access::try_index_only_range(
2691 where_,
2692 cols,
2693 table,
2694 alias_name,
2695 &self.current_snapshot(),
2696 pos,
2697 ) else {
2698 return Ok(None);
2699 };
2700 let schema = alloc::vec![ColumnSchema::new(
2701 out_name,
2702 cols[pos].ty,
2703 cols[pos].nullable
2704 )];
2705 Ok(Some(QueryResult::Rows {
2706 columns: schema,
2707 rows: values
2708 .into_iter()
2709 .map(|v| Row::new(alloc::vec![v]))
2710 .collect(),
2711 }))
2712 }
2713
2714 /// v7.39 (round 564) — the same scan, emitting each value instead of
2715 /// building a `Vec<Row>` for the encoder to walk once and drop.
2716 ///
2717 /// A profile of the server serving a 50k-row range put 10.2% of the
2718 /// connection thread's CPU on BUILDING that vector and another 9.7%
2719 /// on dropping it — a fifth of the query, spent allocating and
2720 /// freeing one single-element `Vec` per output row so that the wire
2721 /// encoder could borrow each value for a few nanoseconds. The
2722 /// streaming interface it then hands them to takes `&[Value]`
2723 /// already.
2724 ///
2725 /// Returns `None` when the shape does not apply, so the caller falls
2726 /// back before anything has been emitted.
2727 pub(crate) fn try_index_only_stream<F>(
2728 &self,
2729 stmt: &SelectStatement,
2730 emit: &mut F,
2731 ) -> Result<Option<usize>, EngineError>
2732 where
2733 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
2734 {
2735 let Some((table, alias_name, pos, out_name)) = self.index_only_shape(stmt) else {
2736 return Ok(None);
2737 };
2738 // r1058 — same declines as `try_exec_joined_streaming`: CTEs
2739 // are not materialised here, and a partition parent's own
2740 // heap/indexes are empty (its rows live in the children).
2741 if !stmt.ctes.is_empty() {
2742 return Ok(None);
2743 }
2744 if let Some(from) = &stmt.from
2745 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
2746 {
2747 return Ok(None);
2748 }
2749 let where_ = stmt.where_.as_ref().expect("shape checked it");
2750 let cols = &table.schema().columns;
2751 let schema = alloc::vec![ColumnSchema::new(
2752 out_name,
2753 cols[pos].ty,
2754 cols[pos].nullable
2755 )];
2756 let snapshot = self.current_snapshot();
2757 // The header goes out only once the walk has agreed to run — a
2758 // shape rejection after it would leave the client with a
2759 // RowDescription for a result that never comes.
2760 let mut wrote_header = false;
2761 let counted = crate::index_access::index_only_range_each(
2762 where_,
2763 cols,
2764 table,
2765 alias_name,
2766 &snapshot,
2767 pos,
2768 &mut |v: spg_storage::Value<'_>| {
2769 if !wrote_header {
2770 emit(crate::StreamItem::Header(&schema))?;
2771 wrote_header = true;
2772 }
2773 emit(crate::StreamItem::Row(crate::RowCells::Refs(&[&v])))
2774 },
2775 );
2776 match counted {
2777 None => Ok(None),
2778 Some(Err(e)) => Err(e),
2779 Some(Ok(n)) => {
2780 if !wrote_header {
2781 emit(crate::StreamItem::Header(&schema))?;
2782 }
2783 Ok(Some(n))
2784 }
2785 }
2786 }
2787
2788 /// `DISTINCT ON`'s de-duplication, which runs after the inner
2789 /// SELECT has produced its rows.
2790 ///
2791 /// `#[inline(never)]` and out of `exec_select_cancel_as` for the
2792 /// reason round 848 established: a debug build gives every branch's
2793 /// locals a slot in the frame whichever branch runs, and this one is
2794 /// eighty lines of hashing, key slicing and survivor sorting that a
2795 /// statement without `DISTINCT ON` never touches. Round 867
2796 /// measured `exec_select_cancel_as` holding ~46 KB on a path that
2797 /// reaches none of it — the segment that had been blamed on
2798 /// `exec_bare_select_cancel`, which turned out to hold 2 KB.
2799 #[inline(never)]
2800 fn apply_distinct_on(
2801 &self,
2802 result: QueryResult,
2803 don_hidden: usize,
2804 don_limit: &(
2805 Option<spg_sql::ast::LimitExpr>,
2806 Option<spg_sql::ast::LimitExpr>,
2807 ),
2808 don_top1: usize,
2809 orig_order_by: &[spg_sql::ast::OrderBy],
2810 ) -> Result<QueryResult, EngineError> {
2811 let QueryResult::Rows { columns, rows } = result else {
2812 return Ok(result);
2813 };
2814 // The keys are the hidden trailing columns appended above.
2815 // v7.39 (round 729) — top-1 mode: the trailing columns are the
2816 // DON keys plus the ORDER tail; keep each group's best in one
2817 // hash pass, then sort the SURVIVORS with the original spec.
2818 let mut kept: alloc::vec::Vec<Row<'static>>;
2819 let key_start;
2820 if don_top1 > 0 {
2821 let tail = don_top1 - 1;
2822 key_start = columns.len().saturating_sub(don_hidden + tail);
2823 let ord_start = key_start + don_hidden;
2824 let tail_dirs: alloc::vec::Vec<(bool, Option<bool>)> = orig_order_by[don_hidden..]
2825 .iter()
2826 .map(|o| (o.desc, o.nulls_first))
2827 .collect();
2828 let mysql = self.backslash_escapes;
2829 let better = |a: &Row<'static>, b: &Row<'static>| -> bool {
2830 for (k, (desc, nf)) in tail_dirs.iter().enumerate() {
2831 let av = a.values.get(ord_start + k).unwrap_or(&Value::Null);
2832 let bv = b.values.get(ord_start + k).unwrap_or(&Value::Null);
2833 match crate::order_by_value_cmp_in(*desc, *nf, av, bv, mysql) {
2834 core::cmp::Ordering::Less => return true,
2835 core::cmp::Ordering::Greater => return false,
2836 core::cmp::Ordering::Equal => {}
2837 }
2838 }
2839 false
2840 };
2841 let mut slot: hashbrown::HashMap<String, usize> = hashbrown::HashMap::new();
2842 let mut best: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
2843 let mut keybuf = String::new();
2844 for row in rows {
2845 keybuf.clear();
2846 for v in row.values.get(key_start..ord_start).unwrap_or(&[]) {
2847 aggregate::push_canonical_key(&mut keybuf, v);
2848 }
2849 match slot.get(keybuf.as_str()) {
2850 Some(&i) => {
2851 if better(&row, &best[i]) {
2852 best[i] = row;
2853 }
2854 }
2855 None => {
2856 slot.insert(keybuf.clone(), best.len());
2857 best.push(row);
2858 }
2859 }
2860 }
2861 // Survivors sort with the FULL original spec (keys are still
2862 // aboard as hidden columns).
2863 let full_dirs: alloc::vec::Vec<(bool, Option<bool>)> = orig_order_by
2864 .iter()
2865 .map(|o| (o.desc, o.nulls_first))
2866 .collect();
2867 best.sort_by(|a, b| {
2868 for (k, (desc, nf)) in full_dirs.iter().enumerate() {
2869 let av = a.values.get(key_start + k).unwrap_or(&Value::Null);
2870 let bv = b.values.get(key_start + k).unwrap_or(&Value::Null);
2871 match crate::order_by_value_cmp_in(*desc, *nf, av, bv, mysql) {
2872 core::cmp::Ordering::Equal => {}
2873 o => return o,
2874 }
2875 }
2876 core::cmp::Ordering::Equal
2877 });
2878 for r in &mut best {
2879 r.values.truncate(key_start);
2880 }
2881 kept = best;
2882 } else {
2883 key_start = columns.len().saturating_sub(don_hidden);
2884 let mut seen: alloc::vec::Vec<alloc::vec::Vec<Value<'static>>> = alloc::vec::Vec::new();
2885 kept = alloc::vec::Vec::new();
2886 for mut row in rows {
2887 let key: alloc::vec::Vec<Value<'static>> =
2888 row.values.get(key_start..).unwrap_or(&[]).to_vec();
2889 if seen.iter().any(|k| k == &key) {
2890 continue;
2891 }
2892 seen.push(key);
2893 row.values.truncate(key_start);
2894 kept.push(row);
2895 }
2896 }
2897 let mut columns = columns;
2898 columns.truncate(key_start);
2899 // PG limits what DISTINCT ON left, not what fed it.
2900 let kept = apply_deferred_limit(kept, don_limit);
2901 Ok(QueryResult::Rows {
2902 columns,
2903 rows: kept,
2904 })
2905 }
2906
2907 pub(crate) fn exec_select_cancel_as(
2908 &self,
2909 stmt: &SelectStatement,
2910 cancel: CancelToken<'_>,
2911 as_role: Option<&str>,
2912 ) -> Result<QueryResult, EngineError> {
2913 // v7.39 (round 763, F31-C1) — `SELECT *, count(*) … GROUP BY
2914 // <all columns>` is legal PG (the wildcard expands to grouped
2915 // columns); SPG refused the whole shape. Expand the wildcard
2916 // into explicit column refs up front — the aggregate layer's
2917 // existing "must appear in the GROUP BY clause" validation
2918 // then answers PG's sentence for any non-grouped column.
2919 if let Some(expanded) = self.expand_aggregate_wildcard(stmt) {
2920 return self.exec_select_cancel_as(&expanded, cancel, as_role);
2921 }
2922 // v7.39 (round 559) — `SELECT count(*) FROM t` without touching
2923 // a row.
2924 //
2925 // The aggregate layer already short-circuits this to
2926 // `rows.len()`, so the O(1) part was never the problem — the
2927 // cost is UPSTREAM, materialising every visible row so that
2928 // layer can take its length. Measured over pgwire on 500k rows:
2929 // PG18 8.2 ms with two parallel workers, 10.3 ms with
2930 // parallelism off, SPG 16.5 ms — 1.6x slower than a
2931 // single-threaded PG on the commonest aggregate there is, and no
2932 // ledger entry recorded it.
2933 //
2934 // Counting visible HEADERS needs no row at all. PG cannot do
2935 // this: its visibility lives in the heap tuples themselves, so
2936 // it has to read them (that is why its own count(*) is a full
2937 // scan, parallel or not).
2938 // v7.39 (read01 round 57) — the table-privilege gate on the common
2939 // read core. A superuser session returns from it immediately.
2940 // v7.39 (round 529) — resolve an ORDER BY that names an output
2941 // ALIAS. The statement-level pass never reached a SELECT nested in
2942 // a FROM clause, a CTE or a scalar subquery, so the same query
2943 // worked on its own and failed the moment anything wrapped it —
2944 // which is what generated SQL does constantly.
2945 let aliased;
2946 let stmt = if crate::orderby::order_by_names_an_alias(stmt) {
2947 let mut s = stmt.clone();
2948 crate::orderby::resolve_order_by_position(&mut s);
2949 aliased = s;
2950 &aliased
2951 } else {
2952 stmt
2953 };
2954 // v7.39 (round 529) — DISTINCT ON needs two things it did not have.
2955 //
2956 // Its keys were evaluated against the PROJECTED row, so a key that
2957 // is not in the select list — `SELECT DISTINCT ON (g) v FROM t
2958 // ORDER BY g, v DESC`, the canonical "latest row per group" — could
2959 // not be read at all and the query failed. PG evaluates them on the
2960 // input. They are projected as hidden columns here and stripped
2961 // again below, the same way the grouping-set ordering columns
2962 // already travel.
2963 //
2964 // And the dedup ran AFTER the inner statement's LIMIT, so
2965 // `… DISTINCT ON (g) … LIMIT 2` on four rows answered ONE row where
2966 // PG answers two: the limit had already taken two rows of the same
2967 // group before anything deduplicated them. A paginated DISTINCT ON
2968 // returned short pages, with no error. The limit is deferred to
2969 // after the dedup, which is PG's order.
2970 let don_stmt;
2971 // v7.39 (round 729) — the top-1 consumer needs the ORIGINAL
2972 // order spec (the rewritten stmt's is emptied).
2973 let orig_order_by = stmt.order_by.clone();
2974 let (stmt, don_hidden, don_limit, don_top1) = if stmt.distinct_on.is_empty() {
2975 (stmt, 0, (None, None), 0usize)
2976 } else {
2977 let mut s = stmt.clone();
2978 let hidden = s.distinct_on.len();
2979 for (i, e) in stmt.distinct_on.iter().enumerate() {
2980 s.items.push(SelectItem::Expr {
2981 expr: e.clone(),
2982 alias: Some(alloc::format!("__distinct_on_{i}")),
2983 });
2984 }
2985 // v7.39 (round 729) — group-top-1 short circuit. When the
2986 // DISTINCT ON keys are exactly the ORDER BY's leading keys,
2987 // the answer is "per group, the row that wins the remaining
2988 // order" — a single O(n) hash pass. The old path sorted the
2989 // ENTIRE input first (500k rows, ~180 ms on the panel cell)
2990 // to keep 100. The inner query runs UNSORTED with every
2991 // order key appended as a hidden column; the dedup below
2992 // keeps each group's best, then sorts the SURVIVORS.
2993 // Declared-collation order keys stay on the sorting path
2994 // (the value comparator here is collation-blind).
2995 let prefix_matches = s.order_by.len() >= hidden
2996 && stmt
2997 .distinct_on
2998 .iter()
2999 .zip(s.order_by.iter())
3000 .all(|(d, o)| *d == o.expr && !o.desc && o.nulls_first.is_none());
3001 let colls_plain =
3002 crate::orderby::order_by_collations(&s.order_by, &self.ev_ctx(&[], None))
3003 .map(|cs| cs.iter().all(Option::is_none))
3004 .unwrap_or(false);
3005 let top1_tail = if prefix_matches && colls_plain && s.group_by.is_none() {
3006 let tail = s.order_by.len() - hidden;
3007 for (j, o) in s.order_by[hidden..].iter().enumerate() {
3008 s.items.push(SelectItem::Expr {
3009 expr: o.expr.clone(),
3010 alias: Some(alloc::format!("__don_ord_{j}")),
3011 });
3012 }
3013 // Carry the tail's direction flags through the aliases'
3014 // ORDER; the survivors re-sort below with the full spec.
3015 s.order_by = Vec::new();
3016 tail + 1 // sentinel: 1 + number of tail keys (0 tail is still active)
3017 } else {
3018 0
3019 };
3020 // Only a folded literal is deferred; a placeholder or an
3021 // expression keeps the path it has today rather than being
3022 // resolved a second way here.
3023 let deferrable = matches!(
3024 (&s.limit, &s.offset),
3025 (
3026 None | Some(spg_sql::ast::LimitExpr::Literal(_)),
3027 None | Some(spg_sql::ast::LimitExpr::Literal(_))
3028 )
3029 );
3030 let deferred = if deferrable {
3031 (s.limit.take(), s.offset.take())
3032 } else {
3033 (None, None)
3034 };
3035 don_stmt = s;
3036 (&don_stmt, hidden, deferred, top1_tail)
3037 };
3038 self.acl_check_select_as(stmt, as_role)?;
3039 validate_aggregate_placement(stmt)?;
3040 // v7.39 (round 559) — the bare `count(*)` fast path, AFTER the
3041 // privilege gate above. Placed before it at first, and the
3042 // security-definer e2e caught it immediately: a SECURITY INVOKER
3043 // function whose body is `SELECT count(*) FROM t` answered
3044 // instead of being refused, because the fast path never reached
3045 // the check.
3046 if let Some(r) = self.try_bare_count_star(stmt, as_role)? {
3047 return Ok(r);
3048 }
3049 // v7.39 (round 560) — an index-only range scan. Same placement
3050 // reasoning as the count above: after the privilege gate.
3051 if let Some(r) = self.try_index_only_scan(stmt)? {
3052 return Ok(r);
3053 }
3054 validate_locking_clause(stmt)?;
3055 let result = self.exec_select_cancel_inner(stmt, cancel)?;
3056 // v7.39 (round 135) — drop the synthetic `__grp_ord_*` ordering columns
3057 // the parser injects for GROUPING() in ORDER BY on a grouping-set query.
3058 // They carry the per-branch mask through the UNION-ALL sort and must not
3059 // appear in the output. Stripped per SELECT level (grouping-set queries
3060 // are often wrapped in a derived subquery), before DISTINCT ON.
3061 let result = strip_synthetic_order_cols(result);
3062 // v7.37.17 (17.6 siblings) — `SELECT DISTINCT ON (exprs)`:
3063 // rows arrive here already ORDER BY'd; keep the FIRST row of
3064 // each group the expressions define (PG semantics). The
3065 // expressions evaluate against the projected schema — an
3066 // expression that isn't in the select list errors honestly.
3067 if stmt.distinct_on.is_empty() {
3068 return Ok(result);
3069 }
3070 self.apply_distinct_on(result, don_hidden, &don_limit, don_top1, &orig_order_by)
3071 }
3072
3073 /// The UNION chain: execute the head as a bare block, then fold each
3074 /// peer in with left-associative dedup.
3075 ///
3076 /// `#[inline(never)]` and out of `exec_select_cancel_inner` for the
3077 /// reason round 848 established. A statement with no unions returns
3078 /// one line above the call — and every nested subquery on a deep
3079 /// path is such a statement, so each level of the recursion carried
3080 /// 170 lines of locals it could not reach. Round 867 measured that
3081 /// frame at 34,800 bytes, the largest single one on the descent,
3082 /// after two earlier attributions had blamed its caller and then its
3083 /// callee: the gap between two marks is the frame of everything
3084 /// BETWEEN them, and this function had no mark of its own.
3085 #[inline(never)]
3086 fn exec_union_chain(
3087 &self,
3088 stmt_ref: &SelectStatement,
3089 stmt: &SelectStatement,
3090 cancel: CancelToken<'_>,
3091 ) -> Result<QueryResult, EngineError> {
3092 // UNION path: clone-strip the head into a bare block (its own
3093 // DISTINCT and any inner ORDER BY are dropped by parser rule —
3094 // the wrapper SelectStatement carries them), execute, then chain
3095 // peers with left-associative dedup semantics.
3096 // v7.39 (round 232) — the wrapper's ORDER BY addresses the head's
3097 // output columns; a position past their count is PG's 42P10.
3098 crate::orderby::check_order_by_positions(stmt_ref)?;
3099 let mut head_unknown = branch_unknown_mask(stmt_ref);
3100 let head_regcast = branch_regcast_mask(stmt_ref);
3101 let mut head = stmt_ref.clone();
3102 head.unions = Vec::new();
3103 head.order_by = Vec::new();
3104 head.limit = None;
3105 let QueryResult::Rows {
3106 mut columns,
3107 mut rows,
3108 } = self.exec_bare_select_cancel(&head, cancel)?
3109 else {
3110 unreachable!("bare SELECT cannot return CommandOk")
3111 };
3112 for (kind, peer) in &stmt_ref.unions {
3113 // v7.37.17 (17.6 siblings) — a peer carrying its own
3114 // unions is a nested INTERSECT group (the parser's
3115 // precedence regrouping); recurse through the
3116 // union-aware wrapper for it.
3117 let peer_result = if peer.unions.is_empty() {
3118 self.exec_bare_select_cancel(peer, cancel)?
3119 } else {
3120 self.exec_select_cancel(peer, cancel)?
3121 };
3122 let QueryResult::Rows {
3123 columns: peer_cols,
3124 rows: mut peer_rows,
3125 } = peer_result
3126 else {
3127 unreachable!("bare SELECT cannot return CommandOk")
3128 };
3129 if peer_cols.len() != columns.len() {
3130 // v7.39 (round 232) — PG's wording, which clients match on.
3131 return Err(EngineError::Unsupported(alloc::format!(
3132 "each {} query must have the same number of columns",
3133 set_op_name(*kind)
3134 )));
3135 }
3136 // v7.39 (round 232+233) — PG resolves each result column to one
3137 // type before it merges anything, and refuses the query when the
3138 // two branches have no common type. SPG's unifier
3139 // (`unify_union_columns`) is value-driven and deliberately
3140 // conservative — "a column where any cell fails to coerce is left
3141 // exactly as it was" — so a mismatch produced a column holding
3142 // BOTH types (`SELECT a, b FROM t UNION SELECT b, a FROM t` came
3143 // back with integers and text interleaved) instead of an error.
3144 //
3145 // The check has to read the branch ASTs, not just their schemas:
3146 // SPG has no `Unknown` DataType, so a bare `'a'` literal describes
3147 // as TEXT and is indistinguishable from a real text column by
3148 // schema alone — yet PG treats the two completely differently
3149 // (`SELECT 1 UNION SELECT 'a'` is an input-syntax error on the
3150 // literal, `SELECT 1 UNION SELECT 'a'::text` is a type mismatch).
3151 let peer_unknown = branch_unknown_mask(peer);
3152 let peer_regcast = branch_regcast_mask(peer);
3153 for i in 0..columns.len() {
3154 let hu = head_unknown.get(i).copied().unwrap_or(false);
3155 let pu = peer_unknown.get(i).copied().unwrap_or(false);
3156 let (ht, pt) = (columns[i].ty, peer_cols[i].ty);
3157 let reg_dual = peer_regcast.get(i).copied().unwrap_or(false)
3158 || head_regcast.get(i).copied().unwrap_or(false);
3159 match (hu, pu) {
3160 // Both sides carry a real type: they must share a category.
3161 (false, false) => {
3162 if !reg_dual && !crate::conversions::types_unify(ht, pt) {
3163 return Err(EngineError::Unsupported(alloc::format!(
3164 "{} types {} and {} cannot be matched",
3165 set_op_name(*kind),
3166 crate::conversions::pg_type_name_for_error(ht),
3167 crate::conversions::pg_type_name_for_error(pt),
3168 )));
3169 }
3170 }
3171 // One side is an untyped literal: it takes the other's
3172 // type, and failing to convert is the error PG reports.
3173 (true, false) => {
3174 coerce_branch_column(&mut rows, i, pt, &columns[i].name)?;
3175 columns[i].ty = pt;
3176 head_unknown[i] = false;
3177 }
3178 (false, true) => {
3179 coerce_branch_column(&mut peer_rows, i, ht, &columns[i].name)?;
3180 }
3181 // Both untyped — nothing to resolve against yet.
3182 (true, true) => {}
3183 }
3184 }
3185 // v7.37 D.26 — a UNION result column is nullable when ANY branch is
3186 // nullable (PG semantics). Previously the result kept only the head's
3187 // nullability, so `VALUES (1),(NULL)` (a UNION-ALL chain seeded by the
3188 // non-null `1`) wrongly reported the column NOT NULL, which let
3189 // `count(col)`'s NOT-NULL fast-path count the NULL row.
3190 for (i, pc) in peer_cols.iter().enumerate() {
3191 if pc.nullable {
3192 columns[i].nullable = true;
3193 }
3194 }
3195 // v7.39 (round 410) — under MySQL, set-op dedup / matching folds
3196 // text by the session collation (CI + accent + PAD SPACE), like
3197 // GROUP BY. PG stays byte-exact.
3198 let mysql = self.backslash_escapes;
3199 match kind {
3200 UnionKind::All => rows.extend(peer_rows),
3201 UnionKind::Distinct => {
3202 rows.extend(peer_rows);
3203 rows = dedup_rows(rows, mysql);
3204 }
3205 // v7.37.17 (17.6 siblings) — PG set semantics.
3206 // v7.39 (round 591) — all four ask the same question of the
3207 // right side, and all four used to answer it by scanning it
3208 // once per left row. `PeerIndex` buckets it by the hash
3209 // DISTINCT already uses, so the answer is a lookup.
3210 // INTERSECT: distinct rows present on both sides.
3211 UnionKind::Intersect => {
3212 let idx = PeerIndex::build(&peer_rows, mysql);
3213 rows = dedup_rows(rows, mysql)
3214 .into_iter()
3215 .filter(|r| idx.contains(r))
3216 .collect();
3217 }
3218 // INTERSECT ALL: multiset intersection — each row
3219 // keeps min(left count, right count) occurrences.
3220 UnionKind::IntersectAll => {
3221 let mut idx = PeerIndex::build(&peer_rows, mysql);
3222 let mut kept: Vec<Row<'static>> = Vec::new();
3223 for r in rows {
3224 if idx.take_one(&r) {
3225 kept.push(r);
3226 }
3227 }
3228 rows = kept;
3229 }
3230 // EXCEPT: distinct left rows absent from the right.
3231 UnionKind::Except => {
3232 let idx = PeerIndex::build(&peer_rows, mysql);
3233 rows = dedup_rows(rows, mysql)
3234 .into_iter()
3235 .filter(|r| !idx.contains(r))
3236 .collect();
3237 }
3238 // EXCEPT ALL: multiset subtraction — each right
3239 // occurrence cancels one left occurrence.
3240 UnionKind::ExceptAll => {
3241 let mut idx = PeerIndex::build(&peer_rows, mysql);
3242 let mut kept: Vec<Row<'static>> = Vec::new();
3243 for r in rows {
3244 if !idx.take_one(&r) {
3245 kept.push(r);
3246 }
3247 }
3248 rows = kept;
3249 }
3250 }
3251 }
3252 // PG resolves a UNION / VALUES result column to one common type
3253 // and casts every branch to it (`SELECT '2020-01-01'::date UNION
3254 // ALL SELECT '2020-01-02'` → both DATE, not DATE + TEXT). SPG
3255 // built each branch independently, leaving mixed-type columns
3256 // that broke ORDER BY, comparisons, and value-based window
3257 // frames. Unify + coerce before the combined ORDER BY sees them.
3258 unify_union_columns(&mut columns, &mut rows);
3259 // ORDER BY at the top of a UNION applies to the combined result.
3260 // Eval against the projected schema (NOT the source table).
3261 if !stmt.order_by.is_empty() {
3262 // v7.39 (read01 round 54) — the combined-result ctx must carry the
3263 // catalog, and the projected columns must keep their enum identity
3264 // (`user_enum_type`), or `ORDER BY <enum col>` over a UNION sorts
3265 // by TEXT instead of member order — silently wrong rows, not an
3266 // error. (Same shape as the enum-order knife's GROUP BY fix.)
3267 let synth_ctx = EvalContext::new(&columns, None).with_catalog(self.active_catalog());
3268 // v7.37.17 (17.6 siblings) — positional keys (ORDER BY 1)
3269 // survive to here when the head projects a Wildcard (the
3270 // group-tail wrapper shape): map them onto the Nth
3271 // projected column so the combined sort works.
3272 let resolved_order: Vec<spg_sql::ast::OrderBy> = stmt
3273 .order_by
3274 .iter()
3275 .map(|o| {
3276 let mut o = o.clone();
3277 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &o.expr
3278 && *n >= 1
3279 && let Ok(idx) = usize::try_from(*n - 1)
3280 && idx < columns.len()
3281 {
3282 o.expr = Expr::Column(spg_sql::ast::ColumnName {
3283 qualifier: None,
3284 name: columns[idx].name.clone(),
3285 });
3286 }
3287 o
3288 })
3289 .collect();
3290 let descs: Vec<bool> = resolved_order.iter().map(|o| o.desc).collect();
3291 let mut tagged: Vec<(Vec<OrderKey>, Row)> = Vec::with_capacity(rows.len());
3292 for r in rows {
3293 let keys = build_order_keys(&resolved_order, &r, &synth_ctx)?;
3294 tagged.push((keys, r));
3295 }
3296 sort_by_keys(&mut tagged, &descs);
3297 rows = tagged.into_iter().map(|(_, r)| r).collect();
3298 }
3299 apply_offset_and_limit(&mut rows, stmt.offset_literal(), stmt.limit_literal());
3300 Ok(QueryResult::Rows { columns, rows })
3301 }
3302
3303 fn exec_select_cancel_inner(
3304 &self,
3305 stmt: &SelectStatement,
3306 cancel: CancelToken<'_>,
3307 ) -> Result<QueryResult, EngineError> {
3308 cancel.check()?;
3309 // v7.38 P0 元机制 A — first observable point inside the
3310 // planner / executor. Tests use this to inject a delay or
3311 // a cancellation race before any row is produced. Release
3312 // build expands to `let _ = (...);` — zero cost.
3313 crate::injection_point!("planner_first_row_fetch", &stmt.from);
3314 // v7.39 (round 705) — WINDOW-clause definitions nothing referenced.
3315 // PG analyses every definition, referenced or not, so `SELECT i FROM
3316 // t WINDOW w AS (ORDER BY nosuch)` fails there and silently
3317 // succeeded here (the parser used to drop the unreferenced defs
3318 // whole). The check is the CREATE VIEW check's shape (round 700): a
3319 // LIMIT-0 run of the same FROM with the definitions' key
3320 // expressions as the projection — it cannot disagree with what a
3321 // referencing window would have done, because it resolves the same
3322 // names the same way. Zero cost for the ordinary statement: the
3323 // list is empty unless a WINDOW clause left unreferenced defs.
3324 if !stmt.window_check_exprs.is_empty() {
3325 let mut probe = stmt.clone();
3326 probe.items = stmt
3327 .window_check_exprs
3328 .iter()
3329 .map(|e| spg_sql::ast::SelectItem::Expr {
3330 expr: e.clone(),
3331 alias: None,
3332 })
3333 .collect();
3334 probe.window_check_exprs = Vec::new();
3335 probe.distinct = false;
3336 probe.distinct_on = Vec::new();
3337 probe.group_by = None;
3338 probe.group_by_all = false;
3339 probe.having = None;
3340 probe.unions = Vec::new();
3341 probe.order_by = Vec::new();
3342 probe.locking = None;
3343 probe.limit = Some(spg_sql::ast::LimitExpr::Literal(0));
3344 probe.offset = None;
3345 probe.limit_with_ties = false;
3346 self.exec_select_cancel_inner(&probe, cancel)?;
3347 }
3348 // v7.39 (read01 round 74) — lower `(f(args)).*`. Naming a record's fields
3349 // takes the catalog, so the parser leaves a marker and the rewrite lands
3350 // here: the call moves into a LATERAL FROM item and the item becomes one
3351 // reference per declared column. `SELECT 'p', (rows_of(2)).*` is
3352 // `SELECT 'p', __rec.id, __rec.v FROM rows_of(2) AS __rec` — reusing the
3353 // set-returning FROM machinery of rounds 65 and 69 rather than growing a
3354 // second one.
3355 if let Some(lowered) = self.lower_record_expansion(stmt)? {
3356 return self.exec_select_cancel_inner(&lowered, cancel);
3357 }
3358 // v7.17.0 Phase 1.2 — user-defined VIEW expansion. If the
3359 // FROM / JOIN graph references any catalogued view name,
3360 // re-parse the view body and prepend it as a synthetic
3361 // CTE. Recurses on views-in-views via the regular CTE
3362 // dispatch below. Fast-path: skip the walker entirely when
3363 // the catalog has no views (the typical OLTP load).
3364 if !self.active_catalog().views_all().is_empty() {
3365 if let Some(rewritten) = self.expand_views_in_select(stmt)? {
3366 return self.exec_select_cancel(&rewritten, cancel);
3367 }
3368 }
3369 // v7.37.6-B(sentori Epic 2 P0)— `SELECT … FROM <partition-parent>`
3370 // gets rewritten to a UNION-ALL over the children that overlap
3371 // the WHERE-derived key range. Uses the same CTE-injection
3372 // trick as VIEW expansion above so downstream resolution
3373 // doesn't need a partition-aware code path.
3374 if let Some(rewritten) = self.expand_partition_parents_in_select(stmt)? {
3375 return self.exec_select_cancel(&rewritten, cancel);
3376 }
3377 // v7.16.2 — information_schema / pg_catalog virtual
3378 // views (mailrs round-10 A.3). If the SELECT touches a
3379 // synthetic meta-table name (`__spg_info_*` /
3380 // `__spg_pg_*` — produced by the parser for
3381 // `information_schema.X` / `pg_catalog.X`), clone the
3382 // catalog, materialise the requested view as a real
3383 // temporary table, and re-execute against an enriched
3384 // engine. Same pattern as `exec_with_ctes` for CTEs.
3385 if !self.meta_views_materialised && select_references_meta_view(stmt) {
3386 return self.exec_select_with_meta_views(stmt, cancel);
3387 }
3388 // v6.10.2 — cold-tier time-travel short-circuit. When the
3389 // primary TableRef carries `AS OF SEGMENT '<id>'`, run a
3390 // dedicated cold-segment scan instead of the regular
3391 // hot+index path. The scope is intentionally narrow for
3392 // v6.10.2 — bare `SELECT * FROM <t> AS OF SEGMENT 'id'`,
3393 // optionally with a single-column-equality WHERE. JOINs /
3394 // aggregates / ORDER BY / subqueries on top of a time-
3395 // travelled scan are STABILITY § "Out of v6.10".
3396 if let Some(from) = &stmt.from
3397 && let Some(seg_id) = from.primary.as_of_segment
3398 {
3399 return self.exec_select_as_of_segment(stmt, from, seg_id);
3400 }
3401 // v6.2.0 / v6.5.0 — virtual-table short-circuits. Detected
3402 // pre-CTE because they don't read from the catalog and
3403 // shouldn't participate in regular FROM resolution.
3404 // v6.2.0 / v6.5.0 / v7.38 (read01 P3.NEW3) — virtual-table
3405 // short-circuits. A meta-view FROM materialises to a fixed row
3406 // set. For a bare `SELECT *` we return it directly; otherwise we
3407 // stage it as a temp table and run the normal pipeline, so
3408 // projection / WHERE / ORDER BY / aggregates work over these views
3409 // (they were `SELECT *`-only before). A real table shadowing the
3410 // name wins (checked first), which also stops the staged re-run
3411 // from recursing back into meta-view detection.
3412 if let Some(from) = &stmt.from
3413 && from.joins.is_empty()
3414 && self.active_catalog().get(&from.primary.name).is_none()
3415 {
3416 let lower = from.primary.name.to_ascii_lowercase();
3417 if let Some(result) = self.meta_view_result(&lower) {
3418 let bare = stmt.where_.is_none()
3419 && stmt.group_by.is_none()
3420 && stmt.having.is_none()
3421 && stmt.unions.is_empty()
3422 && stmt.order_by.is_empty()
3423 && stmt.limit.is_none()
3424 && stmt.offset.is_none()
3425 && !stmt.distinct
3426 && stmt.items.iter().all(|i| matches!(i, SelectItem::Wildcard));
3427 if bare {
3428 return Ok(result);
3429 }
3430 if let QueryResult::Rows { columns, rows } = result {
3431 let mut catalog = self.active_catalog().clone();
3432 let cols = infer_column_types(&columns, &rows);
3433 let schema = TableSchema::new(from.primary.name.clone(), cols);
3434 catalog.create_table(schema).map_err(EngineError::Storage)?;
3435 let t = catalog
3436 .get_mut(&from.primary.name)
3437 .expect("just-created meta-view table must exist");
3438 for row in rows {
3439 t.insert(row).map_err(EngineError::Storage)?;
3440 }
3441 let mut eng = Engine::restore(catalog);
3442 if let Some(c) = self.clock {
3443 eng = eng.with_clock(c);
3444 }
3445 if let Some(f) = self.salt_fn {
3446 eng = eng.with_salt_fn(f);
3447 }
3448 // v7.39 (read01 pgstatfuncs.c) — carry the calling-
3449 // connection identity so `WHERE pid = pg_backend_pid()`
3450 // matches inside the staged meta-view run.
3451 if let Some(f) = self.backend_pid_fn {
3452 eng.set_backend_pid_fn(f);
3453 }
3454 return eng.exec_select_cancel(stmt, cancel);
3455 }
3456 return Ok(result);
3457 }
3458 }
3459 // v4.11: CTEs materialise into a temporary enriched catalog
3460 // *before* anything else — the body SELECT can then refer
3461 // to CTE names via the regular FROM-clause resolution.
3462 // Uncorrelated only: each CTE body runs once against the
3463 // current catalog, not against later CTEs' results (left-
3464 // to-right materialisation would relax this, but we keep
3465 // it simple for v4.11 MVP).
3466 if !stmt.ctes.is_empty() {
3467 return self.exec_with_ctes(stmt, cancel);
3468 }
3469 // v4.10: subqueries (uncorrelated) are resolved here, before
3470 // the executor sees the row loop. We clone the statement so
3471 // we can mutate without disturbing the caller's AST — most
3472 // queries pass through with no subquery nodes and the clone
3473 // is cheap; with subqueries the materialisation cost
3474 // dominates anyway.
3475 let mut stmt_owned;
3476 let stmt_ref: &SelectStatement = if expr_tree_has_subquery(stmt) {
3477 stmt_owned = stmt.clone();
3478 // v7.33 (mailrs 7.32.1) — sublink pull-up first: an
3479 // aggregate-wrapped correlated scalar subquery whose
3480 // correlation key is UNIQUE/PK becomes a LEFT JOIN, so the
3481 // executor streams one join instead of splicing a per-row
3482 // subplan. Runs before the per-row/batch resolver, which then
3483 // only sees the subqueries the pull-up left behind.
3484 self.pull_up_unique_correlated_agg_subqueries(&mut stmt_owned);
3485 // v7.37.4 (A — correlated LIMIT 1 ORDER BY DESC pull-up) —
3486 // the "per-key latest" scalar subquery shape (inbox / feed
3487 // / timeline applications) becomes a CTE + LEFT JOIN
3488 // against a GROUP BY pre-aggregation that reuses the v7.33
3489 // first_ordered argmax executor. Runs AFTER unique-key
3490 // pull-up (so the unique-key fast path still wins for
3491 // single-PK lookups) and BEFORE the EXISTS sublink rewrite.
3492 // Phase 1 (this commit) is skeleton only — no-op pass.
3493 self.pull_up_correlated_limit_one_subqueries(&mut stmt_owned);
3494 // v7.34.2 (mailrs prod NOT EXISTS) — plan-time `[NOT] EXISTS`
3495 // sublink pull-up to semi/anti-join, before the resolver gets
3496 // a chance to walk per-row.
3497 self.pull_up_exists_sublinks(&mut stmt_owned);
3498 // v7.37.4 — if the LIMIT 1 pullup added CTEs, route through
3499 // exec_with_ctes so they materialise once before the body
3500 // SELECT runs. exec_with_ctes strips ctes from the body
3501 // clone, then re-enters select.
3502 if !stmt_owned.ctes.is_empty() {
3503 return self.exec_with_ctes(&stmt_owned, cancel);
3504 }
3505 // v7.37.x (docker-fair INSUBQ attack) — short-circuit
3506 // SELECT COUNT(*) FROM A WHERE A.pk IN (<uncorrelated subquery>)
3507 // BEFORE `resolve_select_subqueries` materialises the inner
3508 // result as `Vec<Expr::Literal>` (~150 µs for the 6 k-row
3509 // INSUBQ benchmark). Run the inner once, collect the result
3510 // values into a `HashSet<i64>` directly, then probe A.pk per
3511 // value and tally. Returns `Some` when the shape matches.
3512 if let Some(out) = self.try_count_star_pk_in_subquery_fast(&stmt_owned, cancel)? {
3513 return Ok(out);
3514 }
3515 self.resolve_select_subqueries(&mut stmt_owned, cancel)?;
3516 &stmt_owned
3517 } else {
3518 stmt
3519 };
3520 if stmt_ref.unions.is_empty() {
3521 return self.exec_bare_select_cancel(stmt_ref, cancel);
3522 }
3523 self.exec_union_chain(stmt_ref, stmt, cancel)
3524 }
3525
3526 #[allow(clippy::too_many_lines)]
3527 #[allow(clippy::too_many_lines)] // huge match — splitting fragments the planner
3528 /// v7.11.7 — execute `SELECT … FROM unnest(expr) [AS] alias …`.
3529 /// Synthesises a single-column virtual table whose column type
3530 /// is TEXT and whose rows are the array elements. Routes
3531 /// through the regular projection / WHERE / ORDER BY / LIMIT
3532 /// machinery so set-returning UNNEST composes naturally with
3533 /// the rest of the SELECT surface.
3534 fn exec_select_unnest(
3535 &self,
3536 stmt: &SelectStatement,
3537 primary: &TableRef,
3538 cancel: CancelToken<'_>,
3539 ) -> Result<QueryResult, EngineError> {
3540 let expr = primary
3541 .unnest_expr
3542 .as_deref()
3543 .expect("caller guards unnest_expr.is_some()");
3544 // Multi-arg unnest(a, b, …) — parallel zip, NULL-padded.
3545 // N value columns instead of one; the shared builder does
3546 // the work and the tail below (WHERE / agg / projection)
3547 // runs against the wider schema.
3548 let multi: Option<(alloc::vec::Vec<DataType>, alloc::vec::Vec<Row<'static>>)> =
3549 match unnest_zip_args(expr) {
3550 Some(args) => Some(unnest_zip_rows(args)?),
3551 None => None,
3552 };
3553 // Evaluate the array expression once. Empty schema / empty
3554 // row — uncorrelated UNNEST cannot reference outer columns.
3555 // v7.39 (read01 round 49) — the ctx must carry the catalog: the enum
3556 // introspection family (enum_range / enum_first / enum_last) resolves
3557 // its labels from the argument's STATIC enum type against the
3558 // catalog's enum registry. Without it `unnest(enum_range(NULL::mood))`
3559 // fell through to the generic arm, got NULL, and expanded to zero rows
3560 // — while the bare `SELECT enum_range(NULL::mood)` (whose ctx does
3561 // carry the catalog) worked.
3562 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
3563 let ctx = EvalContext::new(&empty_schema, None).with_catalog(self.active_catalog());
3564 let dummy_row = Row::new(alloc::vec::Vec::new());
3565 // v7.11.13 — unnest dispatches per array element type so
3566 // INT[] / BIGINT[] surface their PG types in projection.
3567 // v7.39 (round 758, F31-B8a) — the composite SRF names its own
3568 // columns (PG: lexeme | positions | weights); everything else
3569 // keeps the alias / "unnest" defaults below.
3570 let mut composite_names: Option<&[&str]> = None;
3571 let (dtypes, rows): (alloc::vec::Vec<DataType>, alloc::vec::Vec<Row<'static>>) =
3572 if let Some(m) = multi {
3573 m
3574 } else {
3575 // v7.39 (round 236) — flatten a multidimensional array into
3576 // its row-major elements (PG) before the 1-D-only match.
3577 let unnest_src = {
3578 let v = eval::eval_expr(expr, &dummy_row, &ctx).map_err(EngineError::Eval)?;
3579 crate::eval::values::flatten_2d(&v).unwrap_or(v)
3580 };
3581 let mut return_multi: Option<(
3582 alloc::vec::Vec<DataType>,
3583 alloc::vec::Vec<Row<'static>>,
3584 )> = None;
3585 let (elem_dtype, rows): (DataType, alloc::vec::Vec<Row<'static>>) = match unnest_src
3586 {
3587 Value::Null => (DataType::Text, alloc::vec::Vec::new()),
3588 Value::TextArray(items) => {
3589 let rows = items
3590 .into_iter()
3591 .map(|item| {
3592 Row::new(alloc::vec![match item {
3593 Some(s) => Value::text(s),
3594 None => Value::Null,
3595 }])
3596 })
3597 .collect();
3598 (DataType::Text, rows)
3599 }
3600 Value::IntArray(items) => {
3601 let rows = items
3602 .into_iter()
3603 .map(|item| {
3604 Row::new(alloc::vec![match item {
3605 Some(n) => Value::Int(n),
3606 None => Value::Null,
3607 }])
3608 })
3609 .collect();
3610 (DataType::Int, rows)
3611 }
3612 Value::BigIntArray(items) => {
3613 let rows = items
3614 .into_iter()
3615 .map(|item| {
3616 Row::new(alloc::vec![match item {
3617 Some(n) => Value::BigInt(n),
3618 None => Value::Null,
3619 }])
3620 })
3621 .collect();
3622 (DataType::BigInt, rows)
3623 }
3624 Value::Multirange { kind, ranges } => {
3625 let rows = ranges
3626 .iter()
3627 .map(|sp| {
3628 Row::new(alloc::vec![Value::Range {
3629 kind,
3630 lower: sp.lower.clone(),
3631 upper: sp.upper.clone(),
3632 lower_inc: sp.lower_inc,
3633 upper_inc: sp.upper_inc,
3634 empty: false,
3635 }])
3636 })
3637 .collect();
3638 (DataType::Range(kind), rows)
3639 }
3640 // v7.39 (round 758, F31-B8a) — unnest(tsvector):
3641 // one row per lexeme, PG18-measured columns
3642 // lexeme | positions | weights (`a | {1,3} |
3643 // {D,D}`); a position-less lexeme (a stripped
3644 // vector) reads NULL in both array columns.
3645 Value::TsVector(lexemes) => {
3646 composite_names = Some(&["lexeme", "positions", "weights"]);
3647 let rows = lexemes
3648 .iter()
3649 .map(|l| {
3650 let (pos, wts) = if l.positions.is_empty() {
3651 (Value::Null, Value::Null)
3652 } else {
3653 let letter = match l.weight {
3654 3 => "A",
3655 2 => "B",
3656 1 => "C",
3657 _ => "D",
3658 };
3659 (
3660 Value::SmallIntArray(
3661 l.positions
3662 .iter()
3663 .map(|p| {
3664 Some(i16::try_from(*p).unwrap_or(i16::MAX))
3665 })
3666 .collect(),
3667 ),
3668 Value::TextArray(
3669 l.positions
3670 .iter()
3671 .map(|_| Some(letter.into()))
3672 .collect(),
3673 ),
3674 )
3675 };
3676 Row::new(alloc::vec![Value::text(l.word.clone()), pos, wts])
3677 })
3678 .collect();
3679 return_multi = Some((
3680 alloc::vec![
3681 DataType::Text,
3682 DataType::SmallIntArray,
3683 DataType::TextArray
3684 ],
3685 rows,
3686 ));
3687 (DataType::Text, alloc::vec::Vec::new())
3688 }
3689 other => {
3690 // v7.39 (round 622, S05a) — see table_access.rs:
3691 // the same sentence, and it is a type mismatch.
3692 return Err(EngineError::Eval(EvalError::TypeMismatch {
3693 detail: alloc::format!(
3694 "unnest() expects an array argument, got {}",
3695 crate::conversions::pg_type_name_for_error_opt(other.data_type())
3696 ),
3697 }));
3698 }
3699 };
3700 if let Some(m) = return_multi {
3701 m
3702 } else {
3703 (alloc::vec![elem_dtype], rows)
3704 }
3705 };
3706 let alias = primary
3707 .alias
3708 .clone()
3709 .unwrap_or_else(|| "unnest".to_string());
3710 // v7.13.2 — mailrs round-6 S5. Honour PG-standard
3711 // `UNNEST(arr) AS p(col_name)` column-list aliasing:
3712 // entries map positionally over the value columns. Without
3713 // the column list, a single column falls back to the table
3714 // alias (pre-v7.13.2 behaviour); multi-arg columns default
3715 // to PG's `unnest`.
3716 let n_vals = dtypes.len();
3717 let mut schema_cols: alloc::vec::Vec<ColumnSchema> = dtypes
3718 .iter()
3719 .enumerate()
3720 .map(|(i, dt)| {
3721 let name = primary
3722 .unnest_column_aliases
3723 .get(i)
3724 .cloned()
3725 .unwrap_or_else(|| {
3726 if let Some(names) = composite_names {
3727 names
3728 .get(i)
3729 .map_or_else(|| "unnest".to_string(), |n| (*n).to_string())
3730 } else if n_vals == 1 {
3731 alias.clone()
3732 } else {
3733 "unnest".to_string()
3734 }
3735 });
3736 ColumnSchema::new(name, *dt, true)
3737 })
3738 .collect();
3739 // v7.39 (read01 round 78) — the item's row type IS this scalar when the
3740 // parser desugared a base-type-returning function here (see
3741 // TableRef::scalar_fn_item); the marker rides the column so it survives
3742 // every EvalContext an inner stage rebuilds.
3743 if primary.scalar_fn_item && schema_cols.len() == 1 {
3744 schema_cols[0].scalar_row_source = true;
3745 }
3746 // WITH ORDINALITY — trailing BIGINT counting rows from 1
3747 // in element order. The alias entry after the value
3748 // columns renames it (PG default: `ordinality`).
3749 let rows = if primary.with_ordinality {
3750 let ord_name = primary
3751 .unnest_column_aliases
3752 .get(n_vals)
3753 .cloned()
3754 .unwrap_or_else(|| "ordinality".to_string());
3755 schema_cols.push(ColumnSchema::new(ord_name, DataType::BigInt, false));
3756 rows.into_iter()
3757 .enumerate()
3758 .map(|(i, row)| {
3759 let mut vals = row.values.clone();
3760 vals.push(Value::BigInt(i as i64 + 1));
3761 Row::new(vals)
3762 })
3763 .collect()
3764 } else {
3765 rows
3766 };
3767 // v7.39 (read01 round 54) — `ev_ctx` threads the catalog; a bare
3768 // `EvalContext::new` drops it and every catalog-dependent cast
3769 // (regclass / enum / composite / domain) silently degrades.
3770 let scan_ctx = self.ev_ctx(&schema_cols, Some(&alias));
3771 // Apply WHERE.
3772 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
3773 let mut out = alloc::vec::Vec::with_capacity(rows.len());
3774 for row in rows {
3775 cancel.check()?;
3776 let v = eval::eval_expr(w, &row, &scan_ctx).map_err(EngineError::Eval)?;
3777 if matches!(v, Value::Bool(true)) {
3778 out.push(row);
3779 }
3780 }
3781 out
3782 } else {
3783 rows
3784 };
3785 // v7.17.0 Phase 3.P0-48 — aggregate dispatch over the
3786 // unnest source. Same routing the relational scan path
3787 // already takes — without it `SELECT COUNT(*) FROM
3788 // unnest(ARRAY[…])` either errored at projection time or
3789 // returned the wrong shape.
3790 if aggregate::uses_aggregate(stmt) {
3791 // v7.29 — a per-query memo so correlated scalar
3792 // subqueries batch-evaluate once (group map) instead of
3793 // executing per group.
3794 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
3795 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
3796 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
3797 .map_err(|err| match err {
3798 EngineError::Eval(ev) => ev,
3799 other => eval::EvalError::TypeMismatch {
3800 detail: alloc::format!("{other}"),
3801 },
3802 })
3803 };
3804 // v7.39 (round 656) — hand the rows over as they are rather than
3805 // collecting a second vector of `RowRef` wrappers. Note this is
3806 // a set-returning-function path, NOT the relational scan: the
3807 // measured O(rows) cost lived in `run_single_table_aggregate`,
3808 // and converting these four first was a miss that cost a full
3809 // round — every test stayed green and the number did not move.
3810 let agg = aggregate::run(
3811 stmt,
3812 crate::join::AggRows::Owned(&filtered),
3813 &schema_cols,
3814 Some(&alias),
3815 Some(&agg_correlated),
3816 self.parallel_runner.0.as_deref(),
3817 Some(self.active_catalog()),
3818 Some(self),
3819 )?;
3820 return self.finish_agg_result(agg, stmt, cancel);
3821 }
3822 // Projection.
3823 let projection =
3824 build_projection(&stmt.items, &schema_cols, &alias, self.backslash_escapes)?;
3825 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
3826 alloc::vec::Vec::with_capacity(filtered.len());
3827 // v7.19 P5 — Set-Returning-Function in projection
3828 // position (PG `SELECT unnest(arr) FROM t` shape). When a
3829 // SELECT item evaluates to a top-level unnest(arr) call,
3830 // expand it: for each input row, evaluate the array, emit
3831 // one output row per element, broadcasting non-SRF
3832 // projections from the same input row. Multi-SRF + LCM
3833 // padding stays a documented carve-out; mailrs uses
3834 // single-SRF for redirect_uris.
3835 // v7.39 (read01 round 67) — EVERY set-returning item expands, in lockstep
3836 // (see `expand_srf_row`); a user `RETURNS SETOF` function counts too.
3837 let srf_idxs = self.srf_target_idxs(&projection);
3838 // v7.39 (round 621) — which input row each output row came from. An
3839 // SRF turns one input row into many, and the ORDER BY below used to
3840 // index the EXPANDED rows by the INPUT row's position: the result was
3841 // silently truncated to the input row count and left unsorted, so
3842 // `SELECT unnest(ARRAY[1,2]), y FROM unnest(ARRAY[5,6,7]) y ORDER BY 1`
3843 // answered three of its six rows, in no order. Without the ORDER BY
3844 // the same query was already right.
3845 let mut src_of_row: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
3846 if !srf_idxs.is_empty() {
3847 let (rows, src) =
3848 expand_projection_srfs(self, &projection, &srf_idxs, &filtered, &scan_ctx)?;
3849 projected_rows = rows;
3850 src_of_row = src;
3851 } else {
3852 // v7.24 (round-16 B) — select-list subqueries resolve
3853 // per row (correlated-aware; plain exprs take the fast
3854 // path inside).
3855 let mut proj_memo = memoize::MemoizeCache::default();
3856 for row in &filtered {
3857 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
3858 for p in &projection {
3859 vals.push(self.eval_expr_with_correlated(
3860 &p.expr,
3861 row,
3862 &scan_ctx,
3863 cancel,
3864 Some(&mut proj_memo),
3865 )?);
3866 }
3867 projected_rows.push(Row::new(vals));
3868 }
3869 }
3870 // ORDER BY / LIMIT — apply on the projected rows (cheap;
3871 // unnest result sets are small by design).
3872 let columns: alloc::vec::Vec<ColumnSchema> = projection
3873 .iter()
3874 // v7.39 (read01 round 54) — keep the column's enum identity through
3875 // the projection (it lives outside the DataType lattice), or a
3876 // derived table / UNION / windowed result forgets it and any outer
3877 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
3878 .map(|p| {
3879 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
3880 c.user_enum_type = p.user_enum_type.clone();
3881 c.mysql_fsp = p.mysql_fsp;
3882 c
3883 })
3884 .collect();
3885 // Re-evaluate ORDER BY against the source schema (pre-projection
3886 // so col refs by name still resolve through `scan_ctx`).
3887 // v7.39 (read01 round 80) — a positional key means the Nth OUTPUT
3888 // column. Evaluated as an expression it is just the constant N: the same
3889 // key for every row, so the sort ran and changed nothing.
3890 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
3891 if !order_by.is_empty() {
3892 // v7.39 (round 621) — one entry per OUTPUT row, not per input row.
3893 // A key that names a select-list item reads it out of the expanded
3894 // row (PG sorts AFTER the expansion); one that names a source
3895 // column the query does not project is evaluated on the input row
3896 // it came from, which is what `srf_order_output_cols` decides.
3897 let out_cols = if srf_idxs.is_empty() {
3898 alloc::vec![None; order_by.len()]
3899 } else {
3900 srf_order_output_cols(&order_by, &projection)
3901 };
3902 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = projected_rows
3903 .iter()
3904 .enumerate()
3905 .map(|(k, out)| -> Result<_, EngineError> {
3906 let src = src_of_row.get(k).copied().unwrap_or(k);
3907 let keys: Result<Vec<Value<'static>>, EngineError> = order_by
3908 .iter()
3909 .zip(out_cols.iter())
3910 .map(|(ob, oc)| srf_order_key(ob, *oc, out, &filtered[src], &scan_ctx))
3911 .collect();
3912 Ok((k, keys?))
3913 })
3914 .collect::<Result<_, _>>()?;
3915 indexed.sort_by(|a, b| {
3916 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
3917 let o = &order_by[idx];
3918 let cmp = order_by_value_cmp_in(
3919 o.desc,
3920 o.nulls_first,
3921 ka,
3922 kb,
3923 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
3924 );
3925 if cmp != core::cmp::Ordering::Equal {
3926 return cmp;
3927 }
3928 }
3929 core::cmp::Ordering::Equal
3930 });
3931 projected_rows = indexed
3932 .into_iter()
3933 .map(|(i, _)| projected_rows[i].clone())
3934 .collect();
3935 }
3936 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
3937 if stmt.distinct {
3938 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
3939 }
3940 // LIMIT / OFFSET — apply at the tail.
3941 if let Some(offset) = stmt.offset_literal() {
3942 let off = (offset as usize).min(projected_rows.len());
3943 projected_rows.drain(..off);
3944 }
3945 if let Some(limit) = stmt.limit_literal() {
3946 projected_rows.truncate(limit as usize);
3947 }
3948 Ok(QueryResult::Rows {
3949 columns,
3950 rows: projected_rows,
3951 })
3952 }
3953
3954 /// v7.17.0 Phase 3.10 — `FROM generate_series(start, stop [,
3955 /// step])` set-returning source. Mirrors `exec_select_unnest`'s
3956 /// shape: evaluate the arg list once against an empty row,
3957 /// materialise the row stream by stepping start → stop, then
3958 /// route through the standard WHERE / projection / ORDER BY /
3959 /// LIMIT pipeline. Two arg-type combos in v7.17:
3960 /// * integer / integer [/ integer] — SmallInt, Int, BigInt
3961 /// (widened to BigInt internally; step defaults to 1)
3962 /// * timestamp / timestamp / interval — date-range
3963 /// iteration (mailrs's daily-report pattern)
3964 fn exec_select_generate_series(
3965 &self,
3966 stmt: &SelectStatement,
3967 primary: &TableRef,
3968 cancel: CancelToken<'_>,
3969 ) -> Result<QueryResult, EngineError> {
3970 let args = primary
3971 .generate_series_args
3972 .as_ref()
3973 .expect("caller guards generate_series_args.is_some()");
3974 let (elem_dtype, rows) = generate_series_rows(args, &cancel)?;
3975 let alias = primary
3976 .alias
3977 .clone()
3978 .unwrap_or_else(|| "generate_series".to_string());
3979 // `AS t(n)` — the first column-alias entry renames the
3980 // series column (PG semantics); bare alias keeps the
3981 // pre-existing behaviour of naming the column after it.
3982 let col_name = primary
3983 .unnest_column_aliases
3984 .first()
3985 .cloned()
3986 .unwrap_or_else(|| alias.clone());
3987 let col_schema = ColumnSchema::new(col_name, elem_dtype, true);
3988 let mut schema_cols = alloc::vec![col_schema.clone()];
3989 // WITH ORDINALITY — trailing BIGINT counting rows from 1;
3990 // the second column-alias entry renames it.
3991 let rows = if primary.with_ordinality {
3992 let ord_name = primary
3993 .unnest_column_aliases
3994 .get(1)
3995 .cloned()
3996 .unwrap_or_else(|| "ordinality".to_string());
3997 schema_cols.push(ColumnSchema::new(ord_name, DataType::BigInt, false));
3998 rows.into_iter()
3999 .enumerate()
4000 .map(|(i, row)| {
4001 let mut vals = row.values.clone();
4002 vals.push(Value::BigInt(i as i64 + 1));
4003 Row::new(vals)
4004 })
4005 .collect()
4006 } else {
4007 rows
4008 };
4009 // v7.39 (read01 round 54) — `ev_ctx` threads the catalog; a bare
4010 // `EvalContext::new` drops it and every catalog-dependent cast
4011 // (regclass / enum / composite / domain) silently degrades.
4012 let scan_ctx = self.ev_ctx(&schema_cols, Some(&alias));
4013 // WHERE.
4014 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
4015 let mut out = alloc::vec::Vec::with_capacity(rows.len());
4016 for row in rows {
4017 cancel.check()?;
4018 let v = eval::eval_expr(w, &row, &scan_ctx).map_err(EngineError::Eval)?;
4019 if matches!(v, Value::Bool(true)) {
4020 out.push(row);
4021 }
4022 }
4023 out
4024 } else {
4025 rows
4026 };
4027 // v7.17.0 Phase 3.P0-48 — aggregate dispatch for set-
4028 // returning sources. When the SELECT projection contains
4029 // aggregate functions (COUNT/SUM/MIN/MAX/AVG/string_agg/
4030 // …) we route the filtered row stream through the same
4031 // aggregate executor the relational scan path uses, so
4032 // `SELECT COUNT(*) FROM generate_series(1, 100)` returns
4033 // a single 100 row instead of erroring at projection
4034 // time. GROUP BY / HAVING / ORDER BY over the aggregate
4035 // output all ride through `aggregate::run`.
4036 if aggregate::uses_aggregate(stmt) {
4037 // v7.29 — a per-query memo so correlated scalar
4038 // subqueries batch-evaluate once (group map) instead of
4039 // executing per group.
4040 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
4041 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
4042 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
4043 .map_err(|err| match err {
4044 EngineError::Eval(ev) => ev,
4045 other => eval::EvalError::TypeMismatch {
4046 detail: alloc::format!("{other}"),
4047 },
4048 })
4049 };
4050 // v7.39 (round 656) — hand the rows over as they are rather than
4051 // collecting a second vector of `RowRef` wrappers. Note this is
4052 // a set-returning-function path, NOT the relational scan: the
4053 // measured O(rows) cost lived in `run_single_table_aggregate`,
4054 // and converting these four first was a miss that cost a full
4055 // round — every test stayed green and the number did not move.
4056 let agg = aggregate::run(
4057 stmt,
4058 crate::join::AggRows::Owned(&filtered),
4059 &schema_cols,
4060 Some(&alias),
4061 Some(&agg_correlated),
4062 self.parallel_runner.0.as_deref(),
4063 Some(self.active_catalog()),
4064 Some(self),
4065 )?;
4066 return self.finish_agg_result(agg, stmt, cancel);
4067 }
4068 // Projection.
4069 let projection =
4070 build_projection(&stmt.items, &schema_cols, &alias, self.backslash_escapes)?;
4071 // v7.39 (round 621) — and here, for the same reason.
4072 let srf_idxs = self.srf_target_idxs(&projection);
4073 let mut src_of_row: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
4074 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
4075 alloc::vec::Vec::with_capacity(filtered.len());
4076 let mut proj_memo = memoize::MemoizeCache::default();
4077 if !srf_idxs.is_empty() {
4078 let (rows, src) =
4079 expand_projection_srfs(self, &projection, &srf_idxs, &filtered, &scan_ctx)?;
4080 projected_rows = rows;
4081 src_of_row = src;
4082 } else {
4083 for row in &filtered {
4084 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
4085 for p in &projection {
4086 // v7.24 (round-16 B) — correlated-aware.
4087 vals.push(self.eval_expr_with_correlated(
4088 &p.expr,
4089 row,
4090 &scan_ctx,
4091 cancel,
4092 Some(&mut proj_memo),
4093 )?);
4094 }
4095 projected_rows.push(Row::new(vals));
4096 }
4097 }
4098 let columns: alloc::vec::Vec<ColumnSchema> = projection
4099 .iter()
4100 // v7.39 (read01 round 54) — keep the column's enum identity through
4101 // the projection (it lives outside the DataType lattice), or a
4102 // derived table / UNION / windowed result forgets it and any outer
4103 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
4104 .map(|p| {
4105 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
4106 c.user_enum_type = p.user_enum_type.clone();
4107 c.mysql_fsp = p.mysql_fsp;
4108 c
4109 })
4110 .collect();
4111 // ORDER BY against the source schema.
4112 // v7.39 (round 621) — one entry per OUTPUT row (a target-list SRF makes
4113 // more of them than there were inputs), and a positional key means the
4114 // Nth OUTPUT column, which is what `resolve_positional_order_by` does
4115 // and what the other two synthetic-source tails already did.
4116 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
4117 if !order_by.is_empty() {
4118 let out_cols = if srf_idxs.is_empty() {
4119 alloc::vec![None; order_by.len()]
4120 } else {
4121 srf_order_output_cols(&order_by, &projection)
4122 };
4123 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = projected_rows
4124 .iter()
4125 .enumerate()
4126 .map(|(k, out)| -> Result<_, EngineError> {
4127 let r = &filtered[src_of_row.get(k).copied().unwrap_or(k)];
4128 let keys: Result<Vec<Value<'static>>, EngineError> = order_by
4129 .iter()
4130 .zip(out_cols.iter())
4131 .map(|(ob, oc)| srf_order_key(ob, *oc, out, r, &scan_ctx))
4132 .collect();
4133 Ok((k, keys?))
4134 })
4135 .collect::<Result<_, _>>()?;
4136 indexed.sort_by(|a, b| {
4137 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
4138 let o = &stmt.order_by[idx];
4139 let cmp = order_by_value_cmp_in(
4140 o.desc,
4141 o.nulls_first,
4142 ka,
4143 kb,
4144 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
4145 );
4146 if cmp != core::cmp::Ordering::Equal {
4147 return cmp;
4148 }
4149 }
4150 core::cmp::Ordering::Equal
4151 });
4152 projected_rows = indexed
4153 .into_iter()
4154 .map(|(i, _)| projected_rows[i].clone())
4155 .collect();
4156 }
4157 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
4158 if stmt.distinct {
4159 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
4160 }
4161 if let Some(offset) = stmt.offset_literal() {
4162 let off = (offset as usize).min(projected_rows.len());
4163 projected_rows.drain(..off);
4164 }
4165 if let Some(limit) = stmt.limit_literal() {
4166 projected_rows.truncate(limit as usize);
4167 }
4168 Ok(QueryResult::Rows {
4169 columns,
4170 rows: projected_rows,
4171 })
4172 }
4173
4174 /// The FROM shapes that are not an ordinary table scan — joins, the
4175 /// set-returning sources, JSON_TABLE, a derived table, and the rest.
4176 ///
4177 /// `#[inline(never)]` and out of `exec_bare_select_cancel` for the
4178 /// reason round 848 established in the parser: a debug build gives
4179 /// EVERY branch's locals a slot in the frame, whichever branch runs.
4180 /// `exec_bare_select_cancel` measured 64,784 bytes and a nested query
4181 /// stacks several of them; a plain scan reaches none of these
4182 /// branches. Moving them out took the frame to 52,336.
4183 ///
4184 /// `Ok(None)` means "not one of these shapes, carry on".
4185 #[inline(never)]
4186 fn try_from_shape_paths(
4187 &self,
4188 stmt: &SelectStatement,
4189 from: &spg_sql::ast::FromClause,
4190 cancel: CancelToken<'_>,
4191 ) -> Result<Option<QueryResult>, EngineError> {
4192 if !from.joins.is_empty() {
4193 // v7.37.x (docker-fair LEFTJOIN 71 % attack) — LEFT JOIN
4194 // elimination: when a LEFT JOIN's right side is referenced
4195 // ONLY in the ON equality and the right-side join key is
4196 // UNIQUE/PK, the join preserves outer cardinality exactly
4197 // and contributes no values used downstream. Drop the
4198 // entire join. PG does this on the
4199 // `SELECT COUNT(*) FROM A LEFT JOIN B ON B.pk = A.fk` shape
4200 // — A's row count is what survives, B never has to be
4201 // touched.
4202 if let Some(eliminated) = self.try_eliminate_redundant_left_joins(stmt) {
4203 return self.exec_bare_select_cancel(&eliminated, cancel).map(Some);
4204 }
4205 // v7.38 P0 元机制 D — `SPG_TEST_DISABLE_JOINFOLD=1` skips
4206 // the v7.32 joinfold rewrite that turns inner JOINs into a
4207 // single-table scan when the catalogue can prove key-only
4208 // dependency. Tests use this to assert "without joinfold,
4209 // the join still executes correctly" (joinfold is a
4210 // semantically-equivalent rewrite, not a correctness fix).
4211 if !self.env_cfg().disable_joinfold {
4212 if let Some(folded) = self.try_fold_inner_joins(stmt, cancel)? {
4213 return self.exec_bare_select_cancel(&folded, cancel).map(Some);
4214 }
4215 }
4216 return self.exec_joined_select(stmt, from, cancel).map(Some);
4217 }
4218 // v7.11.7 — `FROM unnest(<expr>) [AS] <alias>`. Synthesise a
4219 // single-column table at SELECT entry by evaluating the
4220 // expression once against the empty row (UNNEST is
4221 // uncorrelated in v7.11; correlated / LATERAL unnest is a
4222 // v7.12 carve-out). Build a virtual `Table` in a heap-only
4223 // catalog, then route to the regular scan path.
4224 if from.primary.unnest_expr.is_some() {
4225 return self
4226 .exec_select_unnest(stmt, &from.primary, cancel)
4227 .map(Some);
4228 }
4229 // v7.37.43-T4.5 — `FROM jsonb_each_text(<expr>)` set-
4230 // returning function. Same dispatch shape as unnest but
4231 // emits a two-column (key TEXT, value TEXT) row stream.
4232 if from.primary.jsonb_each_text_arg.is_some() {
4233 return self
4234 .exec_select_jsonb_each_text(stmt, &from.primary, cancel)
4235 .map(Some);
4236 }
4237 // v7.39 (read01 partitionfuncs.c) — FROM-position table functions
4238 // (pg_partition_tree / pg_partition_ancestors) dispatched by name.
4239 // v7.39 (read01 round 74) — `ROWS FROM (f(a), g(b))` whose entries have no
4240 // array form. Each function runs; the results zip in LOCKSTEP with the
4241 // shorter padded to NULL — the SAME rule the target-list SRFs follow
4242 // (round 67), which is why `srf_values` is what evaluates each entry.
4243 if from.primary.rows_from.is_some() {
4244 let (rows, mut schema_cols) = self.rows_from_rows(&from.primary)?;
4245 for (i, new_name) in from.primary.unnest_column_aliases.iter().enumerate() {
4246 if let Some(col) = schema_cols.get_mut(i) {
4247 col.name = new_name.clone();
4248 }
4249 }
4250 let alias = from
4251 .primary
4252 .alias
4253 .clone()
4254 .unwrap_or_else(|| from.primary.name.clone());
4255 return self
4256 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4257 .map(Some);
4258 }
4259 // v7.39 (round 205, JSON_TABLE) — `FROM JSON_TABLE(doc, '$p'
4260 // COLUMNS (...))`. Materialise the row stream + schema by
4261 // walking the row path, then run the regular pipeline over it.
4262 if let Some(jt) = &from.primary.json_table {
4263 let (rows, schema_cols) = self.json_table_rows(jt, None)?;
4264 let alias = from
4265 .primary
4266 .alias
4267 .clone()
4268 .unwrap_or_else(|| from.primary.name.clone());
4269 return self
4270 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4271 .map(Some);
4272 }
4273 if from.primary.table_fn_call.is_some() {
4274 let (rows, mut schema_cols) = self.table_fn_rows(&from.primary)?;
4275 // v7.39 (read01 round 68) — WITH ORDINALITY appends a BIGINT counter
4276 // (from 1, in output order) AFTER the function's own columns. The
4277 // alias list names it like any other, which is why it is appended
4278 // BEFORE the renaming pass below.
4279 let rows = if from.primary.with_ordinality {
4280 schema_cols.push(ColumnSchema::new(
4281 "ordinality".to_string(),
4282 DataType::BigInt,
4283 false,
4284 ));
4285 rows.into_iter()
4286 .enumerate()
4287 .map(|(i, r)| {
4288 let mut vals = r.values;
4289 vals.push(Value::BigInt(i as i64 + 1));
4290 Row::new(vals)
4291 })
4292 .collect()
4293 } else {
4294 rows
4295 };
4296 for (i, new_name) in from.primary.unnest_column_aliases.iter().enumerate() {
4297 if let Some(col) = schema_cols.get_mut(i) {
4298 col.name = new_name.clone();
4299 }
4300 }
4301 let alias = from
4302 .primary
4303 .alias
4304 .clone()
4305 .unwrap_or_else(|| from.primary.name.clone());
4306 return self
4307 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4308 .map(Some);
4309 }
4310 // v7.37.17 (17.6 siblings) — plain derived table in primary
4311 // position: `FROM ( SELECT … ) alias` (no joins). The inner
4312 // SELECT materialises once (it is uncorrelated by
4313 // construction), then the outer projection / WHERE /
4314 // aggregate / ORDER BY pipeline runs over the synthetic
4315 // table. Joined derived tables keep riding the LATERAL
4316 // machinery in join.rs.
4317 if from.joins.is_empty() && from.primary.lateral_subquery.is_some() {
4318 // v7.39 (round 727) — flatten first. A simple derived table
4319 // (bare-column projection over one stored table, nothing that
4320 // changes cardinality or order) used to force the inner
4321 // SELECT through the SERIAL row-at-a-time projection pipeline
4322 // just to materialise a synthetic table the outer query then
4323 // re-scans: `count(*) FROM (SELECT id v FROM d WHERE …) q`
4324 // measured 18.6 ms against PG's 5 — and bare count over the
4325 // same filter WITHOUT the wrapper is 2 ms here, because it
4326 // rides the fused parallel lane. Rewriting to the unwrapped
4327 // form is PG's subquery pull-up; the whole tree gets the
4328 // fast lanes back.
4329 if let Some(flat) = try_flatten_derived(stmt, &from.primary) {
4330 return self.exec_select_cancel(&flat, cancel).map(Some);
4331 }
4332 // v7.39 (round 742) — `SELECT count(*) FROM (SELECT … ORDER
4333 // BY … OFFSET k) q` is `greatest(count_of_inner - k, 0)`:
4334 // ORDER BY never changes the row count, and OFFSET drops
4335 // exactly k. The materialising path sorted 500k rows to
4336 // count 10k (57 ms); PG runs its parallel sort anyway
4337 // (28 ms). The rewrite skips the sort entirely on both
4338 // counts — a plan PG itself does not have.
4339 if let Some(rewritten) = try_count_over_offset(stmt, &from.primary) {
4340 return self.exec_select_cancel(&rewritten, cancel).map(Some);
4341 }
4342 // v7.39 (round 743) — `count(*) OVER a derived whose only
4343 // item is unnest(ARRAY[k elements])` is `k * count(WHERE)`:
4344 // a constant-length array unnests to exactly k rows per
4345 // input row, NULL elements included. PG expands the set to
4346 // count it (6.6 ms on the panel cell); the identity doesn't.
4347 if let Some(rewritten) = try_count_over_const_unnest(stmt, &from.primary) {
4348 return self.exec_select_cancel(&rewritten, cancel).map(Some);
4349 }
4350 return self
4351 .exec_select_derived(stmt, &from.primary, cancel)
4352 .map(Some);
4353 }
4354 // v7.17.0 Phase 3.10 — `FROM generate_series(start, stop
4355 // [, step])` set-returning source. Dispatch mirrors UNNEST:
4356 // materialise the row stream from a single eval pass, then
4357 // run the regular projection / WHERE / ORDER BY / LIMIT
4358 // pipeline over the synthetic single-column table.
4359 if from.primary.generate_series_args.is_some() {
4360 return self
4361 .exec_select_generate_series(stmt, &from.primary, cancel)
4362 .map(Some);
4363 }
4364 Ok(None)
4365 }
4366
4367 /// Pick an index seek for this WHERE, if any of the four apply:
4368 /// BTree equality, GIN `@@`, trigram LIKE, or JSONB `@>`.
4369 ///
4370 /// `#[inline(never)]` and out of `exec_bare_select_cancel` for the
4371 /// frame reason on `try_from_shape_paths`: in a debug build a
4372 /// closure's locals belong to the enclosing frame, and this one is
4373 /// four seek attempts wide on a function that nests.
4374 #[inline(never)]
4375 fn pick_indexed_rows<'r>(
4376 &'r self,
4377 stmt: &SelectStatement,
4378 table: &'r spg_storage::Table,
4379 schema_cols: &[spg_storage::ColumnSchema],
4380 alias: &str,
4381 ctx: &crate::eval::EvalContext<'_>,
4382 seek_snapshot: &crate::Snapshot,
4383 ) -> Option<Vec<Cow<'r, Row<'static>>>> {
4384 stmt.where_.as_ref().and_then(|w| {
4385 // BTree / col=literal seek first — covers the v7.11.3 multi-
4386 // column AND case and the leading-column equality lookup.
4387 try_index_seek(
4388 w,
4389 schema_cols,
4390 self.active_catalog(),
4391 table,
4392 alias,
4393 seek_snapshot,
4394 )
4395 .or_else(|| {
4396 // v7.12.3 — GIN-accelerated `WHERE col @@
4397 // tsquery` when the column has a `USING gin`
4398 // index. Returns an over-approximate candidate
4399 // set; the WHERE re-eval loop below verifies
4400 // the full `@@` predicate per row.
4401 try_gin_seek(
4402 w,
4403 schema_cols,
4404 self.active_catalog(),
4405 table,
4406 alias,
4407 ctx,
4408 seek_snapshot,
4409 )
4410 })
4411 .or_else(|| {
4412 // v7.15.0 — trigram-GIN-accelerated
4413 // `WHERE col LIKE / ILIKE '<pat>'` when the
4414 // column has a `gin_trgm_ops` GIN index.
4415 // Over-approximate candidate set; the WHERE
4416 // re-eval verifies the LIKE per row.
4417 try_trgm_seek(w, schema_cols, table, alias, seek_snapshot)
4418 })
4419 .or_else(|| {
4420 // v7.37.8(sentori Epic 5 P2)— real JSONB-GIN
4421 // accelerated `WHERE col @> <jsonb_literal>`
4422 // when the column has a `USING gin` index. The
4423 // posting-list intersection returns an over-
4424 // approximate candidate set; the WHERE re-eval
4425 // verifies the full `@>` predicate per row.
4426 try_gin_jsonb_seek(w, schema_cols, table, alias, seek_snapshot)
4427 })
4428 })
4429 }
4430
4431 /// Index-seek fast paths: NSW kNN, the primary-key top-N walk, and
4432 /// the two `count(*)` short-circuits. Out-of-line for the frame
4433 /// reason on `try_from_shape_paths` — an ordinary scan reaches none
4434 /// of them, and in a debug build their locals sit in the frame
4435 /// regardless.
4436 #[inline(never)]
4437 fn try_seek_fast_paths(
4438 &self,
4439 stmt: &SelectStatement,
4440 table: &spg_storage::Table,
4441 schema_cols: &[spg_storage::ColumnSchema],
4442 alias: &str,
4443 seek_snapshot: &crate::Snapshot,
4444 cancel: CancelToken<'_>,
4445 ) -> Result<Option<QueryResult>, EngineError> {
4446 if let Some(nsw_rows) = try_nsw_knn(stmt, table, schema_cols, alias, seek_snapshot) {
4447 // NSW kNN dispatches against the hot-tier vector index only
4448 // (vector cells aren't promoted to cold segments), so wrap
4449 // the returned row indices as `Cow::Borrowed` for the
4450 // unified `materialise_in_order` shape.
4451 let ordered: Vec<Cow<'_, Row<'static>>> = nsw_rows
4452 .into_iter()
4453 .filter_map(|i| table.rows().get(i).map(Cow::Borrowed))
4454 .collect();
4455 return materialise_in_order(
4456 stmt,
4457 schema_cols,
4458 alias,
4459 &ordered,
4460 self.backslash_escapes,
4461 )
4462 .map(Some);
4463 }
4464
4465 // v7.34.5 — ORDER BY <indexed col> [DESC|ASC] LIMIT N drives
4466 // the scan via the BTree iterator in the requested direction
4467 // and stops after `OFFSET + LIMIT` candidates pass WHERE. The
4468 // 80 ms `mailrs_prod_plain_limit` baseline at 250 k rows is
4469 // the load-bearing consumer; this skips the materialise-every-
4470 // row + partial-sort tail entirely. Walker output is already
4471 // in ORDER BY order so `materialise_in_order` (no extra sort)
4472 // is the natural sink.
4473 if let Some(walked) = try_pk_walk_top_n(
4474 stmt,
4475 self.active_catalog(),
4476 table,
4477 schema_cols,
4478 alias,
4479 self,
4480 cancel,
4481 ) {
4482 return materialise_in_order(stmt, schema_cols, alias, &walked, self.backslash_escapes)
4483 .map(Some);
4484 }
4485
4486 // Index seek: if WHERE is `col = literal` (or commuted) and the
4487 // referenced column has an index, dispatch each locator through
4488 // the catalog (hot tier → borrow, cold tier → page-read +
4489 // decode) and iterate just those rows. Otherwise fall back to a
4490 // v7.37.x (docker-fair INSUBQ attack) — short-circuit COUNT(*)
4491 // FROM A WHERE A.pk IN (large literal list). The post-subquery-
4492 // replacement shape of INSUBQ. Runs BEFORE `indexed_rows` so
4493 // we don't pay the row materialisation cost twice. Returns
4494 // a bare `Rows{count}` if the shape matches.
4495 if aggregate::uses_aggregate(stmt)
4496 && let Some(out) = self.try_count_star_pk_in_list_fast(stmt, table, schema_cols, alias)
4497 {
4498 return Ok(Some(out));
4499 }
4500 // v7.38 (perf) — `count(*) WHERE <indexed BETWEEN>`: count the in-range
4501 // locators directly, skipping row materialisation + WHERE re-eval.
4502 if aggregate::uses_aggregate(stmt)
4503 && let Some(out) = self.try_count_star_indexed_range_fast(
4504 stmt,
4505 table,
4506 schema_cols,
4507 alias,
4508 seek_snapshot,
4509 )
4510 {
4511 return Ok(Some(out));
4512 }
4513 Ok(None)
4514 }
4515
4516 /// The two rewrites that must happen before the FROM clause is even
4517 /// looked at: a meta-view reference needs the catalog views
4518 /// materialised, and a windowed projection belongs to the window
4519 /// executor. Out-of-line for the frame reason on
4520 /// `try_from_shape_paths`.
4521 #[inline(never)]
4522 fn try_pre_from_paths(
4523 &self,
4524 stmt: &SelectStatement,
4525 cancel: CancelToken<'_>,
4526 ) -> Result<Option<QueryResult>, EngineError> {
4527 if !self.meta_views_materialised && select_references_meta_view(stmt) {
4528 return self.exec_select_with_meta_views(stmt, cancel).map(Some);
4529 }
4530 // v4.12: window-function path. When the projection contains
4531 // any `name(args) OVER (...)` we route to the dedicated
4532 // executor — partition + sort + per-row window value before
4533 // the regular projection.
4534 if select_has_window(stmt) {
4535 // v7.37 D.23 — window functions run AFTER GROUP BY aggregation.
4536 // `SELECT g, sum(v), rank() OVER (ORDER BY sum(v)) FROM t GROUP BY g`
4537 // needs the aggregation done first, then windows over the grouped
4538 // rows. Rewrite to an aggregate derived subquery + outer window query
4539 // (which the window-over-derived path, D.13, executes). Only fires on
4540 // the currently-erroring agg+window+GROUP BY shape, so it can't
4541 // regress working window-only or aggregate-only queries.
4542 if let Some(rewritten) = rewrite_agg_before_window(stmt) {
4543 return self.exec_select_cancel(&rewritten, cancel).map(Some);
4544 }
4545 return self.exec_select_with_window(stmt, cancel).map(Some);
4546 }
4547 Ok(None)
4548 }
4549
4550 /// A projection naming `ctid` or another system column: the schema
4551 /// has to be widened with them before the scan. Out-of-line for the
4552 /// frame reason on `try_from_shape_paths`.
4553 #[inline(never)]
4554 fn try_ctid_projection(
4555 &self,
4556 stmt: &SelectStatement,
4557 primary: &spg_sql::ast::TableRef,
4558 table: &spg_storage::Table,
4559 schema_cols: &[spg_storage::ColumnSchema],
4560 alias: &str,
4561 cancel: CancelToken<'_>,
4562 ) -> Result<Option<QueryResult>, EngineError> {
4563 if references_ctid(stmt) {
4564 let snapshot = self.current_snapshot();
4565 let mut ext_cols = schema_cols.to_vec();
4566 for name in SYSTEM_COLUMNS {
4567 ext_cols.push(ColumnSchema::new(name.to_string(), DataType::Text, false));
4568 }
4569 let table_oid =
4570 crate::system_catalog::relation_oid(self.active_catalog(), &primary.name)
4571 .unwrap_or(0);
4572 let headers = table.headers();
4573 let rows: Vec<Row<'static>> = table
4574 .scan_visible(&snapshot)
4575 .map(|(i, r)| {
4576 let mut vals = r.values.clone();
4577 // One block, offsets from 1, as PG numbers them.
4578 vals.push(Value::Tid(0, i as u32 + 1));
4579 let h = headers.get(i);
4580 vals.push(Value::Xid(h.map_or(0, |h| h.xmin as u32)));
4581 vals.push(Value::Xid(h.map_or(0, |h| h.xmax as u32)));
4582 // SPG keeps no per-statement command ids; PG shows 0 for
4583 // every row a reader can see, which is every row here.
4584 vals.push(Value::Cid(0));
4585 vals.push(Value::Cid(0));
4586 vals.push(Value::BigInt(table_oid));
4587 Row::new(vals)
4588 })
4589 .collect();
4590 return self
4591 .exec_select_over_rows(stmt, rows, ext_cols, alias, cancel)
4592 .map(Some);
4593 }
4594 Ok(None)
4595 }
4596
4597 /// A sequence read as a one-row relation (`SELECT last_value FROM
4598 /// seq`), which PG allows and psql's \\d relies on. Out-of-line for
4599 /// the frame reason on `try_from_shape_paths`.
4600 #[inline(never)]
4601 fn try_sequence_relation(
4602 &self,
4603 stmt: &SelectStatement,
4604 primary: &spg_sql::ast::TableRef,
4605 cancel: CancelToken<'_>,
4606 ) -> Result<Option<QueryResult>, EngineError> {
4607 if self.active_catalog().get(&primary.name).is_none()
4608 && let Some(seq) = self.active_catalog().sequence(&primary.name)
4609 {
4610 let rows = alloc::vec![Row::new(alloc::vec![
4611 Value::BigInt(seq.last_value),
4612 Value::BigInt(0),
4613 Value::Bool(seq.is_called),
4614 ])];
4615 let schema_cols = alloc::vec![
4616 ColumnSchema::new("last_value", DataType::BigInt, false),
4617 ColumnSchema::new("log_cnt", DataType::BigInt, false),
4618 ColumnSchema::new("is_called", DataType::Bool, false),
4619 ];
4620 let alias = primary
4621 .alias
4622 .clone()
4623 .unwrap_or_else(|| primary.name.clone());
4624 return self
4625 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4626 .map(Some);
4627 }
4628 Ok(None)
4629 }
4630
4631 pub(crate) fn exec_bare_select_cancel(
4632 &self,
4633 stmt: &SelectStatement,
4634 cancel: CancelToken<'_>,
4635 ) -> Result<QueryResult, EngineError> {
4636 // v7.17.0 Phase 3.P0-49 — `FETCH FIRST N ROWS WITH TIES`
4637 // is meaningless without an ORDER BY; PG raises a hard
4638 // error and SPG mirrors the surface so the same DDL/app
4639 // path behaves identically on cutover.
4640 check_with_ties_requires_order_by(stmt)?;
4641 // v7.39 (round 229) — WHERE / HAVING run before the window pass, so
4642 // PG rejects window calls there outright. Checked here rather than
4643 // on the window path: `HAVING row_number() OVER () = 1` has no
4644 // window in its projection at all.
4645 crate::window::reject_window_in_row_clauses(stmt)?;
4646 // v7.39 (round 232) — the ORDER BY legality rules (positional
4647 // bounds, DISTINCT, DISTINCT ON). Same placement as the window
4648 // check: before anything scans.
4649 crate::orderby::check_order_by_legality(stmt)?;
4650 // v7.37.16 — resolve `USING` column-merge + `NATURAL JOIN` into an
4651 // equivalent statement the regular executor handles (merged join
4652 // columns collapse to a single unqualified output column; NATURAL
4653 // gets its common-column ON synthesised). The rewrite clears the
4654 // flags, so this re-entrant call is a no-op on the second pass.
4655 if let Some(rewritten) = self.desugar_using_natural(stmt)? {
4656 return self.exec_bare_select_cancel(&rewritten, cancel);
4657 }
4658 // v7.39 (RLS) Phase 3 — cross-table joins: wrap each RLS-enabled join
4659 // operand in a security-barrier subquery, then re-enter (the wrapped
4660 // operands are no longer bare RLS tables, so this is a no-op on the
4661 // second pass).
4662 if let Some(rewritten) = self.rls_rewrite_joins(stmt) {
4663 return self.exec_bare_select_cancel(&rewritten, cancel);
4664 }
4665 // v7.39 (RLS) Phase 1 — for a policy-subject (non-superuser) session,
4666 // AND the RLS USING predicate into a single-table SELECT's WHERE.
4667 // Superuser sessions and non-RLS tables get `None` (no clone, no
4668 // change). Applied inline (shadowing `stmt`) rather than via re-entry
4669 // so it can't re-inject on a recursive pass.
4670 let rls_stmt;
4671 let stmt = match self.rls_select_predicate(stmt)? {
4672 Some(pred) => {
4673 let mut s = stmt.clone();
4674 s.where_ = Some(match s.where_.take() {
4675 Some(existing) => spg_sql::ast::Expr::Binary {
4676 lhs: alloc::boxed::Box::new(existing),
4677 op: spg_sql::ast::BinOp::And,
4678 rhs: alloc::boxed::Box::new(pred),
4679 },
4680 None => pred,
4681 });
4682 rls_stmt = s;
4683 &rls_stmt
4684 }
4685 None => stmt,
4686 };
4687 // v7.16.2 — same meta-view dispatch as
4688 // `exec_select_cancel`, applied here too because
4689 // `subquery_replacement` enters this function directly
4690 // for Exists / ScalarSubquery / InSubquery resolution
4691 // (bypassing the top-level entry to avoid double
4692 // subquery walking). Without this dispatch the subquery
4693 // hits `__spg_info_columns` and reports TableNotFound.
4694 if let Some(done) = self.try_pre_from_paths(stmt, cancel)? {
4695 return Ok(done);
4696 }
4697 // Constant SELECT (no FROM) — evaluate each item once against an
4698 // empty dummy row. Useful for `SELECT 1`, `SELECT coalesce(...)`,
4699 // `SELECT '7'::INT`. Column references will surface as
4700 // ColumnNotFound on eval since the schema is empty.
4701 let Some(from) = &stmt.from else {
4702 return self.exec_constant_select(stmt);
4703 };
4704 // Multi-table FROM (one or more joined peers) goes through the
4705 // nested-loop join executor. Single-table FROM stays on the
4706 // existing scan + index-seek path.
4707 if let Some(done) = self.try_from_shape_paths(stmt, from, cancel)? {
4708 return Ok(done);
4709 }
4710 // NOT hooked up. `try_spill_sorted_scan` is written, correct and
4711 // tested — eight ORDER BY shapes byte-identical spilled against
4712 // in-memory, with 103 runs opened to prove the spill ran — and it
4713 // loses on wall clock, which is a hard stop whatever the memory
4714 // buys. Measured round 865, same psql client both sides, same
4715 // machine, row counts verified, and both sides confirmed to be
4716 // doing an external merge rather than an indexed walk:
4717 //
4718 // PG18 178.7 - 187.0 ms Sort Method: external merge, 85 MB
4719 // SPG spilled 269.7 - 299.6 ms 33 spill files at peak
4720 //
4721 // Non-overlapping, about 1.55x. Re-enable by restoring the call
4722 // below once that closes; nothing else has to change, which is
4723 // the point of it being a separate path.
4724 //
4725 // if let Some(done) = self.try_spill_sorted_scan(stmt, from, cancel)? {
4726 // return Ok(done);
4727 // }
4728 //
4729 // v7.37 (round 882) — this walk stays unhooked, but its streaming
4730 // twin `try_spill_sorted_stream` IS hooked, above the ORDER BY
4731 // bail in `try_exec_joined_streaming`. Collecting the answer was
4732 // most of what this one cost: handing rows over as the merge
4733 // produces them holds peak to the budget plus one row, and the
4734 // wall clock lands inside PG18's range rather than 1.55x outside
4735 // it. Numbers in `extsort.rs`'s header.
4736 let primary = &from.primary;
4737 // v7.39 (round 244) — a sequence is selectable as a one-row relation
4738 // in PG (`SELECT last_value FROM seq` — psql's \d and several ORMs
4739 // read it). Synthesize PG's three columns.
4740 if let Some(done) = self.try_sequence_relation(stmt, primary, cancel)? {
4741 return Ok(done);
4742 }
4743 let table = self.active_catalog().get(&primary.name).ok_or_else(|| {
4744 StorageError::TableNotFound {
4745 name: primary.name.clone(),
4746 }
4747 })?;
4748 let schema_cols = &table.schema().columns;
4749 // The qualifier accepted on column refs is the alias (if any) else the
4750 // bare table name.
4751 let alias = primary.alias.as_deref().unwrap_or(primary.name.as_str());
4752 // v7.39 (round 511) — `ctid`, PG's physical row identity. SPG had no
4753 // system columns at all: `SELECT ctid FROM t` answered "column
4754 // \"ctid\" does not exist", which takes out the dedup idiom every
4755 // PG user knows — `DELETE … WHERE ctid NOT IN (SELECT min(ctid) …
4756 // GROUP BY key)`.
4757 //
4758 // The value comes from the row's position, which the scan already
4759 // yields; the column is appended to the schema and the rows only
4760 // when the statement asks for it, so nothing else pays for it. That
4761 // also routes the query down the general path, past the index fast
4762 // paths below — they hand back rows without positions, and a ctid
4763 // that was sometimes right would be worse than none.
4764 if let Some(done) =
4765 self.try_ctid_projection(stmt, primary, table, schema_cols, alias, cancel)?
4766 {
4767 return Ok(done);
4768 }
4769 let ctx = self.ev_ctx(schema_cols, Some(alias));
4770
4771 // NSW kNN planner: `ORDER BY col <-> literal LIMIT k` with no
4772 // WHERE and an NSW index on `col` skips the full scan. The
4773 // walk returns rows already in ascending-distance order, so
4774 // ORDER BY / LIMIT are honoured implicitly.
4775 // Phase C.3 step 2c — compute the reader's MVCC snapshot once
4776 // and thread it into every index-seek fast path below. No-op
4777 // today (every hot header is committed-alive).
4778 let seek_snapshot = self.current_snapshot();
4779 if let Some(done) =
4780 self.try_seek_fast_paths(stmt, table, schema_cols, alias, &seek_snapshot, cancel)?
4781 {
4782 return Ok(done);
4783 }
4784 // full scan over the hot tier (cold-tier rows are only reached
4785 // via index seek in v5.1 — full table scans against cold-tier
4786 // data ship in v5.2 with the freezer's per-segment scan API).
4787 let indexed_rows =
4788 self.pick_indexed_rows(stmt, table, schema_cols, alias, &ctx, &seek_snapshot);
4789
4790 // Aggregate path: filter rows first, then hand off to the
4791 // aggregate executor which does its own projection + ORDER BY.
4792 if aggregate::uses_aggregate(stmt) {
4793 return self.run_single_table_aggregate(
4794 stmt,
4795 table,
4796 schema_cols,
4797 alias,
4798 indexed_rows,
4799 cancel,
4800 );
4801 }
4802 self.run_single_table_scan(stmt, table, schema_cols, alias, indexed_rows, cancel)
4803 }
4804
4805 /// v7.37.43-T4.5 — execute `SELECT … FROM jsonb_each_text(<expr>)`.
4806 /// Sentori migration 0067 uses this with `CROSS JOIN LATERAL`; the
4807 /// uncorrelated FROM-primary case is the simpler shape, used by
4808 /// e2e pins. Materialises the (key, value) pair stream into a
4809 /// synthetic two-column TEXT table, then routes through the
4810 /// regular projection / WHERE / ORDER BY pipeline.
4811 /// v7.39 (read01 partitionfuncs.c) — materialise a FROM-position
4812 /// v7.39 (round 205, JSON_TABLE) — materialise a JSON_TABLE FROM
4813 /// item into (rows, schema). `outer_doc` is `Some` only when this
4814 /// is a NESTED level being expanded against a parent row item's
4815 /// already-parsed sub-document; the top-level call parses the doc
4816 /// expr itself. Row/column paths reuse the existing jsonpath
4817 /// evaluator (`json::json_table_path`); coercion reuses
4818 /// `coerce_value` on the JSON scalar text, so a json string
4819 /// coerces to DATE by its content, matching PG.
4820 #[allow(clippy::type_complexity)]
4821 pub(crate) fn json_table_rows(
4822 &self,
4823 jt: &spg_sql::ast::JsonTable,
4824 outer_doc: Option<&crate::json::JsonValue>,
4825 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
4826 // Column schema is static (independent of data): flatten the
4827 // COLUMNS tree in declaration order (NESTED contributes its
4828 // children inline, the PG output shape).
4829 let schema = json_table_schema(&jt.columns);
4830
4831 // PASSING variables → a single JsonValue object the jsonpath
4832 // engine reads `$name` from.
4833 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
4834 let ctx = EvalContext::new(&empty_schema, None);
4835 let dummy = Row::new(alloc::vec::Vec::new());
4836 let vars: Option<crate::json::JsonValue> = if jt.passing.is_empty() {
4837 None
4838 } else {
4839 let mut entries = alloc::vec::Vec::new();
4840 for (name, e) in &jt.passing {
4841 let v = eval::eval_expr(e, &dummy, &ctx).map_err(EngineError::Eval)?;
4842 entries.push((name.clone(), value_to_json_value(&v)));
4843 }
4844 Some(crate::json::JsonValue::Object(entries))
4845 };
4846
4847 // The document root: a NESTED level gets it from the parent;
4848 // the top level parses its doc expr.
4849 let root_owned;
4850 let root: &crate::json::JsonValue = match outer_doc {
4851 Some(d) => d,
4852 None => {
4853 let doc_val = eval::eval_expr(&jt.doc, &dummy, &ctx).map_err(EngineError::Eval)?;
4854 let src = match &doc_val {
4855 Value::Null => return Ok((alloc::vec::Vec::new(), schema)),
4856 Value::Json(s) | Value::Text(s) => s.as_ref().to_string(),
4857 other => {
4858 return Err(EngineError::Unsupported(alloc::format!(
4859 "JSON_TABLE document must be json/text, got {}",
4860 crate::conversions::pg_type_name_for_error_opt(other.data_type())
4861 )));
4862 }
4863 };
4864 root_owned = crate::json::parse_doc(&src).map_err(EngineError::Eval)?;
4865 &root_owned
4866 }
4867 };
4868
4869 let items = crate::json::json_table_path(root, &jt.row_path, vars.as_ref())
4870 .map_err(EngineError::Eval)?;
4871 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
4872 for (idx, item) in items.iter().enumerate() {
4873 self.json_table_emit_item(jt, item, idx, vars.as_ref(), &mut rows)?;
4874 }
4875 Ok((rows, schema))
4876 }
4877
4878 /// v7.39 (round 205) — emit the row(s) for one row-pattern item.
4879 /// Regular columns produce one value each; a NESTED column expands
4880 /// as an outer join (each nested match → one row sharing the
4881 /// parent cells; no nested match → one row with the nested cells
4882 /// NULL). Sibling NESTED at one level cross by concatenation of
4883 /// their independent expansions (PG's UNION-of-outer shape).
4884 fn json_table_emit_item(
4885 &self,
4886 jt: &spg_sql::ast::JsonTable,
4887 item: &crate::json::JsonValue,
4888 ordinality: usize,
4889 vars: Option<&crate::json::JsonValue>,
4890 out: &mut alloc::vec::Vec<Row<'static>>,
4891 ) -> Result<(), EngineError> {
4892 use spg_sql::ast::JsonTableColumn as C;
4893 // Parent cells (regular + ordinality), left-to-right; NESTED
4894 // columns contribute a run of child cells appended after.
4895 let mut parent_cells: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
4896 let mut nested_runs: alloc::vec::Vec<alloc::vec::Vec<Row<'static>>> =
4897 alloc::vec::Vec::new();
4898 let mut nested_widths: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
4899 for col in &jt.columns {
4900 match col {
4901 C::Ordinality { .. } => {
4902 parent_cells.push(Value::BigInt(ordinality as i64 + 1));
4903 }
4904 C::Regular { .. } => {
4905 parent_cells.push(self.json_table_column_value(col, item, vars)?);
4906 }
4907 C::Nested { path, columns } => {
4908 // Recurse: a nested JSON_TABLE over `item` filtered
4909 // by `path`, with the same PASSING vars.
4910 let sub = spg_sql::ast::JsonTable {
4911 doc: jt.doc.clone(), // unused (outer_doc provided)
4912 row_path: path.clone(),
4913 columns: columns.clone(),
4914 passing: alloc::vec::Vec::new(),
4915 };
4916 let (nrows, nschema) = self.json_table_rows(&sub, Some(item))?;
4917 nested_widths.push(nschema.len());
4918 nested_runs.push(nrows);
4919 }
4920 }
4921 }
4922 if nested_runs.is_empty() {
4923 out.push(Row::new(parent_cells));
4924 return Ok(());
4925 }
4926 // PG sibling-NESTED semantics: each sibling expands
4927 // INDEPENDENTLY and the results CONCATENATE — a row from
4928 // sibling s fills only s's cells, every other sibling's cells
4929 // NULL. An empty sibling contributes ZERO rows (not a NULL
4930 // row). Only when EVERY sibling is empty does the parent still
4931 // emit one all-NULL row (the outer-join guarantee that a parent
4932 // item is never dropped). Verified vs PG18 (r207): a=1,b=2 → 3
4933 // rows; a=1,b=[] → 1 row; all-empty → 1 NULL row.
4934 let before = out.len();
4935 for (s_idx, run) in nested_runs.iter().enumerate() {
4936 for nrow in run {
4937 let mut cells = parent_cells.clone();
4938 for (o_idx, w) in nested_widths.iter().enumerate() {
4939 if o_idx == s_idx {
4940 cells.extend(nrow.values.iter().cloned());
4941 } else {
4942 for _ in 0..*w {
4943 cells.push(Value::Null);
4944 }
4945 }
4946 }
4947 out.push(Row::new(cells));
4948 }
4949 }
4950 if out.len() == before {
4951 // Every sibling empty → one all-NULL nested row.
4952 let mut cells = parent_cells.clone();
4953 for w in &nested_widths {
4954 for _ in 0..*w {
4955 cells.push(Value::Null);
4956 }
4957 }
4958 out.push(Row::new(cells));
4959 }
4960 Ok(())
4961 }
4962
4963 /// v7.39 (round 205) — evaluate one Regular column against a row
4964 /// item: EXISTS → bool; else path → at most one value, coerced to
4965 /// the declared type with ON EMPTY / ON ERROR / DEFAULT behaviour.
4966 fn json_table_column_value(
4967 &self,
4968 col: &spg_sql::ast::JsonTableColumn,
4969 item: &crate::json::JsonValue,
4970 vars: Option<&crate::json::JsonValue>,
4971 ) -> Result<Value<'static>, EngineError> {
4972 use spg_sql::ast::{JsonTableColumn as C, JsonTableOnBehavior as B};
4973 let C::Regular {
4974 name,
4975 ty,
4976 path,
4977 exists,
4978 format_json,
4979 wrapper,
4980 on_empty,
4981 on_error,
4982 } = col
4983 else {
4984 unreachable!("caller guards Regular");
4985 };
4986 let matches = crate::json::json_table_path(item, path, vars).map_err(EngineError::Eval)?;
4987 if *exists {
4988 return Ok(Value::Bool(!matches.is_empty()));
4989 }
4990 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
4991 let ctx = EvalContext::new(&empty_schema, None);
4992 let dummy = Row::new(alloc::vec::Vec::new());
4993 let default_of = |b: &B| -> Result<Option<Value<'static>>, EngineError> {
4994 match b {
4995 B::Null => Ok(Some(Value::Null)),
4996 B::Error => Ok(None),
4997 B::Default(e) => Ok(Some(
4998 eval::eval_expr(e, &dummy, &ctx).map_err(EngineError::Eval)?,
4999 )),
5000 }
5001 };
5002 // Empty match set → ON EMPTY.
5003 if matches.is_empty() {
5004 return match default_of(on_empty)? {
5005 Some(v) => coerce_json_table_default(v, *ty, name),
5006 None => Err(EngineError::Unsupported(alloc::format!(
5007 "no SQL/JSON item found for JSON_TABLE column {name:?}"
5008 ))),
5009 };
5010 }
5011 let first = &matches[0];
5012 // FORMAT JSON: return the PG-canonical json representation.
5013 // WITH WRAPPER wraps the whole match SET in an array (even a
5014 // single scalar → `[5]`); without it, the single match's json.
5015 if *format_json {
5016 let text = if *wrapper {
5017 crate::json::JsonValue::Array(matches.clone()).canonical_json_text()
5018 } else {
5019 first.canonical_json_text()
5020 };
5021 return Ok(Value::Json(alloc::borrow::Cow::Owned(text)));
5022 }
5023 if first.is_json_null() {
5024 return Ok(Value::Null);
5025 }
5026 // Coerce the scalar text to the declared type; on failure → ON
5027 // ERROR (default NULL, DEFAULT expr, or raise).
5028 let dt = crate::conversions::column_type_to_data_type(*ty);
5029 let scalar = Value::Text(alloc::borrow::Cow::Owned(first.scalar_text()));
5030 match crate::conversions::coerce_value(scalar, dt, name, 0) {
5031 Ok(v) => Ok(v),
5032 Err(e) => match default_of(on_error)? {
5033 Some(v) => coerce_json_table_default(v, *ty, name),
5034 None => Err(e),
5035 },
5036 }
5037 }
5038
5039 /// table function into (rows, default schema). Dispatch by name.
5040 pub(crate) fn table_fn_rows(
5041 &self,
5042 primary: &TableRef,
5043 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
5044 let (fn_name, args) = primary
5045 .table_fn_call
5046 .as_deref()
5047 .expect("caller guards table_fn_call.is_some()");
5048 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
5049 let ctx = EvalContext::new(&empty_schema, None);
5050 let dummy_row = Row::new(alloc::vec::Vec::new());
5051 let arg0: Option<Value<'static>> = match args.first() {
5052 Some(e) => Some(eval::eval_expr(e, &dummy_row, &ctx).map_err(EngineError::Eval)?),
5053 None => None,
5054 };
5055 match fn_name.as_str() {
5056 // v7.39 (read01 round 76) — `jsonb_populate_record(NULL::t, j)` /
5057 // `…_recordset` (+ json_ variants). The row shape is the BASE
5058 // argument's declared type — a table's or a composite type's
5059 // column list — which only the catalog knows, so the parser hands
5060 // the raw arguments here rather than desugaring blind.
5061 "jsonb_populate_record"
5062 | "json_populate_record"
5063 | "jsonb_populate_recordset"
5064 | "json_populate_recordset" => {
5065 let type_name = match args.first() {
5066 Some(Expr::Cast {
5067 target: spg_sql::ast::CastTarget::Named(n),
5068 ..
5069 }) => n.clone(),
5070 _ => {
5071 return Err(EngineError::Unsupported(alloc::format!(
5072 "{fn_name}(): first argument must name a row type, \
5073 e.g. NULL::mytable"
5074 )));
5075 }
5076 };
5077 let cat = self.active_catalog();
5078 let cols: alloc::vec::Vec<ColumnSchema> = if let Some(t) = cat.get(&type_name) {
5079 t.schema().columns.clone()
5080 } else if let Some(c) = cat.composite_types().get(&type_name) {
5081 c.fields
5082 .iter()
5083 .map(|(n, ty)| ColumnSchema::new(n.clone(), *ty, true))
5084 .collect()
5085 } else {
5086 return Err(EngineError::Unsupported(alloc::format!(
5087 "type \"{type_name}\" does not exist"
5088 )));
5089 };
5090 let json_arg = match args.get(1) {
5091 Some(e) => eval::eval_expr(e, &dummy_row, &ctx).map_err(EngineError::Eval)?,
5092 None => Value::Null,
5093 };
5094 // The set form iterates the JSON array; the scalar form is
5095 // the one-element case of the same walk.
5096 let docs: alloc::vec::Vec<Value<'static>> = if fn_name.ends_with("recordset") {
5097 crate::json::array_element_rows(&json_arg, false, fn_name)
5098 .map_err(EngineError::Eval)?
5099 .into_iter()
5100 .map(|s| s.map_or(Value::Null, Value::json))
5101 .collect()
5102 } else if matches!(json_arg, Value::Null) {
5103 alloc::vec::Vec::new()
5104 } else {
5105 alloc::vec![json_arg]
5106 };
5107 let mut rows = alloc::vec::Vec::with_capacity(docs.len());
5108 for doc in &docs {
5109 let mut vals = alloc::vec::Vec::with_capacity(cols.len());
5110 for c in &cols {
5111 // `->>` semantics: a missing key is NULL, present keys
5112 // arrive as text and cast to the declared column type.
5113 let raw = crate::json::path_get(doc, &Value::text(c.name.clone()), true)
5114 .map_err(EngineError::Eval)?;
5115 let v = if matches!(raw, Value::Null) {
5116 Value::Null
5117 } else {
5118 crate::conversions::coerce_value(raw, c.ty, "", 0)
5119 .map_err(|e| EngineError::Unsupported(alloc::format!("{e:?}")))?
5120 };
5121 vals.push(v);
5122 }
5123 rows.push(Row::new(vals));
5124 }
5125 Ok((rows, cols))
5126 }
5127 // 7.38.1 S5.1 (pg_dump wall #3) — pg_options_to_table:
5128 // a text[] of 'name=value' reloptions/fdw options → one
5129 // (option_name, option_value) row per element. NULL or an
5130 // empty array yields zero rows (PG); an element without
5131 // '=' carries a NULL option_value, matching PG's split.
5132 "pg_options_to_table" => {
5133 let schema = alloc::vec![
5134 ColumnSchema::new("option_name", DataType::Text, true),
5135 ColumnSchema::new("option_value", DataType::Text, true),
5136 ];
5137 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
5138 if let Some(Value::TextArray(items)) = arg0 {
5139 for item in items.into_iter().flatten() {
5140 let (name, value) = match item.split_once('=') {
5141 Some((n, v)) => (Value::text(n), Value::text(v)),
5142 None => (Value::text(item.as_str()), Value::Null),
5143 };
5144 rows.push(Row::new(alloc::vec![name, value]));
5145 }
5146 }
5147 Ok((rows, schema))
5148 }
5149 // 7.38.1 S5.1 (pg_dump wall) — pg_get_sequence_data(oid):
5150 // PG18's per-sequence state SRF, (last_value, is_called).
5151 // pg_dump reads it joined to pg_sequence for every dumped
5152 // sequence's setval line. The oid resolves through the
5153 // same relation_oid mapping seqrelid publishes.
5154 "pg_get_sequence_data" => {
5155 let schema = alloc::vec![
5156 ColumnSchema::new("last_value", DataType::BigInt, false),
5157 ColumnSchema::new("is_called", DataType::Bool, false),
5158 ];
5159 let want = match arg0 {
5160 Some(Value::Int(n)) => i64::from(n),
5161 Some(Value::BigInt(n)) => n,
5162 _ => {
5163 return Err(EngineError::Unsupported(
5164 "pg_get_sequence_data(): argument must be a sequence oid".into(),
5165 ));
5166 }
5167 };
5168 let cat = self.active_catalog();
5169 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
5170 for (name, def) in cat.sequences_all() {
5171 if crate::system_catalog::relation_oid(cat, name) == Some(want) {
5172 rows.push(Row::new(alloc::vec![
5173 Value::BigInt(def.last_value),
5174 Value::Bool(def.is_called),
5175 ]));
5176 break;
5177 }
5178 }
5179 Ok((rows, schema))
5180 }
5181 "pg_partition_tree" => {
5182 let cols = alloc::vec![
5183 ColumnSchema::new("relid".to_string(), DataType::Text, true),
5184 ColumnSchema::new("parentrelid".to_string(), DataType::Text, true),
5185 ColumnSchema::new("isleaf".to_string(), DataType::Bool, true),
5186 ColumnSchema::new("level".to_string(), DataType::Int, true),
5187 ];
5188 let Some(Value::Text(name)) = &arg0 else {
5189 // NULL (or missing) argument → zero rows (PG).
5190 return Ok((alloc::vec::Vec::new(), cols));
5191 };
5192 let entries = crate::partition_walks::tree_of(self.active_catalog(), name.as_ref());
5193 if entries.is_empty() && self.active_catalog().get(name.as_ref()).is_none() {
5194 return Err(EngineError::Unsupported(alloc::format!(
5195 "relation \"{name}\" does not exist"
5196 )));
5197 }
5198 let rows = entries
5199 .into_iter()
5200 .map(|(relid, parent, isleaf, level)| {
5201 Row::new(alloc::vec![
5202 Value::text(relid),
5203 parent.map_or(Value::Null, Value::text),
5204 Value::Bool(isleaf),
5205 #[allow(clippy::cast_possible_truncation)]
5206 Value::Int(level as i32),
5207 ])
5208 })
5209 .collect();
5210 Ok((rows, cols))
5211 }
5212 "pg_partition_ancestors" => {
5213 let cols =
5214 alloc::vec![ColumnSchema::new("relid".to_string(), DataType::Text, true)];
5215 let Some(Value::Text(name)) = &arg0 else {
5216 return Ok((alloc::vec::Vec::new(), cols));
5217 };
5218 let cat = self.active_catalog();
5219 if cat.get(name.as_ref()).is_none() {
5220 return Err(EngineError::Unsupported(alloc::format!(
5221 "relation \"{name}\" does not exist"
5222 )));
5223 }
5224 // A relation outside any partition tree yields no rows (PG).
5225 let in_tree = cat
5226 .get(name.as_ref())
5227 .is_some_and(|t| t.schema().partition_role.is_some());
5228 let rows = if in_tree {
5229 crate::partition_walks::ancestors_of(cat, name.as_ref())
5230 .into_iter()
5231 .map(|n| Row::new(alloc::vec![Value::text(n)]))
5232 .collect()
5233 } else {
5234 alloc::vec::Vec::new()
5235 };
5236 Ok((rows, cols))
5237 }
5238 // v7.39 (round 651) — `ts_debug(config, text)`: what the parser
5239 // saw, what each token was called, which dictionary took it
5240 // and what came out. It is a projection of the same tokenizer
5241 // and the same map the indexer uses, so it cannot describe a
5242 // pipeline other than the one that runs.
5243 "ts_debug" => {
5244 use crate::fts::{TokenType, TsDict};
5245 let cols = alloc::vec![
5246 ColumnSchema::new("alias".to_string(), DataType::Text, false),
5247 ColumnSchema::new("description".to_string(), DataType::Text, false),
5248 ColumnSchema::new("token".to_string(), DataType::Text, false),
5249 ColumnSchema::new("dictionaries".to_string(), DataType::TextArray, false),
5250 ColumnSchema::new("dictionary".to_string(), DataType::Text, true),
5251 ColumnSchema::new("lexemes".to_string(), DataType::TextArray, true),
5252 ];
5253 // PG's one-arg form uses the session configuration; the
5254 // two-arg form names one.
5255 let (cfg_name, text) = match (&arg0, args.get(1)) {
5256 (Some(Value::Text(c)), Some(t)) => {
5257 let v = eval::eval_expr(t, &dummy_row, &ctx).map_err(EngineError::Eval)?;
5258 (c.to_string(), crate::eval::value_to_text(&v))
5259 }
5260 (Some(v), None) => (
5261 alloc::string::String::from("english"),
5262 crate::eval::value_to_text(v),
5263 ),
5264 _ => return Ok((alloc::vec::Vec::new(), cols)),
5265 };
5266 let english = match cfg_name
5267 .trim()
5268 .trim_start_matches("pg_catalog.")
5269 .to_ascii_lowercase()
5270 .as_str()
5271 {
5272 "english" => true,
5273 "simple" => false,
5274 other => {
5275 return Err(EngineError::Unsupported(alloc::format!(
5276 "text search configuration \"{other}\" does not exist"
5277 )));
5278 }
5279 };
5280 let rows = crate::fts::tokenize_typed(&text)
5281 .into_iter()
5282 .map(|tok| {
5283 let dict = tok.ty.dictionary(english);
5284 let dname = dict.map(|d| match d {
5285 TsDict::Simple => "simple",
5286 TsDict::EnglishStem => "english_stem",
5287 });
5288 let folded = tok.text.to_lowercase();
5289 let lexemes = dict.map(|d| match d {
5290 TsDict::Simple => alloc::vec![Some(folded.clone())],
5291 TsDict::EnglishStem => {
5292 if crate::fts::is_english_stopword(&folded) {
5293 alloc::vec::Vec::new()
5294 } else {
5295 alloc::vec![Some(crate::fts::porter_stem(&folded))]
5296 }
5297 }
5298 });
5299 Row::new(alloc::vec![
5300 Value::text(tok.ty.alias()),
5301 Value::text(tok.ty.description()),
5302 Value::text(tok.text),
5303 Value::TextArray(
5304 dname
5305 .map(|n| alloc::vec![Some(alloc::string::String::from(n))])
5306 .unwrap_or_default(),
5307 ),
5308 dname.map_or(Value::Null, Value::text),
5309 lexemes.map_or(Value::Null, Value::TextArray),
5310 ])
5311 })
5312 .collect();
5313 let _ = TokenType::AsciiWord;
5314 Ok((rows, cols))
5315 }
5316 // v7.39 (round 651) — `ts_token_type('default')`, the list the
5317 // parser actually produces. It is a projection of the
5318 // `TokenType` enum the tokenizer and `pg_ts_config_map` both
5319 // read, so the three cannot disagree about what a token is.
5320 "ts_token_type" => {
5321 use crate::fts::TokenType as T;
5322 let cols = alloc::vec![
5323 ColumnSchema::new("tokid".to_string(), DataType::Int, false),
5324 ColumnSchema::new("alias".to_string(), DataType::Text, false),
5325 ColumnSchema::new("description".to_string(), DataType::Text, false),
5326 ];
5327 // PG takes the parser by name or oid; SPG has the one.
5328 if let Some(Value::Text(p)) = &arg0
5329 && !p.eq_ignore_ascii_case("default")
5330 && !p.eq_ignore_ascii_case("pg_catalog.default")
5331 {
5332 return Err(EngineError::Unsupported(alloc::format!(
5333 "text search parser \"{p}\" does not exist"
5334 )));
5335 }
5336 const TYPES: &[T] = &[
5337 T::AsciiWord,
5338 T::Word,
5339 T::NumWord,
5340 T::Email,
5341 T::Url,
5342 T::Host,
5343 T::SFloat,
5344 T::Version,
5345 T::HwordNumPart,
5346 T::HwordPart,
5347 T::HwordAsciiPart,
5348 T::Blank,
5349 T::Tag,
5350 T::Protocol,
5351 T::NumHword,
5352 T::AsciiHword,
5353 T::Hword,
5354 T::UrlPath,
5355 T::File,
5356 T::Float,
5357 T::Int,
5358 T::Uint,
5359 T::Entity,
5360 ];
5361 let rows = TYPES
5362 .iter()
5363 .map(|t| {
5364 Row::new(alloc::vec![
5365 Value::Int(*t as i32),
5366 Value::text(t.alias()),
5367 Value::text(t.description()),
5368 ])
5369 })
5370 .collect();
5371 Ok((rows, cols))
5372 }
5373 // v7.39 (read01 round 65) — a set-returning USER function in FROM
5374 // (`FROM rows_of(2)`). Its body runs through the real executor, like
5375 // every other function body since round 63.
5376 other => {
5377 if !self.active_catalog().functions_named(other).is_empty() {
5378 return self.exec_setof_user_function(other, args, primary.alias.as_deref());
5379 }
5380 Err(EngineError::Unsupported(alloc::format!(
5381 "table function {other}() is not supported in FROM"
5382 )))
5383 }
5384 }
5385 }
5386
5387 /// v7.39 (read01 round 65) — run a `RETURNS SETOF <type>` / `RETURNS
5388 /// TABLE(…)` function in FROM position. The body is a SELECT; the arguments
5389 /// are bound into it as literals and it goes through the read path, so the
5390 /// rows it yields are exactly the rows a hand-written query would see.
5391 ///
5392 /// The column NAMES come from the declared shape: `RETURNS TABLE(id int, v
5393 /// text)` names them, and a `SETOF <scalar>` yields a single column named
5394 /// after the function — PG's rule, and what a bare `SELECT * FROM f()`
5395 /// shows.
5396 fn exec_setof_user_function(
5397 &self,
5398 name: &str,
5399 args: &[spg_sql::ast::Expr],
5400 // v7.39 (read01 round 65) — `FROM evens() AS x` names the single column
5401 // `x`: for a scalar SETOF, the table alias IS the column name (PG).
5402 alias: Option<&str>,
5403 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
5404 // The call's arguments belong to the ENCLOSING query, so they are
5405 // evaluated here and the body sees values.
5406 let empty: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
5407 let arg_ctx = self.ev_ctx(&empty, None);
5408 let dummy = Row::new(alloc::vec::Vec::new());
5409 let mut vals: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
5410 for a in args {
5411 vals.push(eval::eval_expr(a, &dummy, &arg_ctx).map_err(EngineError::Eval)?);
5412 }
5413 self.setof_rows_of(name, &vals, alias)
5414 }
5415
5416 /// v7.39 (read01 round 67) — the set-returning core, on already-evaluated
5417 /// arguments. Shared by the FROM position and the target-list expansion, so
5418 /// a function cannot behave differently depending on where it is called.
5419 pub(crate) fn setof_rows_of(
5420 &self,
5421 name: &str,
5422 arg_values: &[Value<'static>],
5423 alias: Option<&str>,
5424 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
5425 let cat = self.active_catalog();
5426 let overloads = cat.functions_named(name);
5427 let def = overloads
5428 .iter()
5429 .find(|f| spg_storage::function_arg_types(&f.args_repr).len() == arg_values.len())
5430 .ok_or_else(|| {
5431 EngineError::Unsupported(alloc::format!(
5432 "function {name} does not exist with {} argument(s)",
5433 arg_values.len()
5434 ))
5435 })?;
5436 let declared = def.returns.trim().to_string();
5437 let upper = declared.to_ascii_uppercase();
5438 if !upper.starts_with("SETOF") && !upper.starts_with("TABLE(") {
5439 return Err(EngineError::Unsupported(alloc::format!(
5440 "function {name}() does not return a set — it cannot be used in FROM"
5441 )));
5442 }
5443
5444 let arg_names_pl = spg_storage::function_arg_names(&def.args_repr);
5445 // v7.39 (read01 round 66) — a plpgsql SETOF body builds its rows with
5446 // RETURN NEXT / RETURN QUERY; the interpreter collects them.
5447 if def.language.eq_ignore_ascii_case("plpgsql") {
5448 let out_rows = self
5449 .call_plpgsql_setof_fn(def, &arg_names_pl, arg_values)
5450 .map_err(EngineError::Eval)?;
5451 let cols = setof_column_shape(&declared, name, alias, out_rows.first());
5452 let rows = out_rows.into_iter().map(Row::new).collect();
5453 return Ok((rows, cols));
5454 }
5455 let body = def.body.trim().trim_end_matches(';');
5456 let stmt = spg_sql::parser::parse_statement(body).map_err(|e| {
5457 EngineError::Unsupported(alloc::format!("function {name} body does not parse: {e}"))
5458 })?;
5459 let spg_sql::ast::Statement::Select(body_select) = stmt else {
5460 return Err(EngineError::Unsupported(alloc::format!(
5461 "function {name}(): a set-returning body must be a SELECT"
5462 )));
5463 };
5464 let arg_names = spg_storage::function_arg_names(&def.args_repr);
5465 let bound = crate::eval::bind_user_fn_args(
5466 self.active_catalog(),
5467 &body_select,
5468 &arg_names,
5469 arg_values,
5470 )
5471 .map_err(EngineError::Eval)?;
5472 let out = self.exec_select_cancel(&bound, crate::CancelToken::none())?;
5473 let QueryResult::Rows { columns, rows } = out else {
5474 return Ok((alloc::vec::Vec::new(), alloc::vec::Vec::new()));
5475 };
5476 // Name the columns from the DECLARED shape — the same rule the plpgsql
5477 // path above uses, so a body's language cannot change the row shape.
5478 let cols = setof_column_shape_from(&declared, name, alias, &columns);
5479 Ok((rows, cols))
5480 }
5481
5482 fn exec_select_jsonb_each_text(
5483 &self,
5484 stmt: &SelectStatement,
5485 primary: &TableRef,
5486 cancel: CancelToken<'_>,
5487 ) -> Result<QueryResult, EngineError> {
5488 let (each_fn, arg_expr) = primary
5489 .jsonb_each_text_arg
5490 .as_ref()
5491 .map(|(name, expr)| (name.as_str(), expr.as_ref()))
5492 .expect("caller guards jsonb_each_text_arg.is_some()");
5493 // v7.37.17 (17.6 siblings) — the plain jsonb_each / json_each
5494 // forms keep JSON rendering in the value column (JSON null
5495 // stays jsonb 'null', strings keep their quotes).
5496 let as_text = each_fn.ends_with("_text");
5497 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
5498 let ctx = EvalContext::new(&empty_schema, None);
5499 let dummy_row = Row::new(alloc::vec::Vec::new());
5500 let arg_value = eval::eval_expr(arg_expr, &dummy_row, &ctx).map_err(EngineError::Eval)?;
5501 let pairs =
5502 crate::json::each_rows(&arg_value, as_text, each_fn).map_err(EngineError::Eval)?;
5503 let rows: alloc::vec::Vec<Row<'static>> = pairs
5504 .into_iter()
5505 .map(|(k, v)| {
5506 let key_val = Value::text(k);
5507 let value_val = match v {
5508 Some(s) if as_text => Value::text(s),
5509 Some(s) => Value::Json(alloc::borrow::Cow::Owned(s)),
5510 None => Value::Null,
5511 };
5512 Row::new(alloc::vec![key_val, value_val])
5513 })
5514 .collect();
5515 let alias = primary.alias.clone().unwrap_or_else(|| each_fn.to_string());
5516 let value_dtype = if as_text {
5517 spg_storage::DataType::Text
5518 } else {
5519 spg_storage::DataType::Json
5520 };
5521 let key_col = ColumnSchema::new("key".to_string(), spg_storage::DataType::Text, false);
5522 let value_col = ColumnSchema::new("value".to_string(), value_dtype, as_text);
5523 let mut schema_cols = alloc::vec![key_col, value_col];
5524 // `AS t(k, v)` renames key/value positionally (PG behaviour); the
5525 // LATERAL-position form of the same call already honours it.
5526 for (i, new_name) in primary.unnest_column_aliases.iter().enumerate() {
5527 if let Some(col) = schema_cols.get_mut(i) {
5528 col.name = new_name.clone();
5529 }
5530 }
5531 // v7.39 (read01 round 54) — `ev_ctx` threads the catalog; a bare
5532 // `EvalContext::new` drops it and every catalog-dependent cast
5533 // (regclass / enum / composite / domain) silently degrades.
5534 let scan_ctx = self.ev_ctx(&schema_cols, Some(&alias));
5535 // WHERE.
5536 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
5537 let mut out = alloc::vec::Vec::with_capacity(rows.len());
5538 for row in rows {
5539 cancel.check()?;
5540 let v = eval::eval_expr(w, &row, &scan_ctx).map_err(EngineError::Eval)?;
5541 if matches!(v, Value::Bool(true)) {
5542 out.push(row);
5543 }
5544 }
5545 out
5546 } else {
5547 rows
5548 };
5549 // Aggregate dispatch (e.g. SELECT COUNT(*) FROM jsonb_each_text…).
5550 if aggregate::uses_aggregate(stmt) {
5551 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
5552 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
5553 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
5554 .map_err(|err| match err {
5555 EngineError::Eval(ev) => ev,
5556 other => eval::EvalError::TypeMismatch {
5557 detail: alloc::format!("{other}"),
5558 },
5559 })
5560 };
5561 // v7.39 (round 656) — hand the rows over as they are rather than
5562 // collecting a second vector of `RowRef` wrappers. Note this is
5563 // a set-returning-function path, NOT the relational scan: the
5564 // measured O(rows) cost lived in `run_single_table_aggregate`,
5565 // and converting these four first was a miss that cost a full
5566 // round — every test stayed green and the number did not move.
5567 let agg = aggregate::run(
5568 stmt,
5569 crate::join::AggRows::Owned(&filtered),
5570 &schema_cols,
5571 Some(&alias),
5572 Some(&agg_correlated),
5573 self.parallel_runner.0.as_deref(),
5574 Some(self.active_catalog()),
5575 Some(self),
5576 )?;
5577 return self.finish_agg_result(agg, stmt, cancel);
5578 }
5579 // Projection.
5580 let projection =
5581 build_projection(&stmt.items, &schema_cols, &alias, self.backslash_escapes)?;
5582 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
5583 alloc::vec::Vec::with_capacity(filtered.len());
5584 for row in &filtered {
5585 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
5586 for p in &projection {
5587 let v = eval::eval_expr(&p.expr, row, &scan_ctx).map_err(EngineError::Eval)?;
5588 vals.push(v);
5589 }
5590 projected_rows.push(Row::new(vals));
5591 }
5592 let columns: alloc::vec::Vec<ColumnSchema> = projection
5593 .iter()
5594 // v7.39 (read01 round 54) — keep the column's enum identity through
5595 // the projection (it lives outside the DataType lattice), or a
5596 // derived table / UNION / windowed result forgets it and any outer
5597 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
5598 .map(|p| {
5599 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
5600 c.user_enum_type = p.user_enum_type.clone();
5601 c.mysql_fsp = p.mysql_fsp;
5602 c
5603 })
5604 .collect();
5605 // ORDER BY.
5606 if !stmt.order_by.is_empty() {
5607 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = filtered
5608 .iter()
5609 .enumerate()
5610 .map(|(i, r)| -> Result<_, EngineError> {
5611 let keys: Result<Vec<Value<'static>>, EngineError> = stmt
5612 .order_by
5613 .iter()
5614 .map(|ob| {
5615 eval::eval_expr(&ob.expr, r, &scan_ctx).map_err(EngineError::Eval)
5616 })
5617 .collect();
5618 Ok((i, keys?))
5619 })
5620 .collect::<Result<_, _>>()?;
5621 indexed.sort_by(|a, b| {
5622 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
5623 let o = &stmt.order_by[idx];
5624 let cmp = order_by_value_cmp_in(
5625 o.desc,
5626 o.nulls_first,
5627 ka,
5628 kb,
5629 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
5630 );
5631 if cmp != core::cmp::Ordering::Equal {
5632 return cmp;
5633 }
5634 }
5635 core::cmp::Ordering::Equal
5636 });
5637 projected_rows = indexed
5638 .into_iter()
5639 .map(|(i, _)| projected_rows[i].clone())
5640 .collect();
5641 }
5642 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
5643 if stmt.distinct {
5644 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
5645 }
5646 if let Some(offset) = stmt.offset_literal() {
5647 let off = (offset as usize).min(projected_rows.len());
5648 projected_rows.drain(..off);
5649 }
5650 if let Some(limit) = stmt.limit_literal() {
5651 projected_rows.truncate(limit as usize);
5652 }
5653 Ok(QueryResult::Rows {
5654 columns,
5655 rows: projected_rows,
5656 })
5657 }
5658
5659 /// v7.37.17 (17.6 siblings) — execute `SELECT … FROM
5660 /// ( SELECT … ) alias` in primary position. The inner SELECT
5661 /// materialises once through the regular bare-select executor
5662 /// (UNION tails included), then the outer WHERE / aggregate /
5663 /// projection / ORDER BY / LIMIT pipeline runs over the
5664 /// synthetic table — the same post-materialisation shape as
5665 /// exec_select_jsonb_each_text, generalised to N columns.
5666 fn exec_select_derived(
5667 &self,
5668 stmt: &SelectStatement,
5669 primary: &TableRef,
5670 cancel: CancelToken<'_>,
5671 ) -> Result<QueryResult, EngineError> {
5672 let inner = primary
5673 .lateral_subquery
5674 .as_deref()
5675 .expect("caller guards lateral_subquery.is_some()");
5676 // exec_select_cancel is the union-aware wrapper — the inner
5677 // SELECT may carry UNION tails on stmt.unions.
5678 let QueryResult::Rows {
5679 columns: inner_cols,
5680 rows,
5681 } = self.exec_select_cancel(inner, cancel)?
5682 else {
5683 return Err(EngineError::Unsupported(
5684 "derived table subquery must return rows".into(),
5685 ));
5686 };
5687 let alias = primary
5688 .alias
5689 .clone()
5690 .unwrap_or_else(|| primary.name.clone());
5691 // `AS t(a, b)` renames the materialised columns positionally
5692 // (extra inner columns keep their own names, PG behaviour).
5693 let mut schema_cols: alloc::vec::Vec<ColumnSchema> = inner_cols;
5694 // v7.39 (read01 round 78) — a column-alias list longer than the item is
5695 // the error PG reports; SPG used to let the extra names through and then
5696 // fail two layers downstream with "column not found: <the extra name>".
5697 let n_out = schema_cols.len() + usize::from(primary.with_ordinality);
5698 if primary.unnest_column_aliases.len() > n_out {
5699 return Err(EngineError::Unsupported(alloc::format!(
5700 "table \"{alias}\" has {n_out} columns available but {} columns specified",
5701 primary.unnest_column_aliases.len()
5702 )));
5703 }
5704 if primary.scalar_fn_item && schema_cols.len() == 1 {
5705 schema_cols[0].scalar_row_source = true;
5706 }
5707 // v7.39 (read01 round 78) — WITH ORDINALITY on a table function that
5708 // rides this channel (regexp_matches): a trailing bigint counter, 1-based.
5709 // The column-alias list, if given, names it like any other column.
5710 let mut rows = rows;
5711 if primary.with_ordinality {
5712 schema_cols.push(ColumnSchema::new(
5713 "ordinality".to_string(),
5714 DataType::BigInt,
5715 false,
5716 ));
5717 rows = rows
5718 .into_iter()
5719 .enumerate()
5720 .map(|(i, r)| {
5721 let mut v = r.values;
5722 #[allow(clippy::cast_possible_wrap)]
5723 v.push(Value::BigInt(i as i64 + 1));
5724 Row::new(v)
5725 })
5726 .collect();
5727 }
5728 for (i, new_name) in primary.unnest_column_aliases.iter().enumerate() {
5729 if let Some(col) = schema_cols.get_mut(i) {
5730 col.name = new_name.clone();
5731 }
5732 }
5733 self.exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
5734 }
5735
5736 /// v7.39 (read01 partitionfuncs.c) — shared synthetic-source SELECT
5737 /// pipeline (WHERE / aggregate / projection / ORDER BY / DISTINCT /
5738 /// OFFSET / LIMIT) over a pre-materialised row set. Drives the
5739 /// derived-table executor and the FROM-position table functions.
5740 fn exec_select_over_rows(
5741 &self,
5742 stmt: &SelectStatement,
5743 rows: alloc::vec::Vec<Row<'static>>,
5744 schema_cols: alloc::vec::Vec<ColumnSchema>,
5745 alias: &str,
5746 cancel: CancelToken<'_>,
5747 ) -> Result<QueryResult, EngineError> {
5748 let scan_ctx = self.ev_ctx(&schema_cols, Some(alias));
5749 // v7.37 D.21 — correlated subqueries in the WHERE / projection may
5750 // reference this derived table's columns (`… WHERE u.gg = t.g` where t
5751 // is `(VALUES …) t`). Resolve them per-row via eval_expr_with_correlated
5752 // (the same path the aggregate branch uses); the old plain eval_expr let
5753 // a ScalarSubquery reach row-eval unresolved ("engine resolver bug").
5754 let corr_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
5755 // WHERE.
5756 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
5757 let mut out = alloc::vec::Vec::with_capacity(rows.len());
5758 for row in rows {
5759 cancel.check()?;
5760 let v = self.eval_expr_with_correlated(
5761 w,
5762 &row,
5763 &scan_ctx,
5764 cancel,
5765 Some(&mut corr_memo.borrow_mut()),
5766 )?;
5767 if matches!(v, Value::Bool(true)) {
5768 out.push(row);
5769 }
5770 }
5771 out
5772 } else {
5773 rows
5774 };
5775 // Aggregate dispatch.
5776 if aggregate::uses_aggregate(stmt) {
5777 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
5778 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
5779 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
5780 .map_err(|err| match err {
5781 EngineError::Eval(ev) => ev,
5782 other => eval::EvalError::TypeMismatch {
5783 detail: alloc::format!("{other}"),
5784 },
5785 })
5786 };
5787 // v7.39 (round 656) — hand the rows over as they are rather than
5788 // collecting a second vector of `RowRef` wrappers. Note this is
5789 // a set-returning-function path, NOT the relational scan: the
5790 // measured O(rows) cost lived in `run_single_table_aggregate`,
5791 // and converting these four first was a miss that cost a full
5792 // round — every test stayed green and the number did not move.
5793 let agg = aggregate::run(
5794 stmt,
5795 crate::join::AggRows::Owned(&filtered),
5796 &schema_cols,
5797 Some(alias),
5798 Some(&agg_correlated),
5799 self.parallel_runner.0.as_deref(),
5800 Some(self.active_catalog()),
5801 Some(self),
5802 )?;
5803 return self.finish_agg_result(agg, stmt, cancel);
5804 }
5805 // Projection.
5806 let projection =
5807 build_projection(&stmt.items, &schema_cols, alias, self.backslash_escapes)?;
5808 // v7.39 (round 621) — a target-list SRF expands here too. This tail
5809 // serves VALUES, a derived table and `ROWS FROM (…)`, and knew nothing
5810 // about them: `SELECT unnest(ARRAY[1,2]), x FROM (VALUES (3),(4)) v(x)`
5811 // answered `function unnest(integer[]) does not exist` for a query PG
5812 // answers.
5813 let srf_idxs = self.srf_target_idxs(&projection);
5814 let mut src_of_row: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
5815 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
5816 alloc::vec::Vec::with_capacity(filtered.len());
5817 if !srf_idxs.is_empty() {
5818 let (rows, src) =
5819 expand_projection_srfs(self, &projection, &srf_idxs, &filtered, &scan_ctx)?;
5820 projected_rows = rows;
5821 src_of_row = src;
5822 } else {
5823 for row in &filtered {
5824 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
5825 for p in &projection {
5826 let v = self.eval_expr_with_correlated(
5827 &p.expr,
5828 row,
5829 &scan_ctx,
5830 cancel,
5831 Some(&mut corr_memo.borrow_mut()),
5832 )?;
5833 vals.push(v);
5834 }
5835 projected_rows.push(Row::new(vals));
5836 }
5837 }
5838 let columns: alloc::vec::Vec<ColumnSchema> = projection
5839 .iter()
5840 // v7.39 (read01 round 54) — keep the column's enum identity through
5841 // the projection (it lives outside the DataType lattice), or a
5842 // derived table / UNION / windowed result forgets it and any outer
5843 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
5844 .map(|p| {
5845 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
5846 c.user_enum_type = p.user_enum_type.clone();
5847 c.mysql_fsp = p.mysql_fsp;
5848 c
5849 })
5850 .collect();
5851 // ORDER BY over the source rows (same shape as the other
5852 // synthetic-table executors).
5853 // v7.39 (read01 round 80) — a positional key (`ORDER BY 1`) means the Nth
5854 // OUTPUT column. Evaluated as an expression, as it was here, the literal
5855 // `1` is just the constant 1: the same sort key for every row, so the
5856 // sort ran and changed nothing. `SELECT unnest(ARRAY['B','a','A','b'])
5857 // ORDER BY 1` (which the parser turns into `SELECT * FROM unnest(…)`,
5858 // landing on this executor) came back in input order.
5859 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
5860 if !order_by.is_empty() {
5861 // v7.39 (round 621) — one entry per OUTPUT row, since a target-list
5862 // SRF makes more of them than there were inputs.
5863 let out_cols = if srf_idxs.is_empty() {
5864 alloc::vec![None; order_by.len()]
5865 } else {
5866 srf_order_output_cols(&order_by, &projection)
5867 };
5868 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = projected_rows
5869 .iter()
5870 .enumerate()
5871 .map(|(k, out)| -> Result<_, EngineError> {
5872 let r = &filtered[src_of_row.get(k).copied().unwrap_or(k)];
5873 let keys: Result<Vec<Value<'static>>, EngineError> = order_by
5874 .iter()
5875 .zip(out_cols.iter())
5876 .map(|(ob, oc)| {
5877 // v7.39 (read01 round 54) — this path builds its
5878 // sort keys itself instead of going through
5879 // `build_order_keys`, so it skipped the enum-ordinal
5880 // substitution: an OUTER `ORDER BY <enum col>` over
5881 // a DERIVED TABLE sorted by the label TEXT, not by
5882 // member order. Silently wrong rows, not an error.
5883 let v = srf_order_key(ob, *oc, out, r, &scan_ctx)?;
5884 Ok(
5885 match crate::orderby::enum_order_ordinal(&ob.expr, &v, &scan_ctx) {
5886 Some(ord) => Value::Float(ord),
5887 None => v,
5888 },
5889 )
5890 })
5891 .collect();
5892 Ok((k, keys?))
5893 })
5894 .collect::<Result<_, _>>()?;
5895 indexed.sort_by(|a, b| {
5896 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
5897 let o = &stmt.order_by[idx];
5898 let cmp = order_by_value_cmp_in(
5899 o.desc,
5900 o.nulls_first,
5901 ka,
5902 kb,
5903 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
5904 );
5905 if cmp != core::cmp::Ordering::Equal {
5906 return cmp;
5907 }
5908 }
5909 core::cmp::Ordering::Equal
5910 });
5911 projected_rows = indexed
5912 .into_iter()
5913 .map(|(i, _)| projected_rows[i].clone())
5914 .collect();
5915 }
5916 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
5917 if stmt.distinct {
5918 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
5919 }
5920 if let Some(offset) = stmt.offset_literal() {
5921 let off = (offset as usize).min(projected_rows.len());
5922 projected_rows.drain(..off);
5923 }
5924 if let Some(limit) = stmt.limit_literal() {
5925 projected_rows.truncate(limit as usize);
5926 }
5927 Ok(QueryResult::Rows {
5928 columns,
5929 rows: projected_rows,
5930 })
5931 }
5932
5933 /// Constant `SELECT` with no FROM: evaluate each projection item
5934 /// once against an empty dummy row (`SELECT 1`, `SELECT '7'::INT`).
5935 fn exec_constant_select(&self, stmt: &SelectStatement) -> Result<QueryResult, EngineError> {
5936 let empty_schema: Vec<ColumnSchema> = Vec::new();
5937 let ctx = self.ev_ctx(&empty_schema, None);
5938 // v7.39 (read01 round 106) — an aggregate with no FROM runs over the
5939 // single implicit row (`SELECT count(*)` → 1, `SELECT sum(5)` → 5,
5940 // `SELECT string_agg('x',',')` → x). Before this it fell through to the
5941 // scalar projection, where the aggregate name looked like an unknown
5942 // function. The WHERE filters that one row, so `… WHERE false` leaves
5943 // the aggregate zero input rows (`count(*)` → 0).
5944 if aggregate::uses_aggregate(stmt) {
5945 let dummy = Row::new(Vec::new());
5946 let passes = match &stmt.where_ {
5947 Some(w) => matches!(eval::eval_expr(w, &dummy, &ctx)?, Value::Bool(true)),
5948 None => true,
5949 };
5950 let rows: Vec<RowRef<'_>> = if passes {
5951 alloc::vec![RowRef::Owned(&dummy)]
5952 } else {
5953 Vec::new()
5954 };
5955 let agg = aggregate::run(
5956 stmt,
5957 crate::join::AggRows::Refs(&rows),
5958 &empty_schema,
5959 None,
5960 None,
5961 self.parallel_runner.0.as_deref(),
5962 Some(self.active_catalog()),
5963 Some(self),
5964 )?;
5965 return self.finish_agg_result(agg, stmt, CancelToken::none());
5966 }
5967 let projection = build_projection(&stmt.items, &empty_schema, "", self.backslash_escapes)?;
5968 // `SELECT … WHERE cond` with no FROM — the one conceptual
5969 // row survives only when the condition is true (previously
5970 // the WHERE was silently ignored: `SELECT 1 WHERE false`
5971 // returned a row).
5972 let dummy_row = Row::new(Vec::new());
5973 if let Some(w) = &stmt.where_ {
5974 let cond = eval::eval_expr(w, &dummy_row, &ctx)?;
5975 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
5976 let columns: Vec<ColumnSchema> = projection
5977 .into_iter()
5978 .map(|p| {
5979 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
5980 c.user_enum_type = p.user_enum_type;
5981 c.collation_name = p.collation_name;
5982 c.mysql_fsp = p.mysql_fsp;
5983 c
5984 })
5985 .collect();
5986 return Ok(QueryResult::Rows {
5987 columns,
5988 rows: Vec::new(),
5989 });
5990 }
5991 }
5992 // v7.38 (read01, T15) — a top-level SRF that the parser did NOT rewrite
5993 // into a FROM item (regexp_matches, whose rows are arrays and so cannot
5994 // desugar to unnest) expands here: one output row per SRF row, sibling
5995 // scalar columns repeated. unnest / array_elements / path_query reach a
5996 // real FROM via the parser rewrite and never land here.
5997 // v7.39 (read01 round 67) — every SRF in the list, in lockstep.
5998 let srf_idxs = self.srf_target_idxs(&projection);
5999 if !srf_idxs.is_empty() {
6000 let mut rows = expand_srf_row(self, &projection, &srf_idxs, &dummy_row, &ctx)?;
6001 let columns: Vec<ColumnSchema> = projection
6002 .into_iter()
6003 .map(|p| {
6004 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
6005 c.user_enum_type = p.user_enum_type;
6006 c.collation_name = p.collation_name;
6007 c.mysql_fsp = p.mysql_fsp;
6008 c
6009 })
6010 .collect();
6011 // v7.39 (read01 round 80) — a FROM-less SELECT still has an ORDER BY,
6012 // an OFFSET and a LIMIT, and they apply to the rows the SRF expanded
6013 // to. This returned straight out of the expansion, so
6014 // `SELECT unnest(ARRAY['B','a','A','b']) ORDER BY 1` came back in
6015 // input order — the sort was not wrong, it never ran. (There is
6016 // exactly one conceptual input row here, which is why the ordinary
6017 // scan pipeline is not on this path at all.)
6018 if !stmt.order_by.is_empty() {
6019 let synth_ctx =
6020 EvalContext::new(&columns, None).with_catalog(self.active_catalog());
6021 let resolved: Vec<spg_sql::ast::OrderBy> = stmt
6022 .order_by
6023 .iter()
6024 .map(|o| {
6025 let mut o = o.clone();
6026 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &o.expr
6027 && *n >= 1
6028 && let Ok(idx) = usize::try_from(*n - 1)
6029 && idx < columns.len()
6030 {
6031 o.expr = Expr::Column(spg_sql::ast::ColumnName {
6032 qualifier: None,
6033 name: columns[idx].name.clone(),
6034 });
6035 }
6036 o
6037 })
6038 .collect();
6039 let descs: Vec<bool> = resolved.iter().map(|o| o.desc).collect();
6040 let mut tagged: Vec<(Vec<OrderKey>, Row)> = Vec::with_capacity(rows.len());
6041 for r in rows {
6042 let keys = build_order_keys(&resolved, &r, &synth_ctx)?;
6043 tagged.push((keys, r));
6044 }
6045 sort_by_keys(&mut tagged, &descs);
6046 rows = tagged.into_iter().map(|(_, r)| r).collect();
6047 }
6048 apply_offset_and_limit(&mut rows, stmt.offset_literal(), stmt.limit_literal());
6049 return Ok(QueryResult::Rows { columns, rows });
6050 }
6051 let mut values = Vec::with_capacity(projection.len());
6052 for p in &projection {
6053 values.push(eval::eval_expr(&p.expr, &dummy_row, &ctx)?);
6054 }
6055 let columns: Vec<ColumnSchema> = projection
6056 .into_iter()
6057 .map(|p| {
6058 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
6059 c.user_enum_type = p.user_enum_type;
6060 c.collation_name = p.collation_name;
6061 c.mysql_fsp = p.mysql_fsp;
6062 c
6063 })
6064 .collect();
6065 // v7.39 (round 239) — the FROM-less scalar path ignored LIMIT and
6066 // OFFSET entirely, so `SELECT 1 LIMIT 0` returned its row where PG
6067 // returns none. (The SRF and aggregate arms above already applied
6068 // them; this tail was the one that didn't.)
6069 let mut rows = alloc::vec![Row::new(values)];
6070 apply_offset_and_limit(&mut rows, stmt.offset_literal(), stmt.limit_literal());
6071 Ok(QueryResult::Rows { columns, rows })
6072 }
6073
6074 /// v7.37.x (docker-fair INSUBQ attack) — pre-replacement short-
6075 /// circuit. Catches
6076 /// SELECT COUNT(*) FROM A WHERE A.pk IN (<uncorrelated subquery>)
6077 /// BEFORE `resolve_select_subqueries` materialises the inner result
6078 /// as `Vec<Expr::Literal>`. Runs the inner once, collects the
6079 /// values into a `HashSet<i64>` directly, then probes A.pk per
6080 /// HashSet entry and tallies. Saves the Expr-literal roundtrip
6081 /// (~150 µs / query at INSUBQ benchmark scale).
6082 pub(crate) fn try_count_star_pk_in_subquery_fast(
6083 &self,
6084 stmt: &SelectStatement,
6085 cancel: CancelToken<'_>,
6086 ) -> Result<Option<QueryResult>, EngineError> {
6087 use spg_sql::ast::SelectItem;
6088 if stmt.distinct
6089 || stmt.limit_with_ties
6090 || stmt.group_by.is_some()
6091 || stmt.having.is_some()
6092 || !stmt.unions.is_empty()
6093 || !stmt.order_by.is_empty()
6094 || stmt.limit.is_some()
6095 || stmt.offset.is_some()
6096 || stmt.items.len() != 1
6097 {
6098 return Ok(None);
6099 }
6100 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
6101 return Ok(None);
6102 };
6103 let is_count_star = matches!(expr, Expr::FunctionCall { name, args }
6104 if name.eq_ignore_ascii_case("count_star") && args.is_empty());
6105 if !is_count_star {
6106 return Ok(None);
6107 }
6108 let Some(from) = stmt.from.as_ref() else {
6109 return Ok(None);
6110 };
6111 if !from.joins.is_empty()
6112 || from.primary.lateral_subquery.is_some()
6113 || from.primary.unnest_expr.is_some()
6114 || from.primary.generate_series_args.is_some()
6115 || from.primary.table_fn_call.is_some()
6116 || from.primary.as_of_segment.is_some()
6117 {
6118 return Ok(None);
6119 }
6120 let Some(where_expr) = stmt.where_.as_ref() else {
6121 return Ok(None);
6122 };
6123 // The WHERE conjunct must be a bare `<col> IN (subquery)` with
6124 // negated=false; no other predicates.
6125 let Expr::InSubquery {
6126 expr: col_expr,
6127 subquery,
6128 negated: false,
6129 } = where_expr
6130 else {
6131 return Ok(None);
6132 };
6133 let Expr::Column(c) = col_expr.as_ref() else {
6134 return Ok(None);
6135 };
6136 let outer_alias = from
6137 .primary
6138 .alias
6139 .as_deref()
6140 .unwrap_or(from.primary.name.as_str());
6141 if let Some(q) = c.qualifier.as_deref()
6142 && !q.eq_ignore_ascii_case(outer_alias)
6143 {
6144 return Ok(None);
6145 }
6146 // Outer column must be a single-column PK on integer family.
6147 let catalog = self.active_catalog();
6148 let Some(outer_table) = catalog.get(from.primary.name.as_str()) else {
6149 return Ok(None);
6150 };
6151 let outer_schema = outer_table.schema();
6152 let Some(outer_pos) = outer_schema
6153 .columns
6154 .iter()
6155 .position(|s| s.name.eq_ignore_ascii_case(&c.name))
6156 else {
6157 return Ok(None);
6158 };
6159 if !matches!(
6160 outer_schema.columns[outer_pos].ty,
6161 spg_storage::DataType::BigInt
6162 | spg_storage::DataType::Int
6163 | spg_storage::DataType::SmallInt
6164 ) {
6165 return Ok(None);
6166 }
6167 if !outer_schema
6168 .uniqueness_constraints
6169 .iter()
6170 .any(|u| u.is_primary_key && u.columns.as_slice() == [outer_pos])
6171 {
6172 return Ok(None);
6173 }
6174 let Some(idx) = outer_table.index_on(outer_pos) else {
6175 return Ok(None);
6176 };
6177 // Inner must be uncorrelated. The cheap-correlation pre-check
6178 // exists upstream; here we just attempt the bare exec.
6179 if crate::subquery::select_is_correlated(subquery) {
6180 return Ok(None);
6181 }
6182 let mut inner = (**subquery).clone();
6183 self.resolve_select_subqueries(&mut inner, cancel)?;
6184 let r = match self.exec_bare_select_cancel(&inner, cancel) {
6185 Ok(r) => r,
6186 Err(_) => return Ok(None),
6187 };
6188 let QueryResult::Rows { columns, rows, .. } = r else {
6189 return Ok(None);
6190 };
6191 if columns.len() != 1 {
6192 return Ok(None);
6193 }
6194 // v7.37.43 (INSUBQ B-1) — inner-uniqueness check. If the inner
6195 // subquery projects a column known to be UNIQUE/PK on its table
6196 // (statically: `SELECT <col> FROM <tbl> WHERE …` where <col> is
6197 // in `tbl.uniqueness_constraints`), survivor values are
6198 // guaranteed distinct and the per-survivor `HashSet::insert`
6199 // dedup check is redundant. ~25 ns × N_inner-survivors saved.
6200 //
6201 // Inlined check — gated on: no DISTINCT/GROUP/UNION/JOIN, single
6202 // projection that is a bare Column ref, table-column lookup in
6203 // catalog confirms the column appears as a unique constraint's
6204 // sole member. UNIQUE NOT NULL is required — a nullable unique
6205 // column may have multiple NULLs, but NULLs are already skipped
6206 // above (`Value::Null => continue`), so a UNIQUE-only column is
6207 // still safe to dedup-skip.
6208 let inner_unique = (|| -> bool {
6209 if inner.distinct
6210 || inner.group_by.is_some()
6211 || !inner.unions.is_empty()
6212 || inner.having.is_some()
6213 || inner.items.len() != 1
6214 {
6215 return false;
6216 }
6217 let Some(inner_from) = inner.from.as_ref() else {
6218 return false;
6219 };
6220 if !inner_from.joins.is_empty()
6221 || inner_from.primary.lateral_subquery.is_some()
6222 || inner_from.primary.unnest_expr.is_some()
6223 || inner_from.primary.generate_series_args.is_some()
6224 || inner_from.primary.table_fn_call.is_some()
6225 {
6226 return false;
6227 }
6228 let SelectItem::Expr { expr: proj, .. } = &inner.items[0] else {
6229 return false;
6230 };
6231 let Expr::Column(pc) = proj else {
6232 return false;
6233 };
6234 let inner_alias = inner_from
6235 .primary
6236 .alias
6237 .as_deref()
6238 .unwrap_or(inner_from.primary.name.as_str());
6239 if let Some(q) = pc.qualifier.as_deref()
6240 && !q.eq_ignore_ascii_case(inner_alias)
6241 {
6242 return false;
6243 }
6244 let Some(inner_table) = catalog.get(inner_from.primary.name.as_str()) else {
6245 return false;
6246 };
6247 let isch = inner_table.schema();
6248 let Some(ipos) = isch
6249 .columns
6250 .iter()
6251 .position(|s| s.name.eq_ignore_ascii_case(&pc.name))
6252 else {
6253 return false;
6254 };
6255 isch.uniqueness_constraints
6256 .iter()
6257 .any(|u| u.columns.as_slice() == [ipos])
6258 })();
6259 // Collect inner i64 values directly into a HashSet, then probe.
6260 let mut count: i64 = 0;
6261 let mut probed = if inner_unique {
6262 hashbrown::HashSet::<i64>::new()
6263 } else {
6264 hashbrown::HashSet::<i64>::with_capacity(rows.len())
6265 };
6266 for row in &rows {
6267 let v = row.values.first().cloned().unwrap_or(Value::Null);
6268 let n = match v {
6269 Value::BigInt(n) => n,
6270 Value::Int(n) => i64::from(n),
6271 Value::SmallInt(n) => i64::from(n),
6272 Value::Null => continue,
6273 _ => return Ok(None),
6274 };
6275 // De-duplicate inner key set so a duplicate inner value
6276 // doesn't double-count the same outer row. Skipped when
6277 // the inner projection is statically unique.
6278 if !inner_unique && !probed.insert(n) {
6279 continue;
6280 }
6281 // v7.37.43 (INSUBQ B-2 + B-4) — direct i64 PK probe, skipping
6282 // the `IndexKey::from_value` enum-dispatch and the per-call
6283 // `IndexKey` wrapper construction. The outer column is
6284 // already gated to integer-family above, so an i64 key
6285 // always corresponds to a valid PK lookup.
6286 if !idx.lookup_eq_i64(n).is_empty() {
6287 count += 1;
6288 }
6289 }
6290 let columns_out = alloc::vec![ColumnSchema::new(
6291 "count".to_string(),
6292 spg_storage::DataType::BigInt,
6293 false,
6294 )];
6295 let rows_out = alloc::vec![Row::new(alloc::vec![Value::BigInt(count)])];
6296 Ok(Some(QueryResult::Rows {
6297 columns: columns_out,
6298 rows: rows_out,
6299 }))
6300 }
6301
6302 /// v7.37.x (docker-fair INSUBQ attack) — short-circuit
6303 /// SELECT COUNT(*) FROM A WHERE A.pk IN (literal list)
6304 /// (the post-subquery-replacement shape of the INSUBQ probe
6305 /// `SELECT COUNT(*) FROM A WHERE A.pk IN (SELECT k FROM B WHERE …)`).
6306 /// The general aggregate path materialises every seeked row into
6307 /// a `Vec<Cow<Row>>`, then runs the aggregate executor over it.
6308 /// For COUNT(*) we only care how many keys hit; iterate the list
6309 /// and tally `idx.lookup_eq(key)` non-empty results, skipping the
6310 /// row materialisation, the aggregate state machine, and the per-
6311 /// row WHERE re-eval (the seek already filtered by the same list).
6312 /// Returns `None` when the shape doesn't match.
6313 fn try_count_star_pk_in_list_fast(
6314 &self,
6315 stmt: &SelectStatement,
6316 table: &spg_storage::Table,
6317 schema_cols: &[ColumnSchema],
6318 alias: &str,
6319 ) -> Option<QueryResult> {
6320 use spg_sql::ast::{ColumnName, SelectItem};
6321 // Gates on the SELECT shape.
6322 if stmt.distinct
6323 || stmt.limit_with_ties
6324 || stmt.group_by.is_some()
6325 || stmt.having.is_some()
6326 || !stmt.unions.is_empty()
6327 || !stmt.order_by.is_empty()
6328 || stmt.limit.is_some()
6329 || stmt.offset.is_some()
6330 || stmt.items.len() != 1
6331 {
6332 return None;
6333 }
6334 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
6335 return None;
6336 };
6337 let is_count_star = matches!(expr, Expr::FunctionCall { name, args }
6338 if name.eq_ignore_ascii_case("count_star") && args.is_empty());
6339 if !is_count_star {
6340 return None;
6341 }
6342 // WHERE must be `<col> IN (literal list)` with no other
6343 // conjuncts (the seek result is a true subset of the row
6344 // population for this predicate).
6345 let where_expr = stmt.where_.as_ref()?;
6346 let Expr::InList {
6347 expr: col_expr,
6348 list,
6349 negated: false,
6350 } = where_expr
6351 else {
6352 return None;
6353 };
6354 let Expr::Column(c) = col_expr.as_ref() else {
6355 return None;
6356 };
6357 if let Some(q) = c.qualifier.as_deref()
6358 && !q.eq_ignore_ascii_case(alias)
6359 {
6360 return None;
6361 }
6362 let col_pos = schema_cols
6363 .iter()
6364 .position(|s| s.name.eq_ignore_ascii_case(&c.name))?;
6365 // The column must be a single-column PK on an integer family
6366 // — the same gate the SCALARSQ + LEFT-ANTI-JOIN fast paths use,
6367 // so the antiset stays collision-free under `HashSet<i64>`.
6368 let schema = table.schema();
6369 if !matches!(
6370 schema.columns[col_pos].ty,
6371 spg_storage::DataType::BigInt
6372 | spg_storage::DataType::Int
6373 | spg_storage::DataType::SmallInt
6374 ) {
6375 return None;
6376 }
6377 if !schema
6378 .uniqueness_constraints
6379 .iter()
6380 .any(|u| u.is_primary_key && u.columns.as_slice() == [col_pos])
6381 {
6382 return None;
6383 }
6384 let idx = table.index_on(col_pos)?;
6385 // Tally non-empty seek results across all literal values.
6386 let mut count: i64 = 0;
6387 for lit in list {
6388 let Expr::Literal(l) = lit else {
6389 return None;
6390 };
6391 // r1039 — through the shared resolver, so a literal spelled
6392 // in another type ('5' against an integer PK) is read as the
6393 // column's before it becomes a key. This tally answers from
6394 // the index alone, so a key in the wrong space would return a
6395 // COUNT of zero rather than fall back to a scan.
6396 let col = schema.columns.get(col_pos)?;
6397 let v = crate::index_access::literal_as_column_value(l, col, col_pos)?;
6398 let key = spg_storage::IndexKey::from_value_for_column(&v, col.ty)?;
6399 if !idx.lookup_eq(&key).is_empty() {
6400 count += 1;
6401 }
6402 }
6403 let columns = alloc::vec![ColumnSchema::new(
6404 "count".to_string(),
6405 spg_storage::DataType::BigInt,
6406 false,
6407 )];
6408 let rows = alloc::vec![Row::new(alloc::vec![Value::BigInt(count)])];
6409 let _ = ColumnName {
6410 qualifier: None,
6411 name: String::new(),
6412 };
6413 Some(QueryResult::Rows { columns, rows })
6414 }
6415
6416 /// v7.38 (perf, exact-range count) — `SELECT count(*) FROM t WHERE <col>
6417 /// BETWEEN a AND b` on an indexed column. The index range walk yields
6418 /// exactly the matching (visible) rows, so we count locators directly —
6419 /// skipping the row materialisation, the aggregate state machine, and the
6420 /// per-row WHERE re-eval the general path pays. Turns the `range_count`
6421 /// endpoint from tied-with-PG (superset re-eval) into a clear win. None
6422 /// when the shape doesn't match.
6423 fn try_count_star_indexed_range_fast(
6424 &self,
6425 stmt: &SelectStatement,
6426 table: &spg_storage::Table,
6427 schema_cols: &[ColumnSchema],
6428 alias: &str,
6429 snapshot: &spg_storage::snapshot::Snapshot,
6430 ) -> Option<QueryResult> {
6431 use spg_sql::ast::SelectItem;
6432 if stmt.distinct
6433 || stmt.limit_with_ties
6434 || stmt.group_by.is_some()
6435 || stmt.having.is_some()
6436 || !stmt.unions.is_empty()
6437 || !stmt.order_by.is_empty()
6438 || stmt.limit.is_some()
6439 || stmt.offset.is_some()
6440 || stmt.items.len() != 1
6441 {
6442 return None;
6443 }
6444 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
6445 return None;
6446 };
6447 let is_count_star = matches!(expr, Expr::FunctionCall { name, args }
6448 if name.eq_ignore_ascii_case("count_star") && args.is_empty());
6449 if !is_count_star {
6450 return None;
6451 }
6452 let where_expr = stmt.where_.as_ref()?;
6453 let count =
6454 crate::index_access::try_range_count(where_expr, schema_cols, table, alias, snapshot)?;
6455 let columns = alloc::vec![ColumnSchema::new(
6456 "count".to_string(),
6457 spg_storage::DataType::BigInt,
6458 false,
6459 )];
6460 let rows = alloc::vec![Row::new(alloc::vec![Value::BigInt(count)])];
6461 Some(QueryResult::Rows { columns, rows })
6462 }
6463
6464 /// Single-table aggregate path: filter the (optionally index-seeked)
6465 /// rows, then hand off to the aggregate executor which does its own
6466 /// projection + ORDER BY before `finish_agg_result` applies LIMIT.
6467 fn run_single_table_aggregate<'a>(
6468 &self,
6469 stmt: &SelectStatement,
6470 table: &'a spg_storage::Table,
6471 schema_cols: &'a [ColumnSchema],
6472 alias: &str,
6473 indexed_rows: Option<Vec<Cow<'a, Row<'static>>>>,
6474 cancel: CancelToken<'_>,
6475 ) -> Result<QueryResult, EngineError> {
6476 // v7.38 (read01 U15) — per-scan sampler cell for TABLESAMPLE
6477 // REPEATABLE (see run_single_table_scan). Aggregates
6478 // (`count(*) FROM t TABLESAMPLE …`) filter through this ctx too.
6479 let sample_cell: core::cell::Cell<Option<u64>> = core::cell::Cell::new(None);
6480 let ctx = self
6481 .ev_ctx(schema_cols, Some(alias))
6482 .with_sample_rng(&sample_cell);
6483 // v7.39 (round 657) — pre-sized. Pushing 500k pointers into a
6484 // `Vec::new()` walks the doubling chain 8, 16, … 262144, 524288,
6485 // and every abandoned buffer on the way stays resident: RSS is a
6486 // high-water mark, so the intermediates are paid for even though
6487 // they are freed. Round 656 measured the scan at 17 bytes/row
6488 // where the survivor list itself only needs 8.
6489 let mut filtered: Vec<&Row<'static>> = if stmt.where_.is_none() {
6490 Vec::with_capacity(table.rows().len())
6491 } else {
6492 // With a WHERE, the row count is an UPPER bound and reserving it
6493 // is the worse trade: `… WHERE id = 5` over 50M rows would take
6494 // 400 MB of pointers to hold one survivor. Let it grow.
6495 Vec::new()
6496 };
6497 // v6.2.6 — Memoize: per-query LRU cache for correlated
6498 // scalar subqueries. Fresh per row-loop entry so each
6499 // SELECT execution gets an isolated cache.
6500 let mut memo = memoize::MemoizeCache::new();
6501 // v7.37 (perf) — single-table aggregate's WHERE filter
6502 // pre-7.37 ran the slow tree-walker (`eval_expr_with_
6503 // correlated`) per row, even for subquery-free WHEREs that
6504 // the single-table SCAN path has compiled since v7.32
6505 // (perf knife D). The asymmetry meant a fold-to-filter
6506 // rewrite (joinfold) that swapped a JOIN for a single-table
6507 // aggregate over a compiled WHERE saw the tree-walker
6508 // instead — 25 k rows × `m.mailbox_id IN (25 lits)` cost
6509 // ~9 ms via the walker, vs ~1 ms via the compiled InSet
6510 // step. Compile once if eligible; fall back to the walker
6511 // for subquery-bearing or non-compilable WHEREs.
6512 let compiled_where: Option<eval::CompiledExpr> = stmt
6513 .where_
6514 .as_ref()
6515 .filter(|w| eval::fully_compilable(w))
6516 .map(|w| eval::compile_expr(w, &ctx));
6517 let mut eval_stack: Vec<Value<'static>> = Vec::new();
6518 let mut row_passes_where = |row: &Row<'static>,
6519 eval_stack: &mut Vec<Value<'static>>,
6520 memo: &mut memoize::MemoizeCache|
6521 -> Result<bool, EngineError> {
6522 match (&compiled_where, &stmt.where_) {
6523 (Some(cw), _) => {
6524 // v7.39 (round 479) — the predicate wants a bool, not a
6525 // Value. The owned entry ended in `Value::into_owned`
6526 // and the caller then dropped it, once per row; round
6527 // 478's profile put that pair above the comparison
6528 // itself.
6529 Ok(eval::compiled::eval_compiled_pred(
6530 cw,
6531 row,
6532 &ctx,
6533 eval_stack,
6534 ctx.mysql_dialect,
6535 )
6536 .map_err(EngineError::Eval)?)
6537 }
6538 (None, Some(w)) => {
6539 let cond = self.eval_expr_with_correlated(w, row, &ctx, cancel, Some(memo))?;
6540 Ok(crate::eval::predicate_is_true(
6541 &cond,
6542 "WHERE",
6543 ctx.mysql_dialect,
6544 )?)
6545 }
6546 (None, None) => Ok(true),
6547 }
6548 };
6549 if let Some(rows) = &indexed_rows {
6550 for cow in rows {
6551 let row = cow.as_ref();
6552 if !row_passes_where(row, &mut eval_stack, &mut memo)? {
6553 continue;
6554 }
6555 filtered.push(row);
6556 }
6557 }
6558 // v7.36 (cold-tier coverage) — single-table aggregate's
6559 // non-indexed full scan was hot-only and silently lost cold
6560 // rows on COUNT/SUM/etc. Materialise cold rows once into
6561 // `cold_rows_storage` (Vec<Row<'static>>) so the `filtered: Vec<&Row<'static>>`
6562 // shape stays unchanged; the cold rows live until the end of
6563 // the aggregate run.
6564 let cold_rows_storage = if indexed_rows.is_none() {
6565 self.iter_cold_rows_of_table(table)
6566 } else {
6567 Vec::new()
6568 };
6569 if indexed_rows.is_none() {
6570 // v7.37.15 (Phase C.3, step 2) — MVCC visibility gate for the
6571 // single-table aggregate full-scan path. Mirrors the gate on
6572 // `run_single_table_scan`: this is a user-query result path,
6573 // so under gate-on (`SPG_MVCC_INPLACE`) it must skip rows the
6574 // reader's snapshot cannot see (e.g. tombstoned versions),
6575 // otherwise COUNT/SUM/etc. would tally dead rows. A no-op
6576 // under the default gate-off: every hot row is frozen or
6577 // committed-and-alive, so `is_row_visible` returns true.
6578 // Cold-tier rows are frozen (visible) by definition — left
6579 // ungated, matching the plain-scan path.
6580 let scan_snapshot = self.current_snapshot();
6581 // v7.39 (pg_stat knife B) — this full-scan branch walks
6582 // headers directly (serial and sharded alike); count the
6583 // sequential scan here.
6584 table.note_seq_scan();
6585 // v7.39 (parallel-agg P2) — the visibility probe + WHERE
6586 // filter dominate the pre-aggregate wall time on big
6587 // scans (P1's ground truth: accumulation is only ~17%).
6588 // Shard THAT work when the host injected an executor and
6589 // the WHERE is compiled (the compiled evaluator is pure
6590 // over &row; the tree-walker fallback can hit correlated
6591 // subqueries and stays serial). Shards return surviving
6592 // ROW INDICES — &Row can't cross the Box<dyn Any>'s
6593 // 'static bound — and the main thread only dereferences.
6594 let n = table.row_count();
6595 let par = self.parallel_runner.0.as_deref().filter(|_| {
6596 n >= crate::PARALLEL_MIN_ROWS && (stmt.where_.is_none() || compiled_where.is_some())
6597 });
6598 if let Some(r) = par {
6599 let n_shards = (n / crate::PARALLEL_MIN_ROWS).clamp(2, 8);
6600 let chunk = n.div_ceil(n_shards);
6601 type ShardOut = Result<alloc::vec::Vec<usize>, EngineError>;
6602 let cw = &compiled_where;
6603 let snap_ref = &scan_snapshot;
6604 let results = r.run_shards(n_shards, &|s| {
6605 let lo = s * chunk;
6606 let hi = ((s + 1) * chunk).min(n);
6607 let mut keep: alloc::vec::Vec<usize> = alloc::vec::Vec::with_capacity(hi - lo);
6608 // EvalContext carries Cells (sampler / row counters)
6609 // and is !Sync — each shard builds its own from the
6610 // same Sync inputs. The compiled WHERE is gated to
6611 // the pure-scalar whitelist, which reads none of the
6612 // session state the engine-built ctx would add
6613 // (TABLESAMPLE's __tsm_fract is not whitelisted, so
6614 // sampled scans never take this branch).
6615 let shard_ctx = EvalContext::new(schema_cols, Some(alias));
6616 let mut stack: Vec<Value<'static>> = Vec::new();
6617 let out: ShardOut = (|| {
6618 for i in lo..hi {
6619 if !table.is_row_visible(i, snap_ref) {
6620 continue;
6621 }
6622 let row = &table.rows()[i];
6623 // v7.39 (round 480) — the parallel full-scan
6624 // shard is the path the aggregate benchmark
6625 // actually takes, and it was still on the OWNED
6626 // entry: round 480's profile attributed 68.7 %
6627 // of `drop_glue<Value>` to this closure, which
6628 // is why round 479's fix to the indexed path
6629 // barely moved the total.
6630 //
6631 // The `matches!(…, Value::Bool(true))` form was
6632 // also a narrower reading than the rest of the
6633 // engine uses — `predicate_is_true` is what
6634 // handles NULL and MySQL truthiness — so the
6635 // bool entry fixes the shape as well as the cost.
6636 let pass = match cw {
6637 Some(c) => eval::compiled::eval_compiled_pred(
6638 c,
6639 row,
6640 &shard_ctx,
6641 &mut stack,
6642 shard_ctx.mysql_dialect,
6643 )
6644 .map_err(EngineError::Eval)?,
6645 None => true,
6646 };
6647 if pass {
6648 keep.push(i);
6649 }
6650 }
6651 Ok(keep)
6652 })();
6653 alloc::boxed::Box::new(out)
6654 });
6655 // v7.39 (round 567) — `rows()` is a 32-way trie, so
6656 // indexing it is four dependent loads and a scan that
6657 // reads every row paid them every row. A profile of
6658 // `SELECT sum(id)` over 500k rows put 37.8% of the
6659 // connection thread's CPU on THIS ONE LINE. The cursor
6660 // holds the leaf, making that one descent per 32.
6661 let mut rows_cur = table.rows().run_cursor();
6662 for boxed in results {
6663 let shard = boxed
6664 .downcast::<ShardOut>()
6665 .expect("runner echoes the closure's box");
6666 for i in (*shard)? {
6667 if let Some(row) = rows_cur.get(i) {
6668 filtered.push(row);
6669 }
6670 }
6671 }
6672 } else {
6673 let mut rows_cur = table.rows().run_cursor();
6674 for i in 0..n {
6675 if !table.is_row_visible(i, &scan_snapshot) {
6676 continue;
6677 }
6678 let Some(row) = rows_cur.get(i) else { continue };
6679 if !row_passes_where(row, &mut eval_stack, &mut memo)? {
6680 continue;
6681 }
6682 filtered.push(row);
6683 }
6684 }
6685 for row in &cold_rows_storage {
6686 if !row_passes_where(row, &mut eval_stack, &mut memo)? {
6687 continue;
6688 }
6689 filtered.push(row);
6690 }
6691 }
6692 // v7.29 — a per-query memo so correlated scalar
6693 // subqueries batch-evaluate once (group map) instead of
6694 // executing per group.
6695 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
6696 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
6697 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
6698 .map_err(|err| match err {
6699 EngineError::Eval(ev) => ev,
6700 other => eval::EvalError::TypeMismatch {
6701 detail: alloc::format!("{other}"),
6702 },
6703 })
6704 };
6705 // v7.39 (round 656) — the plain relational scan. This collect() was
6706 // the measured defect: one 64-byte `RowRef` per surviving row to
6707 // wrap an 8-byte pointer `filtered` already holds. Scalar
6708 // aggregates measured ~81 bytes/row of working memory because of
6709 // it — 40 MB at 500k rows, 3.2 GB at 50M, for a query that returns
6710 // one number. `AggRows::Ptrs` reads the pointers directly.
6711 let agg = aggregate::run(
6712 stmt,
6713 crate::join::AggRows::Ptrs(&filtered),
6714 schema_cols,
6715 Some(alias),
6716 Some(&agg_correlated),
6717 self.parallel_runner.0.as_deref(),
6718 Some(self.active_catalog()),
6719 Some(self),
6720 )?;
6721 self.finish_agg_result(agg, stmt, cancel)
6722 }
6723
6724 /// Single-table scan + projection path: WHERE filter (compiled when
6725 /// subquery-free), ORDER BY keying, SRF expansion / projection, then
6726 /// sort + WITH TIES / DISTINCT / OFFSET-LIMIT.
6727 fn run_single_table_scan<'a>(
6728 &self,
6729 stmt: &SelectStatement,
6730 table: &'a spg_storage::Table,
6731 schema_cols: &'a [ColumnSchema],
6732 alias: &str,
6733 indexed_rows: Option<Vec<Cow<'a, Row<'static>>>>,
6734 cancel: CancelToken<'_>,
6735 ) -> Result<QueryResult, EngineError> {
6736 // v7.38 (read01 U15) — a fresh per-scan sampler cell for
6737 // `TABLESAMPLE … REPEATABLE(seed)`. Created before the ctx so the
6738 // deterministic `__tsm_fract(seed)` draws share one scan-local
6739 // state (isolated from the global random() PRNG); a fresh cell per
6740 // scan makes a repeat / rescan reproduce the same sample. Unused
6741 // and cheap when the query carries no sample.
6742 let sample_cell: core::cell::Cell<Option<u64>> = core::cell::Cell::new(None);
6743 let ctx = self
6744 .ev_ctx(schema_cols, Some(alias))
6745 .with_sample_rng(&sample_cell);
6746 let projection = build_projection(&stmt.items, schema_cols, alias, self.backslash_escapes)?;
6747 // v7.19 P5 — single-table SELECT path for SRF
6748 // `SELECT unnest(arr) FROM t` shape. Detect a top-level
6749 // unnest in the projection list. When present, the
6750 // per-row processor emits one output row per array
6751 // element (broadcasting non-SRF projections from the
6752 // same input row). Empty / NULL arrays emit zero rows
6753 // for that input — PG semantics.
6754 // v7.39 (read01 round 67) — every SRF in the target list, in lockstep.
6755 let srf_idxs = self.srf_target_idxs(&projection);
6756 let srf_position = srf_idxs.first().copied();
6757 // v7.39 (round 599) — the SRF analysis is per QUERY, not per row.
6758 let mut srf_plan = if srf_position.is_some() {
6759 Some(build_srf_plan(self, &projection, &srf_idxs, &ctx)?)
6760 } else {
6761 None
6762 };
6763
6764 // Materialise the filter pass into `(order_key, projected_row)`
6765 // tuples. The order key is `None` when there's no ORDER BY clause.
6766 let mut tagged: Vec<(Vec<OrderKey>, Row<'static>)> = Vec::new();
6767 // v7.33 (C1, ceiling-first/never-die) — charge each accumulated
6768 // output row to the per-query byte budget as it is built, so a
6769 // fat single-table scan / sort REJECTS with QueryBytesExceeded
6770 // at ~the ceiling instead of materialising the whole table and
6771 // only noticing at the final enforce_row_limit check. Without
6772 // this, N concurrent fat scans peak at N×table and OOM the host.
6773 // `max_query_bytes = None` (the embedded default) = no ceiling,
6774 // so existing unbudgeted behaviour is byte-identical.
6775 let mut budget = ByteBudget::new(self.max_query_bytes);
6776 // v6.2.6 — Memoize per-row WHERE eval shares one cache.
6777 let mut memo = memoize::MemoizeCache::new();
6778 // v7.32 (perf knife D) — subquery-free WHERE compiles once;
6779 // the row loop then runs a flat step program instead of a
6780 // tree interpretation per row.
6781 let compiled_where: Option<eval::CompiledExpr> = stmt
6782 .where_
6783 .as_ref()
6784 .filter(|w| eval::fully_compilable(w))
6785 .map(|w| eval::compile_expr(w, &ctx));
6786 let mut eval_stack: Vec<Value<'static>> = Vec::new();
6787 // v7.37.x (docker-fair SCALARSQ attack) — pre-analyse every
6788 // SELECT-item scalar subquery for the PK-probe fast path. The
6789 // analysis (gate checks + catalog lookups) takes ~500 ns; doing
6790 // it once per query instead of once per row × 100 rows saves
6791 // ~50 µs and lets the per-row evaluation reduce to a single
6792 // index probe + outer-column read.
6793 let scalarsq_fast: Vec<Option<crate::ScalarPkProbeFastPath>> = projection
6794 .iter()
6795 .map(|p| {
6796 if let Expr::ScalarSubquery(inner) = &p.expr {
6797 self.analyse_scalar_count_pk_eq_probe(inner, schema_cols, alias)
6798 } else {
6799 None
6800 }
6801 })
6802 .collect();
6803 let any_scalarsq_fast = scalarsq_fast.iter().any(Option::is_some);
6804 // v7.39 (round 487) — a projection item that is a bare column
6805 // reference binds its position ONCE per query.
6806 //
6807 // Per row it used to walk `eval_expr_with_correlated` (a memo
6808 // lookup for "does this have a subquery", then an un-memoised
6809 // `expr_may_use_in_set` tree walk), then `eval_expr`'s dispatch,
6810 // then `resolve_column`, which finds the column by scanning the
6811 // schema and comparing NAMES. On `SELECT g FROM h` that chain was
6812 // 19 % of self time for what is ultimately one cell read.
6813 //
6814 // `compile_column_pos` is the Step VM's resolver, already
6815 // `pub(crate)` and already reused by the aggregate's bind-once
6816 // path: it mirrors `resolve_column`'s happy layers and returns
6817 // None for anything that would reach an error, an ambiguity, or a
6818 // miss, so those still go the interpreter's way and keep its
6819 // exact message. A composite column is excluded for the same
6820 // reason `compile_into` excludes it — it must be rehydrated from
6821 // stored JSON, which is not a cell read.
6822 let proj_direct = bind_direct_columns(&projection, &ctx);
6823 let any_proj_direct = proj_direct.iter().any(Option::is_some);
6824 // v7.39 (round 605) — a projection item that cannot depend on the row
6825 // is evaluated once. `SELECT ('{"a":1}')::JSONB FROM j` cost TEN
6826 // allocations a row against one for a plain column, `'abc' || 'def'`
6827 // six and `upper('abc')` five, all of them producing the same value
6828 // 50,000 times. An item that fails to evaluate is left alone, so its
6829 // error still comes from the row loop in the interpreter's wording.
6830 let proj_const: Vec<Option<Value<'static>>> = projection
6831 .iter()
6832 .map(|p| crate::eval::compiled::constant_projection_value(&p.expr, &ctx))
6833 .collect();
6834 let any_proj_const = proj_const.iter().any(Option::is_some);
6835 crate::bump_counter!(crate::select::SCAN_PATH_ENTERED);
6836 // v7.39 (read01 round 80) — positional ORDER BY over a WILDCARD
6837 // projection. Statement prep (`resolve_order_by_position`) can only map
6838 // `ORDER BY 1` onto the first SELECT item when that item is an
6839 // expression; a `*` is not one, so the literal survived to here and was
6840 // evaluated as the CONSTANT 1 — the same key for every row, i.e. no sort
6841 // at all. The parser rewrites `SELECT unnest(a) x` into
6842 // `SELECT * FROM unnest(a) x`, so that innocuous-looking shape landed
6843 // exactly here: `SELECT unnest(ARRAY['B','a','A','b']) ORDER BY 1` came
6844 // back in input order. The projection is built by now, so the Nth output
6845 // column is known — resolve against it.
6846 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
6847 // v7.39 (round 600) — the ORDER BY of an SRF query is decided on the
6848 // EXPANDED rows, so a key naming a select-list item reads that item.
6849 let srf_order_cols: Vec<Option<usize>> = if srf_position.is_some() {
6850 srf_order_output_cols(&order_by, &projection)
6851 } else {
6852 Vec::new()
6853 };
6854 let srf_key_bound: Vec<Option<usize>> = (0..order_by.len()).map(Some).collect();
6855 // v7.37.x (docker-fair SCALARSQ attack) — early-limit gate for
6856 // the no-ORDER-BY-no-DISTINCT-no-TIES-no-SRF-no-WHERE shape.
6857 // Hoisted above the closure so the projection-eval path can
6858 // gate `memo` passing on it: the SELECT-item correlated-scalar
6859 // batch path scans the FULL inner table once (~5 ms for 12.5 k
6860 // rows) and is only a win when N outer rows is large; for small
6861 // LIMITed shapes a per-row PK seek (~5 µs × 100 = 500 µs) wins.
6862 let early_cap: Option<usize> = if order_by.is_empty()
6863 && !stmt.distinct
6864 && !stmt.limit_with_ties
6865 && srf_position.is_none()
6866 && stmt.where_.is_none()
6867 {
6868 stmt.limit_literal()
6869 .map(|n| n.saturating_add(stmt.offset_literal().unwrap_or(0)) as usize)
6870 } else {
6871 None
6872 };
6873 // v7.38 (read01 B8) — streaming top-N budget. For `ORDER BY …
6874 // LIMIT k` (no DISTINCT / WITH TIES / SRF, and not forced to
6875 // full-sort by the test gate) keep only the running top-`keep`
6876 // rows in memory instead of materialising every projected row,
6877 // so a `… ORDER BY col LIMIT 10` over a huge table is O(keep)
6878 // space, not O(rows). `None` = accumulate everything (the prior
6879 // behaviour). The final `partial_sort_tagged(keep)` below still
6880 // runs and produces the identical rows.
6881 // v7.39 (round 683) — the declared collation for each ORDER BY
6882 // position, resolved once and carried beside `descs` for the same
6883 // reason `descs` is carried: it is per key position, not per row.
6884 let order_colls = crate::orderby::order_by_collations(&order_by, &ctx)?;
6885 let topk_stream: Option<(usize, Vec<bool>)> = if !order_by.is_empty()
6886 && !stmt.distinct
6887 && !stmt.limit_with_ties
6888 && srf_position.is_none()
6889 && !self.env_cfg().disable_topk
6890 {
6891 stmt.limit_literal().and_then(|l| {
6892 let keep = (l as usize).saturating_add(stmt.offset_literal().unwrap_or(0) as usize);
6893 (keep >= 1).then(|| (keep, order_by.iter().map(|o| o.desc).collect()))
6894 })
6895 } else {
6896 None
6897 };
6898 // v7.37.16 — streaming DISTINCT seen-set: norm-hash → indices of
6899 // kept rows in `tagged`. Probing on the PROJECTED row as soon as
6900 // it is built means a duplicate costs neither a build_order_keys
6901 // eval (the dominant per-row cost of `DISTINCT … ORDER BY`) nor
6902 // a tagged slot, and the sort below runs over u survivors, not
6903 // n input rows — PG's hash-distinct-then-sort plan shape.
6904 let mut seen_distinct: hashbrown::HashMap<u64, crate::distinct::DistinctBucket> =
6905 hashbrown::HashMap::new();
6906 let distinct_hb = hashbrown::DefaultHashBuilder::default();
6907 // v7.39 (round 485) — one projection buffer for the whole scan
6908 // rather than a fresh `Vec` per input row. A row that survives
6909 // the DISTINCT probe takes the buffer with it (`mem::take`) and
6910 // the next row allocates a new one; a row that duplicates an
6911 // earlier one leaves the buffer — and its capacity — in place.
6912 // The round-485 counter says 49 900 of `distinct_proj`'s 50 000
6913 // projected rows are duplicates, so that is 49 900 allocate /
6914 // free pairs the scan no longer performs. Shapes where every row
6915 // survives (plain projection, `DISTINCT` over a unique column)
6916 // allocate exactly as often as before.
6917 let mut proj_buf: Vec<Value<'static>> = Vec::new();
6918 // v7.39 (round 571) — buffers handed back by the top-N trim.
6919 // Round 485 made the scan share ONE projection buffer, but a
6920 // surviving row takes it (`mem::take`) and without DISTINCT
6921 // almost every row survives, so the next one starts from zero
6922 // capacity and allocates. The trim drops `keep` rows at a time
6923 // and their buffers come back here instead of being freed.
6924 let mut proj_pool: Vec<Vec<Value<'static>>> = Vec::new();
6925 let mut key_pool: Vec<Vec<crate::orderby::OrderKey>> = Vec::new();
6926 // v7.39 (round 581) — the worst row the accumulator is currently
6927 // keeping. Anything that loses to it cannot reach the answer, so
6928 // it is dropped before its projection is ever built.
6929 let mut topk_boundary: Option<Vec<crate::orderby::OrderKey>> = None;
6930 // v7.39 (round 582) — resolve each ORDER BY column once, not
6931 // once per row. See `order_by_bound_positions`.
6932 let order_bound =
6933 crate::orderby::order_by_bound_positions(&order_by, schema_cols, Some(alias));
6934 // v7.39 (round 581) — and it stops asking when the answer is
6935 // always "keep".
6936 //
6937 // The check earns its place only on rows it rejects. Over
6938 // ascending ids, `ORDER BY id DESC` never rejects one — every
6939 // row beats the current worst — so the comparison is pure
6940 // overhead there, measured at +5.5% in three batches out of
6941 // three. After a window of rows it looks at what it has
6942 // actually rejected and switches itself off if the shape is not
6943 // paying. The answers do not depend on it either way.
6944 const BOUNDARY_WINDOW: u32 = 8192;
6945 let mut boundary_checks: u32 = 0;
6946 let mut boundary_rejects: u32 = 0;
6947 let mut boundary_check_on = true;
6948 // Inline the per-row work in a closure so the indexed and full-
6949 // scan branches share the body.
6950 let mut process_row = |row: &Row<'static>, loop_idx: usize| -> Result<(), EngineError> {
6951 if loop_idx.is_multiple_of(256) {
6952 cancel.check()?;
6953 }
6954 if let Some(cw) = &compiled_where {
6955 let cond = eval::eval_compiled(cw, row, &ctx, &mut eval_stack)
6956 .map_err(EngineError::Eval)?;
6957 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
6958 return Ok(());
6959 }
6960 } else if let Some(where_expr) = &stmt.where_ {
6961 let cond =
6962 self.eval_expr_with_correlated(where_expr, row, &ctx, cancel, Some(&mut memo))?;
6963 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
6964 return Ok(());
6965 }
6966 }
6967 // Under DISTINCT the keys are built AFTER the dup probe
6968 // (survivors only); the non-distinct order is unchanged.
6969 // v7.39 (round 600) — an SRF query's keys are built per EXPANDED
6970 // row further down, and building them here would evaluate the
6971 // ORDER BY against the INPUT row: a key naming the SRF's own
6972 // output became a scalar call to it, which is where
6973 // "function unnest(integer[]) does not exist" came from.
6974 let order_keys = if order_by.is_empty() || stmt.distinct || srf_position.is_some() {
6975 Vec::new()
6976 } else {
6977 let mut buf = key_pool.pop().unwrap_or_default();
6978 crate::orderby::build_order_keys_bound(
6979 &order_by,
6980 &order_bound,
6981 row,
6982 &ctx,
6983 &mut buf,
6984 )?;
6985 // v7.39 (round 581) — reject before projecting.
6986 //
6987 // `ORDER BY g DESC, id DESC LIMIT 10` over 500k rows with
6988 // 50 distinct `g` decides nearly every row on the FIRST
6989 // key, and PG answers it FASTER than the single-key form
6990 // (7.4 ms against 10.4) because a rejected row costs it
6991 // one comparison. SPG built both keys AND the projected
6992 // row for all 500k before throwing them away. The keys
6993 // are needed to compare; the projection is not.
6994 if boundary_check_on
6995 && let Some((_, descs)) = &topk_stream
6996 && let Some(b) = &topk_boundary
6997 {
6998 boundary_checks += 1;
6999 let loses = crate::orderby::cmp_multi_key_in(&buf, b, descs, &order_colls)
7000 == core::cmp::Ordering::Greater;
7001 if loses {
7002 boundary_rejects += 1;
7003 }
7004 if boundary_checks == BOUNDARY_WINDOW {
7005 // Keep asking only if it has been rejecting at
7006 // least a quarter of what it saw.
7007 boundary_check_on = boundary_rejects.saturating_mul(4) >= boundary_checks;
7008 }
7009 if loses {
7010 buf.clear();
7011 key_pool.push(buf);
7012 return Ok(());
7013 }
7014 }
7015 buf
7016 };
7017 if srf_position.is_some() {
7018 let plan = srf_plan.as_mut().expect("srf_position implies a plan");
7019 for out in expand_srf_row_with(self, plan, &projection, row, &ctx)? {
7020 if stmt.distinct {
7021 let bucket = seen_distinct
7022 .entry(norm_hash_row(&out, &distinct_hb, ctx.mysql_dialect))
7023 .or_default();
7024 if bucket
7025 .iter()
7026 .any(|i| row_eq_norm(&tagged[i].1, &out, ctx.mysql_dialect))
7027 {
7028 continue;
7029 }
7030 bucket.push(tagged.len());
7031 }
7032 budget.charge(approx_row_bytes(&out))?;
7033 // The keys come from THIS expanded row: a key naming a
7034 // select-list item reads its value, anything else is
7035 // still evaluated against the input row.
7036 let keys = if order_by.is_empty() {
7037 Vec::new()
7038 } else {
7039 let mut kv: Vec<Value<'static>> = Vec::with_capacity(order_by.len());
7040 for (k, ob) in order_by.iter().enumerate() {
7041 kv.push(match srf_order_cols.get(k).copied().flatten() {
7042 Some(p) => out.values.get(p).cloned().unwrap_or(Value::Null),
7043 None => eval::eval_expr(&ob.expr, row, &ctx)
7044 .map_err(EngineError::Eval)?,
7045 });
7046 }
7047 // Packed by the same code every other ORDER BY uses,
7048 // so DESC / NULLS FIRST / the MySQL rule are not
7049 // restated here.
7050 let key_row = Row::new(kv);
7051 let mut buf = Vec::new();
7052 crate::orderby::build_order_keys_bound(
7053 &order_by,
7054 &srf_key_bound,
7055 &key_row,
7056 &ctx,
7057 &mut buf,
7058 )?;
7059 buf
7060 };
7061 tagged.push((keys, out));
7062 }
7063 } else {
7064 let values = &mut proj_buf;
7065 values.clear();
7066 values.reserve(projection.len());
7067 for (i, p) in projection.iter().enumerate() {
7068 // v7.37.x (docker-fair SCALARSQ attack) — pre-
7069 // analysed PK-probe fast path. The per-row work is
7070 // a read of outer.col from the row plus an index
7071 // probe — no Expr clone, no walker, no
7072 // `eval_expr_with_correlated` framework.
7073 if any_scalarsq_fast && let Some(fp) = &scalarsq_fast[i] {
7074 values.push(self.probe_with_pk_fast_path(fp, row));
7075 continue;
7076 }
7077 // v7.39 (round 605) — the same value every row.
7078 if any_proj_const && let Some(v) = &proj_const[i] {
7079 values.push(v.clone());
7080 continue;
7081 }
7082 // v7.39 (round 487) — bound column: read the cell.
7083 // This is `rehydrate_cell`'s body for a non-composite
7084 // column, which is what the whole chain below reduces
7085 // to once the name has been resolved.
7086 if any_proj_direct && let Some(pos) = proj_direct[i] {
7087 crate::bump_counter!(crate::select::PROJ_DIRECT_FIRE);
7088 values.push(row.values[pos].clone().into_owned());
7089 continue;
7090 }
7091 // v7.24 (round-16 B) — correlated-aware.
7092 // v7.37.x (docker-fair SCALARSQ attack) — share the
7093 // per-row memo with projection. Required for the
7094 // batch-evaluated correlated-scalar path to fire on
7095 // SELECT-item scalar subqueries; otherwise each row
7096 // re-executes the inner.
7097 //
7098 // Skip the memo when the outer row count is small
7099 // (early-limited): the batch path scans the FULL
7100 // inner table to build a GroupMap (~5 ms for a
7101 // 12.5 k-row inner), while per-row execution with a
7102 // PK index seek is ~5 µs per call — much cheaper for
7103 // N ≤ ~1000 outer rows.
7104 let pass_memo = early_cap.is_none_or(|cap| cap > 1000);
7105 let memo_arg = if pass_memo { Some(&mut memo) } else { None };
7106 values.push(
7107 self.eval_expr_with_correlated(&p.expr, row, &ctx, cancel, memo_arg)?,
7108 );
7109 }
7110 crate::bump_counter!(crate::select::PROJ_ROW_BUILT);
7111 if stmt.distinct {
7112 let bucket = seen_distinct
7113 .entry(norm_hash_values(&proj_buf, &distinct_hb, ctx.mysql_dialect))
7114 .or_default();
7115 if bucket
7116 .iter()
7117 .any(|i| values_eq_norm(&tagged[i].1.values, &proj_buf, ctx.mysql_dialect))
7118 {
7119 crate::bump_counter!(crate::select::DISTINCT_DUP_DROPPED);
7120 return Ok(());
7121 }
7122 bucket.push(tagged.len());
7123 }
7124 let out = Row::new(core::mem::replace(
7125 &mut proj_buf,
7126 proj_pool.pop().unwrap_or_default(),
7127 ));
7128 let order_keys = if stmt.distinct && !order_by.is_empty() {
7129 build_order_keys(&order_by, row, &ctx)?
7130 } else {
7131 order_keys
7132 };
7133 budget.charge(approx_row_bytes(&out))?;
7134 tagged.push((order_keys, out));
7135 }
7136 // Streaming top-N: bound the accumulator to O(keep) rows.
7137 if let Some((k, descs)) = &topk_stream {
7138 crate::orderby::topk_trim_recycling(
7139 &mut tagged,
7140 *k,
7141 descs,
7142 &mut proj_pool,
7143 &mut key_pool,
7144 &mut topk_boundary,
7145 );
7146 }
7147 Ok(())
7148 };
7149 // v7.37.15 (Phase C.3, step 2) — MVCC visibility gate for the
7150 // load-bearing full-scan path. This is the primary single-table
7151 // executor; pre-C.3 it read every hot-tier row raw. Once C.3's
7152 // in-place writers retain dead/old versions, an ungated scan
7153 // here would return them, so the gate must land BEFORE the
7154 // writers flip (see the plan's activation-order rule). A no-op
7155 // today: every hot row is frozen or committed-and-alive under
7156 // the reader's snapshot, so `is_row_visible` returns true for
7157 // all of them (verified by the full e2e suite staying green).
7158 let scan_snapshot = self.current_snapshot();
7159 let mut emitted: usize = 0;
7160 if let Some(rows) = &indexed_rows {
7161 for (loop_idx, cow) in rows.iter().enumerate() {
7162 if let Some(cap) = early_cap
7163 && emitted >= cap
7164 {
7165 break;
7166 }
7167 process_row(cow.as_ref(), loop_idx)?;
7168 emitted = emitted.saturating_add(1);
7169 }
7170 } else {
7171 // v7.39 (round 570) — the row store is a 32-way trie, so
7172 // indexing it is four dependent loads. Round 567 measured
7173 // -18% on the aggregate scan from holding the leaf between
7174 // rows; this is the same loop for the projecting scan.
7175 let mut rows_cur = table.rows().run_cursor();
7176 for i in 0..table.row_count() {
7177 if let Some(cap) = early_cap
7178 && emitted >= cap
7179 {
7180 break;
7181 }
7182 // Skip rows this snapshot cannot see (invisible rows do
7183 // not count toward the LIMIT).
7184 if !table.is_row_visible(i, &scan_snapshot) {
7185 continue;
7186 }
7187 let Some(row) = rows_cur.get(i) else { continue };
7188 process_row(row, i)?;
7189 emitted = emitted.saturating_add(1);
7190 }
7191 // v7.35.1 (mailrs prod #6 follow-up) — fold cold-tier
7192 // rows into the same loop. The full-scan path here is the
7193 // load-bearing single-table SELECT executor, and pre-
7194 // 7.35.1 it only walked `table.rows()` (hot), so any
7195 // `SELECT … FROM t` against a table with cold segments
7196 // silently returned a subset.
7197 let cold_rows = self.iter_cold_rows_of_table(table);
7198 for (offset, row) in cold_rows.iter().enumerate() {
7199 if let Some(cap) = early_cap
7200 && emitted >= cap
7201 {
7202 break;
7203 }
7204 process_row(row, table.row_count() + offset)?;
7205 emitted = emitted.saturating_add(1);
7206 }
7207 }
7208
7209 // (DISTINCT already de-duped STREAMING inside process_row, so the
7210 // sort below only sees the u survivors and the partial-sort
7211 // budget applies to DISTINCT too.)
7212 if !order_by.is_empty() {
7213 // Partial-sort fast path: when LIMIT is small relative to
7214 // the row count, select_nth_unstable + sort just the
7215 // prefix is O(n + k log k) instead of O(n log n).
7216 // WITH TIES needs the full sort so the tie extension can
7217 // scan past `limit` to find rows that share the last-kept
7218 // row's key.
7219 let keep = if stmt.limit_with_ties
7220 // v7.38 元机制 D acceptor — `SPG_TEST_DISABLE_TOPK=1`
7221 // forces the full-sort fallback by suppressing the
7222 // partial-sort `keep` budget. See
7223 // `xtests/sigil/test-mode-gucs.md`.
7224 || self.env_cfg().disable_topk
7225 {
7226 None
7227 } else {
7228 stmt.limit_literal()
7229 .map(|l| l as usize + stmt.offset_literal().map_or(0, |o| o as usize))
7230 };
7231 let descs: Vec<bool> = order_by.iter().map(|o| o.desc).collect();
7232 crate::orderby::partial_sort_tagged_in(&mut tagged, keep, &descs, &order_colls);
7233 }
7234
7235 // v7.17.0 Phase 3.P0-49 — `FETCH FIRST … WITH TIES` extends
7236 // past the truncated tail through every row that shares the
7237 // last-kept row's ORDER BY key. The tie check uses the
7238 // already-computed `(order_keys, row)` pairs so it matches
7239 // the sort comparator exactly. DISTINCT + WITH TIES falls
7240 // through to the no-ties path (PG also disallows their
7241 // combination; SPG silently drops the tie extension here so
7242 // the customer doesn't see a hard error mid-query — the
7243 // user-visible result is still correct, just narrower).
7244 let output_rows: Vec<Row<'static>> = if stmt.limit_with_ties && !stmt.distinct {
7245 apply_offset_and_limit_tagged(
7246 &mut tagged,
7247 stmt.offset_literal(),
7248 stmt.limit_literal(),
7249 true,
7250 );
7251 tagged.into_iter().map(|(_, r)| r).collect()
7252 } else {
7253 // DISTINCT already de-duped pre-sort above.
7254 let mut output_rows: Vec<Row<'static>> = tagged.into_iter().map(|(_, r)| r).collect();
7255 apply_offset_and_limit(
7256 &mut output_rows,
7257 stmt.offset_literal(),
7258 stmt.limit_literal(),
7259 );
7260 output_rows
7261 };
7262
7263 let columns: Vec<ColumnSchema> = projection
7264 .into_iter()
7265 .map(|p| {
7266 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
7267 c.user_enum_type = p.user_enum_type;
7268 c.collation_name = p.collation_name;
7269 c.mysql_fsp = p.mysql_fsp;
7270 c
7271 })
7272 .collect();
7273
7274 Ok(QueryResult::Rows {
7275 columns,
7276 rows: output_rows,
7277 })
7278 }
7279
7280 /// v7.31 (perf — PG lesson #1): shared aggregate finisher. Apply
7281 /// OFFSET/LIMIT first, then evaluate the deferred subquery-bearing
7282 /// select items for the surviving rows only — PG's Result-above-
7283 /// Limit shape, where SubPlan loops equal the OUTPUT row count
7284 /// (50) instead of the group count (24k).
7285 fn finish_agg_result(
7286 &self,
7287 mut agg: aggregate::AggResult,
7288 stmt: &SelectStatement,
7289 cancel: CancelToken<'_>,
7290 ) -> Result<QueryResult, EngineError> {
7291 apply_offset_and_limit(&mut agg.rows, stmt.offset_literal(), stmt.limit_literal());
7292 if !agg.deferred.is_empty() {
7293 apply_offset_and_limit(
7294 &mut agg.synth_rows,
7295 stmt.offset_literal(),
7296 stmt.limit_literal(),
7297 );
7298 let ctx = EvalContext::new(&agg.synth_schema, None);
7299 let mut memo = memoize::MemoizeCache::default();
7300 // v7.32 (architecture v2 P3) — keyed index-probe seeding.
7301 // Deferred subqueries are referenced only by surviving
7302 // select-list rows (≤ LIMIT), so their correlation keys are
7303 // exactly the ≤LIMIT group keys in `synth_rows`. Pre-build
7304 // each batchable subquery's group map over just those keys
7305 // via per-key index seek; the per-row splice loop below then
7306 // reuses the seeded map. A join-shaped or un-indexed inner
7307 // falls through to the all-keys batch inside the call (built
7308 // eagerly here instead of lazily on row 0 — same cost), so
7309 // it still pays the full scan, never the 715 ms per-row
7310 // direct eval; its index-nested-loop probe is the next
7311 // knife. Genuinely non-batchable shapes return None and are
7312 // left unseeded for the loop's per-row resolver, as before.
7313 for (_, expr) in &agg.deferred {
7314 let mut subs: Vec<&SelectStatement> = Vec::new();
7315 collect_scalar_subqueries(expr, &mut subs);
7316 for sub in subs {
7317 let repr = alloc::format!("{sub}");
7318 if memo.group_maps.contains_key(&repr) {
7319 continue;
7320 }
7321 if let Some(gm) = self.try_batch_correlated_scalar(
7322 sub,
7323 Some((&agg.synth_rows, &ctx)),
7324 cancel,
7325 )? {
7326 memo.group_maps.insert(repr, Some(alloc::rc::Rc::new(gm)));
7327 }
7328 }
7329 }
7330 for (ri, srow) in agg.synth_rows.iter().enumerate() {
7331 cancel.check()?;
7332 for (col, expr) in &agg.deferred {
7333 let v =
7334 self.eval_expr_with_correlated(expr, srow, &ctx, cancel, Some(&mut memo))?;
7335 if let Some(cell) = agg.rows[ri].values.get_mut(*col) {
7336 *cell = v;
7337 }
7338 }
7339 }
7340 }
7341 Ok(QueryResult::Rows {
7342 columns: agg.columns,
7343 rows: agg.rows,
7344 })
7345 }
7346
7347 /// v7.37 — streaming projection for the joined-non-aggregate
7348 /// shape (multi-table FROM, all projection items bound, no
7349 /// ORDER BY / DISTINCT / GROUP BY / HAVING / LIMIT / OFFSET /
7350 /// UNION). Walks the deferred join survivors and emits
7351 /// `&[&Value]` borrowed straight out of the source tables — no
7352 /// `.cloned()`, no `Vec<Row<'static>>`. Skips the 25 k × 3-TEXT clone tax
7353 /// on the mailrs `PROJ` shape (about 4 ms saved).
7354 ///
7355 /// Returns `Ok(None)` when the shape doesn't qualify; the caller
7356 /// then falls back to the materialising path.
7357 /// v7.37 (round 831) — stream a joinless SELECT straight off the
7358 /// stored table, one row at a time, without ever building a row set.
7359 ///
7360 /// Returns `Ok(None)` for anything this cannot serve, and the caller
7361 /// falls through to the deferred-join path exactly as before: a
7362 /// missing table, or a cold tier whose hydration the fallback handles.
7363 /// Sort a single-table scan through the external sorter, so the
7364 /// answer's size is bounded by `work_mem` and not by the input.
7365 ///
7366 /// Sorting held every row twice — the scan's `Vec<Row>` and the
7367 /// sort's `Vec<(keys, Row)>` beside it — with nothing bounding
7368 /// either: 807 MB at 400k rows, whatever `work_mem` said. A large
7369 /// enough ORDER BY took the server down, which is a liveness
7370 /// problem before it is a performance one.
7371 ///
7372 /// A SEPARATE walk rather than a change to `run_single_table_scan`,
7373 /// following what round 831 did for the joinless shape. That
7374 /// function is 552 lines whose projection loop is entangled with
7375 /// DISTINCT (which indexes back into the tagged vector) and with
7376 /// streaming top-N (whose boundary moves as the scan runs); both
7377 /// assume the projection has already happened when a row is
7378 /// pushed, which is exactly what spilling has to defer. Two earlier
7379 /// attempts tried to rework that loop and were reverted. Here the
7380 /// existing path is untouched and this one only claims shapes it
7381 /// can serve, so a decline costs nothing.
7382 ///
7383 /// Records are SOURCE rows, not projected ones: `finish` re-derives
7384 /// keys from what it decodes, and an ORDER BY key need not be in
7385 /// the projection — `SELECT pad FROM big ORDER BY id` (round 835).
7386 fn try_spill_sorted_scan(
7387 &self,
7388 stmt: &SelectStatement,
7389 from: &FromClause,
7390 cancel: CancelToken<'_>,
7391 ) -> Result<Option<QueryResult>, EngineError> {
7392 // Shapes this walk does not serve. Each one either needs the
7393 // whole tagged vector addressable (DISTINCT probes back into
7394 // it, WITH TIES re-reads its tail) or is already bounded
7395 // without spilling (a LIMIT makes the partial sort O(keep)).
7396 if !self.can_spill()
7397 || stmt.order_by.is_empty()
7398 || stmt.distinct
7399 || stmt.limit_with_ties
7400 || stmt.limit_literal().is_some()
7401 || !from.joins.is_empty()
7402 || from.primary.lateral_subquery.is_some()
7403 || from.primary.unnest_expr.is_some()
7404 || from.primary.generate_series_args.is_some()
7405 || select_has_window(stmt)
7406 {
7407 return Ok(None);
7408 }
7409 // A parent's rows are its children's. These walks scan the named
7410 // relation alone, so a partitioned or inherited parent comes back
7411 // short — and silently: the corpus caught `SELECT id FROM pr
7412 // ORDER BY id` and `SELECT k FROM pl ORDER BY k` returning the
7413 // parent's own rows instead of the partitions'. `ONLY` is exactly
7414 // the case that does not fan out, so it stays, which is the test
7415 // the FROM-clause fan-out itself makes.
7416 if !from.primary.only
7417 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
7418 {
7419 return Ok(None);
7420 }
7421 let Some(table) = self.active_catalog().get(&from.primary.name) else {
7422 return Ok(None);
7423 };
7424 // Cold-tier rows live outside `rows()`; this walk would drop
7425 // them silently, the same reason round 831's walk declines.
7426 if table.has_cold_rows_fast() {
7427 return Ok(None);
7428 }
7429
7430 let alias = from
7431 .primary
7432 .alias
7433 .as_deref()
7434 .unwrap_or(from.primary.name.as_str());
7435 let cols = table.schema().columns.clone();
7436 let sess = self.dml_session();
7437 let ctx = EvalContext::new(&cols, Some(alias))
7438 .with_catalog(self.active_catalog())
7439 .with_session(&sess);
7440 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
7441 let order_by = stmt.order_by.clone();
7442 // The same one-shot resolution the general path does (round
7443 // 582): each ORDER BY column is bound once, not once per row.
7444 let order_bound = crate::orderby::order_by_bound_positions(&order_by, &cols, Some(alias));
7445 let descs: Vec<bool> = order_by.iter().map(|o| o.desc).collect();
7446 // Resolved BEFORE the scan, because it now decides what the sort
7447 // STORES and not just what it decodes (round 995).
7448 let needed = Self::sort_record_columns_needed(&stmt.items, &order_bound, cols.len(), &ctx);
7449
7450 let mut sorter = crate::extsort::ExternalSorter::new(
7451 self.temp_run_factory,
7452 self.session_work_mem_bytes(),
7453 cols.clone(),
7454 &descs,
7455 )
7456 .with_stats(&self.spill_stats)
7457 .with_pruned(&needed);
7458 let snapshot = self.current_snapshot();
7459 // One key buffer for the whole scan: `push` drains it and leaves
7460 // the capacity behind.
7461 let mut keys: Vec<OrderKey> = Vec::new();
7462 // r1024 — compile the predicate once for the scan.
7463 //
7464 // These two sorted-spill scans are the paths a single-table SELECT
7465 // with an ORDER BY takes, and they were the last row-returning ones
7466 // still walking the expression tree per row. r1023 did the
7467 // no-ORDER-BY sibling; the sweep's two remaining losing cells are
7468 // exactly this shape.
7469 //
7470 // Found from the profile's CALL TREE rather than its leaves. The
7471 // leaves say what is expensive — `eval_expr` 320, `apply_binary`
7472 // 261, `mod_op` 178 — and two attempts at reasoning out which
7473 // function asked for it were both wrong. The tree names the caller
7474 // chain, and it named this one.
7475 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
7476 .where_
7477 .as_ref()
7478 .filter(|w| crate::eval::fully_compilable(w))
7479 .map(|w| crate::eval::compile_expr(w, &ctx));
7480 let mut eval_stack: Vec<Value<'static>> = Vec::new();
7481 for (i, row) in table.scan_visible_from(0, &snapshot) {
7482 if i.is_multiple_of(256) {
7483 cancel.check()?;
7484 }
7485 if let Some(c) = &compiled_where {
7486 if !crate::eval::compiled::eval_compiled_pred(
7487 c,
7488 row,
7489 &ctx,
7490 &mut eval_stack,
7491 ctx.mysql_dialect,
7492 )? {
7493 continue;
7494 }
7495 } else if let Some(w) = &stmt.where_ {
7496 let cond = crate::eval::eval_expr(w, row, &ctx).map_err(EngineError::Eval)?;
7497 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
7498 continue;
7499 }
7500 }
7501 keys.clear();
7502 crate::orderby::build_order_keys_bound(&order_by, &order_bound, row, &ctx, &mut keys)?;
7503 sorter.push(&mut keys, row)?;
7504 }
7505
7506 let key_ctx = &ctx;
7507 let rows = sorter.finish(
7508 |src, buf| {
7509 crate::orderby::build_order_keys_bound(&order_by, &order_bound, src, key_ctx, buf)
7510 },
7511 |src| {
7512 let mut values = Vec::with_capacity(projection.len());
7513 for p in &projection {
7514 values.push(
7515 crate::eval::eval_expr(&p.expr, src, key_ctx).map_err(EngineError::Eval)?,
7516 );
7517 }
7518 Ok(Row::new(values))
7519 },
7520 )?;
7521
7522 let columns: Vec<ColumnSchema> = projection
7523 .iter()
7524 .map(|p| {
7525 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
7526 c.user_enum_type = p.user_enum_type.clone();
7527 c.mysql_fsp = p.mysql_fsp;
7528 c
7529 })
7530 .collect();
7531 Ok(Some(QueryResult::Rows { columns, rows }))
7532 }
7533
7534 /// v7.37 (round 882) — the bounded sort of `try_spill_sorted_scan`,
7535 /// handing each row to the consumer instead of collecting the answer.
7536 ///
7537 /// That walk bounds the SORT and then returns `QueryResult::Rows`,
7538 /// which holds every output row. Measured at `work_mem = 4 MB` over
7539 /// 200-byte rows, RSS above the server's own baseline while the
7540 /// query runs grew +30 MB at 100k rows, +68 MB at 200k and +137 MB
7541 /// at 400k — linear — while the spill underneath worked correctly
7542 /// (9 / 17 / 33 runs, witnessed DURING the query; `FileRun::drop`
7543 /// removes each file, so a count taken afterwards reads 0 whatever
7544 /// happened, and an earlier reading of "no spill at all" was that
7545 /// blind witness). The growth is the collected result, not the sort.
7546 ///
7547 /// Emitting makes peak the budget, one buffer per run and a single
7548 /// row — the state a merge already holds at every step. It also
7549 /// frees each projected row as the next is built rather than
7550 /// accumulating them, which is where the time is: a profile of the
7551 /// collecting walk put the allocator at 586 samples, more than every
7552 /// sort comparison combined (420), against 19 for `push` itself.
7553 /// v7.37 (round 923) — which of a sort record's columns the output half
7554 /// reads. The record is the SOURCE row (round 836), so a narrow projection
7555 /// decoded every column: skipping one 200-byte text halves a decode
7556 /// (2.17 -> 1.14 ms per pass at 10k rows, priced additively).
7557 ///
7558 /// Timid on purpose — a wrong mask is a SILENT wrong answer, a pruned
7559 /// column reads NULL. Answers only when every projection item is a bare
7560 /// column reference AND every ORDER BY key is a bound column; anything
7561 /// else returns empty, decoding everything as before.
7562 /// `explain.rs`'s `collect_column_refs` is NOT used: its `_ => {}` arm
7563 /// drops references from expression kinds it does not enumerate.
7564 ///
7565 /// ORDER BY columns are included — the merge re-derives keys from the
7566 /// decoded row on the spilled path, so pruning one would sort NULLs.
7567 pub(crate) fn sort_record_columns_needed(
7568 items: &[SelectItem],
7569 order_bound: &[Option<usize>],
7570 arity: usize,
7571 ctx: &EvalContext,
7572 ) -> Vec<bool> {
7573 let all_bare = items.iter().all(|i| {
7574 matches!(
7575 i,
7576 SelectItem::Expr {
7577 expr: Expr::Column(_),
7578 ..
7579 }
7580 )
7581 });
7582 if !all_bare || order_bound.iter().any(Option::is_none) {
7583 return Vec::new();
7584 }
7585 let mut mask = alloc::vec![false; arity];
7586 for item in items {
7587 if let SelectItem::Expr {
7588 expr: Expr::Column(c),
7589 ..
7590 } = item
7591 {
7592 match crate::eval::find_column_pos(c, ctx) {
7593 Some(p) if p < arity => mask[p] = true,
7594 _ => return Vec::new(),
7595 }
7596 }
7597 }
7598 for p in order_bound.iter().flatten() {
7599 if *p < arity {
7600 mask[*p] = true;
7601 } else {
7602 return Vec::new();
7603 }
7604 }
7605 mask
7606 }
7607
7608 /// r1025 — `ORDER BY <indexed NOT NULL column>` walks the index instead
7609 /// of sorting.
7610 ///
7611 /// PG serves such an ordering from the index and never sorts. We sorted:
7612 /// measured at 400,000 rows, `SELECT pad FROM t ORDER BY id` costs
7613 /// 138-144 ms against PG18's 64-75, and the call tree puts the cost in
7614 /// the sorter's own round trip — `ExternalSorter::finish_each` →
7615 /// `next_row` → `decode_row_body_dense_pruned` → `read_value_body`.
7616 /// Every row is encoded into the sorter's arena and decoded back out,
7617 /// for an order the index already holds.
7618 ///
7619 /// The walk exists — `try_pk_walk_top_n` — and requires a `LIMIT`,
7620 /// because it was built for top-N. This is the unbounded sibling.
7621 ///
7622 /// NOT NULL is a hard gate, not a simplification: a NULL key is absent
7623 /// from a btree, so walking one would silently drop those rows. That is
7624 /// exactly the defect r1020 fixed on the top-N path, where it had
7625 /// shipped.
7626 /// r1044 — the index this statement's ORDER BY can be WALKED on,
7627 /// instead of sorted, or `None`.
7628 ///
7629 /// Extracted so `EXPLAIN` can ask the same question the executor
7630 /// answers. It could not, and said so: `SELECT pad FROM t ORDER BY
7631 /// id` on a 400,000-row table planned as `Sort` over `Seq Scan`
7632 /// while the executor walked the primary key — 34.9 ms against
7633 /// 147.0 for the same query ordered by an unindexed column, so the
7634 /// walk was plainly running. Round 551 fixed a different case of
7635 /// this and wrote the reason down: EXPLAIN is the first thing any
7636 /// performance question opens, and an instrument that misnames the
7637 /// access path is worse than one that says nothing.
7638 ///
7639 /// The gate is here once. Two copies of it is how the plan and the
7640 /// executor come to disagree again.
7641 pub(crate) fn index_order_walk_target(
7642 &self,
7643 stmt: &SelectStatement,
7644 from: &FromClause,
7645 ) -> Option<(String, usize)> {
7646 if stmt.order_by.len() != 1
7647 || !stmt.distinct_on.is_empty()
7648 || stmt.limit_with_ties
7649 || stmt.limit.is_some()
7650 || stmt.offset.is_some()
7651 || stmt.having.is_some()
7652 || stmt.group_by.is_some()
7653 || !stmt.unions.is_empty()
7654 || !from.joins.is_empty()
7655 || from.primary.lateral_subquery.is_some()
7656 || from.primary.unnest_expr.is_some()
7657 || from.primary.as_of_segment.is_some()
7658 || from.primary.generate_series_args.is_some()
7659 || select_has_window(stmt)
7660 || aggregate::uses_aggregate(stmt)
7661 {
7662 return None;
7663 }
7664 if stmt
7665 .items
7666 .iter()
7667 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
7668 {
7669 return None;
7670 }
7671 let table = self.active_catalog().get(&from.primary.name)?;
7672 if table.has_cold_rows_fast() {
7673 return None;
7674 }
7675 if !from.primary.only
7676 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
7677 {
7678 return None;
7679 }
7680 let alias = from
7681 .primary
7682 .alias
7683 .as_deref()
7684 .unwrap_or(from.primary.name.as_str());
7685 let cols = &table.schema().columns;
7686 let order = &stmt.order_by[0];
7687 let Expr::Column(oc) = &order.expr else {
7688 return None;
7689 };
7690 if let Some(q) = &oc.qualifier
7691 && !q.eq_ignore_ascii_case(alias)
7692 {
7693 return None;
7694 }
7695 let order_pos = cols
7696 .iter()
7697 .position(|c| c.name.eq_ignore_ascii_case(&oc.name))?;
7698 // r1047 — DISTINCT joins the walk when the projection IS the
7699 // order column, and only then. The index's keys are canonical
7700 // (r1039: representation equality is value equality — the
7701 // property every seek already depends on), so one key is one
7702 // distinct value and the walk can emit the first passing row of
7703 // each key group instead of hashing every row. On the release
7704 // sweep's `SELECT DISTINCT n FROM t ORDER BY n` — 400,000 rows,
7705 // 1,000 distinct values — the hash path priced at 21.3-22.7 ms
7706 // with an ablation floor of 14.8, because the hash must
7707 // normalize and probe ALL the rows; the walk visits each key
7708 // once. A wider projection makes DISTINCT about the whole tuple,
7709 // not the key, so anything else still declines.
7710 if stmt.distinct {
7711 let only_the_order_column = stmt.items.len() == 1
7712 && match &stmt.items[0] {
7713 SelectItem::Expr {
7714 expr: Expr::Column(c),
7715 ..
7716 } => {
7717 c.name.eq_ignore_ascii_case(&oc.name)
7718 && match &c.qualifier {
7719 Some(q) => q.eq_ignore_ascii_case(alias),
7720 None => true,
7721 }
7722 }
7723 _ => false,
7724 };
7725 if !only_the_order_column {
7726 return None;
7727 }
7728 }
7729 // r1046 — a nullable key no longer refuses the walk; it changes
7730 // what the walk has to do. A NULL key is not in the btree, so
7731 // walking alone would silently drop those rows — the r1020
7732 // defect, which shipped once. The walk emits them separately, at
7733 // the end SQL puts them.
7734 //
7735 // Refusing was costing every nullable indexed column a 3.4x:
7736 // `SELECT id FROM t ORDER BY b` over 400,000 rows measured
7737 // 72.0 ms with the column nullable and 20.2 with the same data
7738 // under NOT NULL. `NOT NULL` is not the default, so that was the
7739 // common case paying for the uncommon one.
7740 let index = table.index_on(order_pos)?;
7741 if !matches!(index.kind, spg_storage::IndexKind::BTree(_))
7742 || index.expression.is_some()
7743 || index.partial_predicate.is_some()
7744 {
7745 return None;
7746 }
7747 Some((index.name.clone(), order_pos))
7748 }
7749
7750 fn try_index_order_stream<F>(
7751 &self,
7752 stmt: &SelectStatement,
7753 from: &FromClause,
7754 cancel: CancelToken<'_>,
7755 emit: &mut F,
7756 ) -> Result<Option<usize>, EngineError>
7757 where
7758 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
7759 {
7760 // r1044 — the shape gate lives in `index_order_walk_target`, so
7761 // `EXPLAIN` answers the same question. What stays here is the
7762 // part that RAISES (an illegal ORDER BY has to keep erroring
7763 // from where it did) and the bindings the walk needs.
7764 crate::orderby::check_order_by_legality(stmt)?;
7765 crate::orderby::check_order_by_positions(stmt)?;
7766 crate::window::reject_window_in_row_clauses(stmt)?;
7767 let Some((_, order_pos)) = self.index_order_walk_target(stmt, from) else {
7768 return Ok(None);
7769 };
7770 let Some(table) = self.active_catalog().get(&from.primary.name) else {
7771 return Ok(None);
7772 };
7773 let alias = from
7774 .primary
7775 .alias
7776 .as_deref()
7777 .unwrap_or(from.primary.name.as_str());
7778 let cols = table.schema().columns.clone();
7779 let order = &stmt.order_by[0];
7780 let Some(index) = table.index_on(order_pos) else {
7781 return Ok(None);
7782 };
7783
7784 let sess = self.dml_session();
7785 let ctx = EvalContext::new(&cols, Some(alias))
7786 .with_catalog(self.active_catalog())
7787 .with_session(&sess);
7788 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
7789 let columns: Vec<ColumnSchema> = projection
7790 .iter()
7791 .map(|p| {
7792 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
7793 c.user_enum_type = p.user_enum_type.clone();
7794 c.mysql_fsp = p.mysql_fsp;
7795 c
7796 })
7797 .collect();
7798 emit(crate::StreamItem::Header(&columns))?;
7799 let bound_pos: Vec<Option<usize>> = projection
7800 .iter()
7801 .map(|p| match &p.expr {
7802 Expr::Column(c) => match crate::eval::locate_column(c, &ctx) {
7803 Ok(Some(pos)) => Some(pos),
7804 _ => None,
7805 },
7806 _ => None,
7807 })
7808 .collect();
7809
7810 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
7811 .where_
7812 .as_ref()
7813 .filter(|w| crate::eval::fully_compilable(w))
7814 .map(|w| crate::eval::compile_expr(w, &ctx));
7815 let mut eval_stack: Vec<Value<'static>> = Vec::new();
7816 let mut values: Vec<Value<'static>> = Vec::with_capacity(projection.len());
7817 let snapshot = self.current_snapshot();
7818
7819 // A btree holds one locator per row VERSION, so a row whose key was
7820 // updated can sit under two keys and a dead one can sit beside its
7821 // replacement. The visibility gate drops the dead; `seen` drops a
7822 // live row that the walk reaches twice, which would otherwise be a
7823 // duplicated output row rather than a slow one.
7824 let mut emitted_rows = alloc::vec![false; table.rows().len()];
7825
7826 // r1046 — the rows the index cannot hold.
7827 //
7828 // A NULL key is not in the btree, so the walk below never reaches
7829 // those rows; they are emitted here, at the end SQL puts them.
7830 // PG's default is NULLS LAST ascending and NULLS FIRST
7831 // descending, and an explicit `NULLS FIRST` / `NULLS LAST` wins —
7832 // the same rule `order_by_value_cmp_raw` applies to the sort this
7833 // replaces, so the two orders agree.
7834 //
7835 // Finding them costs one pass over the column. That pass is why
7836 // this is still worth doing: the sort it replaces encodes and
7837 // decodes every row, and the walk plus the pass measured 72.0 ms
7838 // down to about 22 on 400,000 rows.
7839 let nulls_first = order.nulls_first.unwrap_or(order.desc);
7840 // r1047 — under DISTINCT the walk emits the FIRST passing row of
7841 // each key group and skips the rest; the gate admits DISTINCT
7842 // only when the projection is the order column itself, so one
7843 // canonical key is one output row. NULL is one distinct value,
7844 // so the NULL pass stops at its first emit too.
7845 let distinct = stmt.distinct;
7846 let mut count = 0usize;
7847 let mut visited = 0usize;
7848 let mut emit_null_rows = |emitted_rows: &mut alloc::vec::Vec<bool>,
7849 eval_stack: &mut Vec<Value<'static>>,
7850 values: &mut Vec<Value<'static>>,
7851 visited: &mut usize,
7852 emit: &mut F|
7853 -> Result<usize, EngineError> {
7854 if !cols[order_pos].nullable {
7855 return Ok(0);
7856 }
7857 let mut n = 0usize;
7858 for (ri, row) in table.rows().iter().enumerate() {
7859 if !matches!(row.values.get(order_pos), Some(Value::Null)) {
7860 continue;
7861 }
7862 if emitted_rows.get(ri).copied().unwrap_or(true) {
7863 continue;
7864 }
7865 if !table.is_row_visible(ri, &snapshot) {
7866 continue;
7867 }
7868 *visited += 1;
7869 if visited.is_multiple_of(256) {
7870 cancel.check()?;
7871 }
7872 emitted_rows[ri] = true;
7873 if Self::stream_project_row(
7874 row,
7875 stmt.where_.as_ref(),
7876 compiled_where.as_ref(),
7877 eval_stack,
7878 &projection,
7879 &bound_pos,
7880 &ctx,
7881 values,
7882 emit,
7883 )? {
7884 n += 1;
7885 if distinct {
7886 break;
7887 }
7888 }
7889 }
7890 Ok(n)
7891 };
7892
7893 if nulls_first {
7894 count += emit_null_rows(
7895 &mut emitted_rows,
7896 &mut eval_stack,
7897 &mut values,
7898 &mut visited,
7899 emit,
7900 )?;
7901 }
7902
7903 let walker: alloc::boxed::Box<
7904 dyn Iterator<Item = (&spg_storage::IndexKey, &spg_storage::PostingList)>,
7905 > = if order.desc {
7906 alloc::boxed::Box::new(index.iter_desc())
7907 } else {
7908 alloc::boxed::Box::new(index.iter_asc())
7909 };
7910 for (_key, locators) in walker {
7911 for loc in locators {
7912 let spg_storage::RowLocator::Hot(ri) = *loc else {
7913 continue;
7914 };
7915 if emitted_rows.get(ri).copied().unwrap_or(true) {
7916 continue;
7917 }
7918 if !table.is_row_visible(ri, &snapshot) {
7919 continue;
7920 }
7921 let Some(row) = table.rows().get(ri) else {
7922 continue;
7923 };
7924 visited += 1;
7925 if visited.is_multiple_of(256) {
7926 cancel.check()?;
7927 }
7928 emitted_rows[ri] = true;
7929 if Self::stream_project_row(
7930 row,
7931 stmt.where_.as_ref(),
7932 compiled_where.as_ref(),
7933 &mut eval_stack,
7934 &projection,
7935 &bound_pos,
7936 &ctx,
7937 &mut values,
7938 emit,
7939 )? {
7940 count += 1;
7941 // One row per key group: the rest are the same value.
7942 if distinct {
7943 break;
7944 }
7945 }
7946 }
7947 }
7948
7949 if !nulls_first {
7950 count += emit_null_rows(
7951 &mut emitted_rows,
7952 &mut eval_stack,
7953 &mut values,
7954 &mut visited,
7955 emit,
7956 )?;
7957 }
7958 Ok(Some(count))
7959 }
7960
7961 /// r1031 — `ORDER BY` over NOT NULL integer columns, sorted without
7962 /// building an `OrderKey` vector per row.
7963 ///
7964 /// The row-returning sorted scan allocates twice per row: one
7965 /// `Vec<OrderKey>` for the sort keys and one `Vec<Value>` for the
7966 /// projection. Counted over 400 k rows (r1030,
7967 /// `docs/PERF_SORTED_SCAN_ALLOCATIONS_2026-08-15.md`), that is 800,067
7968 /// allocations and 208 MB of traffic for an answer of four hundred
7969 /// thousand integers.
7970 ///
7971 /// The key half is pure ceremony on this shape.
7972 /// `sort_tagged_by_inline_int_key` already sorts indices rather than
7973 /// rows, so the per-row vector is built, has one integer taken out of
7974 /// it, and is then dragged through the permutation — it exists to carry
7975 /// a number the row's column already held. This lane carries the number
7976 /// instead, in a fixed-size array that lives inside the buffer element
7977 /// and allocates nothing. Same idea as the predicate VM's integer lane.
7978 ///
7979 /// Declines to `None` for anything it does not cover, and every caller
7980 /// falls through to the general path, so the gate list is the
7981 /// specification.
7982 ///
7983 /// Ties: equal keys keep scan order, as the stable sort on the general
7984 /// path does. Rows that tie on every ORDER BY term are entitled to any
7985 /// order among themselves either way — see `STABILITY.md`.
7986 fn try_int_key_sorted_stream<F>(
7987 &self,
7988 stmt: &SelectStatement,
7989 from: &FromClause,
7990 cancel: CancelToken<'_>,
7991 emit: &mut F,
7992 ) -> Result<Option<usize>, EngineError>
7993 where
7994 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
7995 {
7996 /// Sort terms this lane carries inline. Four covers every ORDER BY
7997 /// in the endpoint sweep and in the dogfood corpus; wider ones fall
7998 /// through rather than growing the buffer element for everybody.
7999 const MAX_KEYS: usize = 4;
8000
8001 if stmt.order_by.is_empty()
8002 || stmt.order_by.len() > MAX_KEYS
8003 || stmt.distinct
8004 || stmt.limit_with_ties
8005 || stmt.limit.is_some()
8006 || stmt.offset.is_some()
8007 || stmt.having.is_some()
8008 || stmt.group_by.is_some()
8009 || !stmt.unions.is_empty()
8010 || !from.joins.is_empty()
8011 || from.primary.lateral_subquery.is_some()
8012 || from.primary.unnest_expr.is_some()
8013 || from.primary.as_of_segment.is_some()
8014 || from.primary.generate_series_args.is_some()
8015 || select_has_window(stmt)
8016 || aggregate::uses_aggregate(stmt)
8017 {
8018 return Ok(None);
8019 }
8020 if stmt
8021 .items
8022 .iter()
8023 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
8024 {
8025 return Ok(None);
8026 }
8027 crate::orderby::check_order_by_legality(stmt)?;
8028 crate::orderby::check_order_by_positions(stmt)?;
8029 crate::window::reject_window_in_row_clauses(stmt)?;
8030 let Some(table) = self.active_catalog().get(&from.primary.name) else {
8031 return Ok(None);
8032 };
8033 if table.has_cold_rows_fast() {
8034 return Ok(None);
8035 }
8036 if !from.primary.only
8037 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
8038 {
8039 return Ok(None);
8040 }
8041 let alias = from
8042 .primary
8043 .alias
8044 .as_deref()
8045 .unwrap_or(from.primary.name.as_str());
8046 let cols = table.schema().columns.clone();
8047
8048 // Every ORDER BY term must be a NOT NULL integer column of this
8049 // table. NOT NULL is what lets the key be a bare integer: with
8050 // NULLs the lane would have to carry their ordering too, and
8051 // getting that subtly wrong is the r1020 defect.
8052 let mut key_pos = [0usize; MAX_KEYS];
8053 let mut descs = [false; MAX_KEYS];
8054 // PG's default is NULLS LAST for ASC and NULLS FIRST for DESC,
8055 // which the AST records as `None`; `unwrap_or(desc)` is how the
8056 // rest of the engine resolves it.
8057 let mut nulls_first = [false; MAX_KEYS];
8058 let n_keys = stmt.order_by.len();
8059 for (slot, order) in stmt.order_by.iter().enumerate() {
8060 let Expr::Column(oc) = &order.expr else {
8061 return Ok(None);
8062 };
8063 if let Some(q) = &oc.qualifier
8064 && !q.eq_ignore_ascii_case(alias)
8065 {
8066 return Ok(None);
8067 }
8068 let Some(pos) = cols
8069 .iter()
8070 .position(|c| c.name.eq_ignore_ascii_case(&oc.name))
8071 else {
8072 return Ok(None);
8073 };
8074 if !matches!(
8075 cols[pos].ty,
8076 spg_storage::DataType::SmallInt
8077 | spg_storage::DataType::Int
8078 | spg_storage::DataType::BigInt
8079 ) {
8080 return Ok(None);
8081 }
8082 key_pos[slot] = pos;
8083 descs[slot] = order.desc;
8084 nulls_first[slot] = order.nulls_first.unwrap_or(order.desc);
8085 }
8086
8087 let sess = self.dml_session();
8088 let ctx = EvalContext::new(&cols, Some(alias))
8089 .with_catalog(self.active_catalog())
8090 .with_session(&sess);
8091 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
8092 let columns: Vec<ColumnSchema> = projection
8093 .iter()
8094 .map(|p| {
8095 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8096 c.user_enum_type = p.user_enum_type.clone();
8097 c.mysql_fsp = p.mysql_fsp;
8098 c
8099 })
8100 .collect();
8101 let bound_pos: Vec<Option<usize>> = projection
8102 .iter()
8103 .map(|p| match &p.expr {
8104 Expr::Column(c) => match crate::eval::locate_column(c, &ctx) {
8105 Ok(Some(pos)) => Some(pos),
8106 _ => None,
8107 },
8108 _ => None,
8109 })
8110 .collect();
8111 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
8112 .where_
8113 .as_ref()
8114 .filter(|w| crate::eval::fully_compilable(w))
8115 .map(|w| crate::eval::compile_expr(w, &ctx));
8116
8117 // The same first-observable point the materialising planner fires,
8118 // placed after the gates so it fires exactly once: this lane runs
8119 // BEFORE that planner and would otherwise be a hole in the
8120 // panic-isolation and cancellation-race coverage rather than a
8121 // faster path through it.
8122 crate::injection_point!("planner_first_row_fetch", &stmt.from);
8123
8124 let mut eval_stack: Vec<Value<'static>> = Vec::new();
8125 let mut values: Vec<Value<'static>> = Vec::with_capacity(projection.len());
8126 let mut budget = ByteBudget::new(self.max_query_bytes);
8127 let snapshot = self.current_snapshot();
8128 // Keys, a NULL bit per key slot, and the row. The bitmask keeps
8129 // the element small: a nullable key still costs one bit rather
8130 // than a second array.
8131 let mut sorted: Vec<([i64; MAX_KEYS], u8, Vec<Value<'static>>)> = Vec::new();
8132
8133 for (ri, row) in table.rows().iter().enumerate() {
8134 if ri.is_multiple_of(256) {
8135 cancel.check()?;
8136 }
8137 if !table.is_row_visible(ri, &snapshot) {
8138 continue;
8139 }
8140 // The key comes from the STORED row, before projection: an
8141 // ORDER BY column need not appear in the select list.
8142 let mut keys = [0i64; MAX_KEYS];
8143 let mut nulls = 0u8;
8144 let mut keyed = true;
8145 for slot in 0..n_keys {
8146 match row.values.get(key_pos[slot]) {
8147 Some(Value::SmallInt(v)) => keys[slot] = i64::from(*v),
8148 Some(Value::Int(v)) => keys[slot] = i64::from(*v),
8149 Some(Value::BigInt(v)) => keys[slot] = *v,
8150 Some(Value::Null) | None => nulls |= 1 << slot,
8151 // An integer column holding something else is a row
8152 // this lane cannot order; hand the whole query back
8153 // rather than guess at it.
8154 _ => {
8155 keyed = false;
8156 break;
8157 }
8158 }
8159 }
8160 if !keyed {
8161 return Ok(None);
8162 }
8163 if !Self::stream_filter_project(
8164 row,
8165 stmt.where_.as_ref(),
8166 compiled_where.as_ref(),
8167 &mut eval_stack,
8168 &projection,
8169 &bound_pos,
8170 &ctx,
8171 &mut values,
8172 )? {
8173 continue;
8174 }
8175 budget.charge(crate::bytebudget::approx_values_bytes(&values))?;
8176 sorted.push((keys, nulls, core::mem::take(&mut values)));
8177 values.reserve(projection.len());
8178 }
8179
8180 sorted.sort_by(|a, b| {
8181 use core::cmp::Ordering;
8182 for slot in 0..n_keys {
8183 let bit = 1u8 << slot;
8184 let ord = match (a.1 & bit != 0, b.1 & bit != 0) {
8185 (true, true) => Ordering::Equal,
8186 // Where the NULLs go is already decided — `nulls_first`
8187 // resolved DESC's default when it was read. Reversing
8188 // this for DESC as well would apply the direction
8189 // twice and put them at the wrong end.
8190 (true, false) => {
8191 if nulls_first[slot] {
8192 Ordering::Less
8193 } else {
8194 Ordering::Greater
8195 }
8196 }
8197 (false, true) => {
8198 if nulls_first[slot] {
8199 Ordering::Greater
8200 } else {
8201 Ordering::Less
8202 }
8203 }
8204 (false, false) => {
8205 let o = a.0[slot].cmp(&b.0[slot]);
8206 if descs[slot] { o.reverse() } else { o }
8207 }
8208 };
8209 if ord != Ordering::Equal {
8210 return ord;
8211 }
8212 }
8213 Ordering::Equal
8214 });
8215
8216 emit(crate::StreamItem::Header(&columns))?;
8217 let count = sorted.len();
8218 for (_, _, vals) in &sorted {
8219 emit(crate::StreamItem::Row(crate::RowCells::Values(vals)))?;
8220 }
8221 Ok(Some(count))
8222 }
8223
8224 fn try_spill_sorted_stream<F>(
8225 &self,
8226 stmt: &SelectStatement,
8227 from: &FromClause,
8228 cancel: CancelToken<'_>,
8229 emit: &mut F,
8230 ) -> Result<Option<usize>, EngineError>
8231 where
8232 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8233 {
8234 // The shapes `try_spill_sorted_scan` declines, plus the ones the
8235 // streaming executor does not carry (a LIMIT is already bounded
8236 // by a partial sort; the rest need the answer addressable).
8237 if !self.can_spill()
8238 || stmt.order_by.is_empty()
8239 || stmt.distinct
8240 || stmt.limit_with_ties
8241 || stmt.limit.is_some()
8242 || stmt.offset.is_some()
8243 || stmt.having.is_some()
8244 || stmt.group_by.is_some()
8245 || !stmt.unions.is_empty()
8246 || !from.joins.is_empty()
8247 || from.primary.lateral_subquery.is_some()
8248 || from.primary.unnest_expr.is_some()
8249 || from.primary.as_of_segment.is_some()
8250 || from.primary.generate_series_args.is_some()
8251 || select_has_window(stmt)
8252 || aggregate::uses_aggregate(stmt)
8253 {
8254 return Ok(None);
8255 }
8256 if stmt
8257 .items
8258 .iter()
8259 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
8260 {
8261 return Ok(None);
8262 }
8263 // Everything `exec_bare_select_cancel` does before it scans runs
8264 // BELOW this path, so a statement claimed here skips it. Three of
8265 // those were missed on the way in and each was caught by a
8266 // different gate — the ORDER BY rules by an e2e (`SELECT a FROM t
8267 // ORDER BY 2` sorted happily instead of raising 42P10), the
8268 // cancellation check by another, the partition fan-out by the
8269 // differential corpus. What is reconciled, item by item: with-ties
8270 // needs ORDER BY (gated above), USING/NATURAL and RLS join
8271 // rewrites (joins gated above), the single-table RLS predicate
8272 // (the dispatcher declines a policy-subject table before this is
8273 // reached), the meta-view dispatch (those names are not in the
8274 // catalog, so the lookup below declines). These three are calls,
8275 // so the message and SQLSTATE are the ones the fall-back gives —
8276 // `select_has_window` above reads the select list and ORDER BY but
8277 // not WHERE, which is the case the third one covers.
8278 crate::orderby::check_order_by_legality(stmt)?;
8279 crate::orderby::check_order_by_positions(stmt)?;
8280 crate::window::reject_window_in_row_clauses(stmt)?;
8281 // A parent's rows are its children's. These walks scan the named
8282 // relation alone, so a partitioned or inherited parent comes back
8283 // short — and silently: the corpus caught `SELECT id FROM pr
8284 // ORDER BY id` and `SELECT k FROM pl ORDER BY k` returning the
8285 // parent's own rows instead of the partitions'. `ONLY` is exactly
8286 // the case that does not fan out, so it stays, which is the test
8287 // the FROM-clause fan-out itself makes.
8288 if !from.primary.only
8289 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
8290 {
8291 return Ok(None);
8292 }
8293 let Some(table) = self.active_catalog().get(&from.primary.name) else {
8294 return Ok(None);
8295 };
8296 // Cold-tier rows live outside `rows()`; this walk would drop
8297 // them silently, the same reason round 831's walk declines.
8298 if table.has_cold_rows_fast() {
8299 return Ok(None);
8300 }
8301
8302 let alias = from
8303 .primary
8304 .alias
8305 .as_deref()
8306 .unwrap_or(from.primary.name.as_str());
8307 let cols = table.schema().columns.clone();
8308 let sess = self.dml_session();
8309 let ctx = EvalContext::new(&cols, Some(alias))
8310 .with_catalog(self.active_catalog())
8311 .with_session(&sess);
8312 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
8313 let order_by = stmt.order_by.clone();
8314 // The same one-shot resolution the general path does (round
8315 // 582): each ORDER BY column is bound once, not once per row.
8316 let order_bound = crate::orderby::order_by_bound_positions(&order_by, &cols, Some(alias));
8317 let descs: Vec<bool> = order_by.iter().map(|o| o.desc).collect();
8318 // Resolved BEFORE the scan, because it now decides what the sort
8319 // STORES and not just what it decodes (round 995).
8320 let needed = Self::sort_record_columns_needed(&stmt.items, &order_bound, cols.len(), &ctx);
8321
8322 let mut sorter = crate::extsort::ExternalSorter::new(
8323 self.temp_run_factory,
8324 self.session_work_mem_bytes(),
8325 cols.clone(),
8326 &descs,
8327 )
8328 .with_stats(&self.spill_stats)
8329 .with_pruned(&needed);
8330 let snapshot = self.current_snapshot();
8331 // One key buffer for the whole scan: `push` drains it and leaves
8332 // the capacity behind.
8333 let mut keys: Vec<OrderKey> = Vec::new();
8334 // r1024 — compile the predicate once for the scan.
8335 //
8336 // These two sorted-spill scans are the paths a single-table SELECT
8337 // with an ORDER BY takes, and they were the last row-returning ones
8338 // still walking the expression tree per row. r1023 did the
8339 // no-ORDER-BY sibling; the sweep's two remaining losing cells are
8340 // exactly this shape.
8341 //
8342 // Found from the profile's CALL TREE rather than its leaves. The
8343 // leaves say what is expensive — `eval_expr` 320, `apply_binary`
8344 // 261, `mod_op` 178 — and two attempts at reasoning out which
8345 // function asked for it were both wrong. The tree names the caller
8346 // chain, and it named this one.
8347 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
8348 .where_
8349 .as_ref()
8350 .filter(|w| crate::eval::fully_compilable(w))
8351 .map(|w| crate::eval::compile_expr(w, &ctx));
8352 let mut eval_stack: Vec<Value<'static>> = Vec::new();
8353 for (i, row) in table.scan_visible_from(0, &snapshot) {
8354 if i.is_multiple_of(256) {
8355 cancel.check()?;
8356 }
8357 if let Some(c) = &compiled_where {
8358 if !crate::eval::compiled::eval_compiled_pred(
8359 c,
8360 row,
8361 &ctx,
8362 &mut eval_stack,
8363 ctx.mysql_dialect,
8364 )? {
8365 continue;
8366 }
8367 } else if let Some(w) = &stmt.where_ {
8368 let cond = crate::eval::eval_expr(w, row, &ctx).map_err(EngineError::Eval)?;
8369 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
8370 continue;
8371 }
8372 }
8373 keys.clear();
8374 crate::orderby::build_order_keys_bound(&order_by, &order_bound, row, &ctx, &mut keys)?;
8375 sorter.push(&mut keys, row)?;
8376 }
8377
8378 let columns: Vec<ColumnSchema> = projection
8379 .iter()
8380 .map(|p| {
8381 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8382 c.user_enum_type = p.user_enum_type.clone();
8383 c.mysql_fsp = p.mysql_fsp;
8384 c
8385 })
8386 .collect();
8387 emit(crate::StreamItem::Header(&columns))?;
8388
8389 let key_ctx = &ctx;
8390 let mut emitted_since_check = 0usize;
8391 let n = sorter.finish_each(
8392 |src, buf| {
8393 crate::orderby::build_order_keys_bound(&order_by, &order_bound, src, key_ctx, buf)
8394 },
8395 |src, values| {
8396 for p in &projection {
8397 values.push(
8398 crate::eval::eval_expr(&p.expr, src, key_ctx).map_err(EngineError::Eval)?,
8399 );
8400 }
8401 Ok(())
8402 },
8403 |cells| {
8404 // The merge is the long half of a big sort, and the scan's
8405 // check above stops running once it ends: a cancelled
8406 // `SELECT pad FROM big ORDER BY id` delivered all 120k rows
8407 // anyway. Same stride as the scan.
8408 emitted_since_check += 1;
8409 if emitted_since_check >= 256 {
8410 emitted_since_check = 0;
8411 cancel.check()?;
8412 }
8413 emit(crate::StreamItem::Row(crate::RowCells::Values(cells)))
8414 },
8415 )?;
8416 Ok(Some(n))
8417 }
8418
8419 /// One row of the single-table streaming walk: the WHERE test, the
8420 /// projection, the emit. Returns whether a row was emitted.
8421 ///
8422 /// v7.39 (round 970) — factored out because the walk now has two ways
8423 /// to reach a row, the sequential scan and an index seek's candidate
8424 /// positions, and both must do IDENTICALLY this. A copy in each is how
8425 /// two paths for one job drift; this file already carries the cost of
8426 /// that lesson twice (rounds 823 and 961, both resolvers).
8427 ///
8428 /// `#[inline]` so the scan loop keeps the shape round 957 measured it
8429 /// in — a shared hot path pays for a new abstraction whether or not it
8430 /// uses it, and this one is on the scan.
8431 #[inline]
8432 #[allow(clippy::too_many_arguments)]
8433 fn stream_filter_project(
8434 row: &spg_storage::Row<'static>,
8435 where_: Option<&Expr>,
8436 // r1023 — the same WHERE, compiled once by the caller. `None` means
8437 // the expression did not qualify and `where_` is evaluated as before.
8438 compiled_where: Option<&crate::eval::CompiledExpr>,
8439 eval_stack: &mut Vec<Value<'static>>,
8440 projection: &[ProjectedItem],
8441 bound_pos: &[Option<usize>],
8442 ctx: &crate::eval::EvalContext<'_>,
8443 values: &mut Vec<Value<'static>>,
8444 ) -> Result<bool, EngineError> {
8445 // r1023 — this scan ran its predicate through the TREE INTERPRETER,
8446 // once per row, and it was the only row-returning path that did.
8447 // The aggregate path, `table_access`, and the PK walker all compile
8448 // theirs. Profiled: on `SELECT pad FROM d WHERE id % 3 = 0` the
8449 // server's live samples were `eval_expr` 99, `apply_binary` 81,
8450 // `mod_op` 29 — the interpreter, not delivery.
8451 //
8452 // The arithmetic accounted for it exactly. Over the wire, the same
8453 // filter costs 6.375 ms returning rows and 0.679 ms counting them;
8454 // the 5.70 ms difference over 50,000 scanned rows is 114 ns each,
8455 // which is what an interpreted predicate costs against the compiled
8456 // lane's 11.7. It was named "delivery after a filter" before this
8457 // profile, and it was never delivery.
8458 if let Some(c) = compiled_where {
8459 if !crate::eval::compiled::eval_compiled_pred(
8460 c,
8461 row,
8462 ctx,
8463 eval_stack,
8464 ctx.mysql_dialect,
8465 )? {
8466 return Ok(false);
8467 }
8468 } else if let Some(w) = where_ {
8469 let cond = crate::eval::eval_expr(w, row, ctx).map_err(EngineError::Eval)?;
8470 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
8471 return Ok(false);
8472 }
8473 }
8474 values.clear();
8475 for (p, bound) in projection.iter().zip(bound_pos) {
8476 values.push(match bound {
8477 Some(pos) => crate::eval::column_at(*pos, row, ctx).map_err(EngineError::Eval)?,
8478 None => crate::eval::eval_expr(&p.expr, row, ctx).map_err(EngineError::Eval)?,
8479 });
8480 }
8481 Ok(true)
8482 }
8483
8484 /// The same filter and projection, then emit. Split from
8485 /// [`Self::stream_filter_project`] so a path that has to BUFFER rows
8486 /// before it can emit them — a sort — runs the identical predicate and
8487 /// projection rather than a second copy of them.
8488 #[allow(clippy::too_many_arguments)]
8489 fn stream_project_row<F>(
8490 row: &spg_storage::Row<'static>,
8491 where_: Option<&Expr>,
8492 compiled_where: Option<&crate::eval::CompiledExpr>,
8493 eval_stack: &mut Vec<Value<'static>>,
8494 projection: &[ProjectedItem],
8495 bound_pos: &[Option<usize>],
8496 ctx: &crate::eval::EvalContext<'_>,
8497 values: &mut Vec<Value<'static>>,
8498 emit: &mut F,
8499 ) -> Result<bool, EngineError>
8500 where
8501 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8502 {
8503 if !Self::stream_filter_project(
8504 row,
8505 where_,
8506 compiled_where,
8507 eval_stack,
8508 projection,
8509 bound_pos,
8510 ctx,
8511 values,
8512 )? {
8513 return Ok(false);
8514 }
8515 emit(crate::StreamItem::Row(crate::RowCells::Values(values)))?;
8516 Ok(true)
8517 }
8518
8519 fn try_stream_single_table<F>(
8520 &self,
8521 stmt: &SelectStatement,
8522 from: &FromClause,
8523 cancel: CancelToken<'_>,
8524 emit: &mut F,
8525 ) -> Result<Option<usize>, EngineError>
8526 where
8527 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8528 {
8529 let Some(table) = self.active_catalog().get(&from.primary.name) else {
8530 return Ok(None);
8531 };
8532 // Cold-tier rows live outside `rows()`; the materialising fallback
8533 // covers both tiers and this walk would silently drop them.
8534 if table.has_cold_rows_fast() {
8535 return Ok(None);
8536 }
8537 let alias = from
8538 .primary
8539 .alias
8540 .as_deref()
8541 .unwrap_or(from.primary.name.as_str());
8542 let cols = table.schema().columns.clone();
8543 let sess = self.dml_session();
8544 let ctx = EvalContext::new(&cols, Some(alias))
8545 .with_catalog(self.active_catalog())
8546 .with_session(&sess);
8547 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
8548
8549 let columns: Vec<ColumnSchema> = projection
8550 .iter()
8551 .map(|p| {
8552 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8553 c.user_enum_type = p.user_enum_type.clone();
8554 c.mysql_fsp = p.mysql_fsp;
8555 c
8556 })
8557 .collect();
8558 emit(crate::StreamItem::Header(&columns))?;
8559
8560 // v7.37 (round 957) — resolve each bare-column projection ONCE
8561 // instead of once per row. `find_column_pos`-style resolution is a
8562 // linear walk of the schema comparing column-name strings, and the
8563 // row loop below ran it for every cell of every row: measured at
8564 // 400k rows, binding it out of the loop took `SELECT pad` from
8565 // 16.5-17.5 ms to 10.9-11.7 ms (-41%, two windows, round 954).
8566 //
8567 // ORDER BY has bound its keys this way since round 582
8568 // (`order_by_bound_positions`); the projection never did.
8569 //
8570 // `locate_column` is the same resolution `resolve_column` performs,
8571 // returning the site instead of the value, so the two cannot drift
8572 // apart the way a second hand-written resolver would. Anything it
8573 // declines — an expression, a whole-row reference, a name that does
8574 // not resolve — binds to `None` and takes the general path below,
8575 // errors included, so an empty table still reports nothing rather
8576 // than raising at bind time.
8577 let bound_pos: Vec<Option<usize>> = projection
8578 .iter()
8579 .map(|p| match &p.expr {
8580 Expr::Column(c) => match crate::eval::locate_column(c, &ctx) {
8581 Ok(Some(pos)) => Some(pos),
8582 _ => None,
8583 },
8584 _ => None,
8585 })
8586 .collect();
8587
8588 // One snapshot for the whole scan, as the materialising path takes.
8589 let snapshot = self.current_snapshot();
8590
8591 // v7.39 (round 970) — ask the indices BEFORE walking the table.
8592 //
8593 // This walk had no index step at all, and it is preferred over the
8594 // materialising path, which does have one (`pick_indexed_rows` ->
8595 // `try_index_seek`). So a primary-key point lookup — the commonest
8596 // statement there is — read every row: measured on 500k rows,
8597 // `SELECT * FROM big WHERE id = 250000` took 14.947 ms against
8598 // PG18.4's 0.172 ms, and the cost tracked the TABLE (1k 0.315 ms,
8599 // 10k 1.660, 100k 3.518), which is not what O(log n) looks like.
8600 //
8601 // The control that named it: `... OFFSET 0` — semantically the same
8602 // query — answered in 0.159 ms, because OFFSET is one of the shape
8603 // gates that declines this walk and sends the statement to the path
8604 // that seeks. `LIMIT 1` and `GROUP BY` did the same. The three have
8605 // no semantics in common; what they share is making this function
8606 // stand down.
8607 //
8608 // The seek only NARROWS: every candidate still goes through the
8609 // full WHERE below, exactly as the mutation paths use it, so a
8610 // partial index match cannot change an answer. Positions come back
8611 // already visibility-filtered and already capped at a quarter of the
8612 // table (round 490), so a seek can never cost more than the scan it
8613 // replaces, and `None` means "walk the table" as before.
8614 //
8615 // Sorted because the scan would have produced table order and the
8616 // index produces key order. Without an ORDER BY neither is promised,
8617 // but a walk that silently reorders its answer when an index happens
8618 // to exist is a difference nobody asked for.
8619 let seek_positions: Option<Vec<usize>> = stmt.where_.as_ref().and_then(|w| {
8620 crate::index_access::try_index_seek_positions(w, &cols, table, alias, &snapshot)
8621 });
8622
8623 let mut values: Vec<Value<'static>> = Vec::with_capacity(projection.len());
8624 // r1023 — compile the predicate once for the whole scan. Same gate
8625 // every other path uses: `fully_compilable` or keep the interpreter,
8626 // so a shape the VM cannot take answers exactly as it did before.
8627 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
8628 .where_
8629 .as_ref()
8630 .filter(|w| crate::eval::fully_compilable(w))
8631 .map(|w| crate::eval::compile_expr(w, &ctx));
8632 let mut eval_stack: Vec<Value<'static>> = Vec::new();
8633 let mut count: usize = 0;
8634 match seek_positions {
8635 Some(mut positions) => {
8636 positions.sort_unstable();
8637 for (n, pos) in positions.into_iter().enumerate() {
8638 if n.is_multiple_of(256) {
8639 cancel.check()?;
8640 }
8641 let Some(row) = table.rows().get(pos) else {
8642 continue;
8643 };
8644 if Self::stream_project_row(
8645 row,
8646 stmt.where_.as_ref(),
8647 compiled_where.as_ref(),
8648 &mut eval_stack,
8649 &projection,
8650 &bound_pos,
8651 &ctx,
8652 &mut values,
8653 emit,
8654 )? {
8655 count += 1;
8656 }
8657 }
8658 }
8659 None => {
8660 for (i, row) in table.scan_visible_from(0, &snapshot) {
8661 if i.is_multiple_of(256) {
8662 cancel.check()?;
8663 }
8664 if Self::stream_project_row(
8665 row,
8666 stmt.where_.as_ref(),
8667 compiled_where.as_ref(),
8668 &mut eval_stack,
8669 &projection,
8670 &bound_pos,
8671 &ctx,
8672 &mut values,
8673 emit,
8674 )? {
8675 count += 1;
8676 }
8677 }
8678 }
8679 }
8680 Ok(Some(count))
8681 }
8682
8683 pub(crate) fn try_exec_joined_streaming<F>(
8684 &self,
8685 stmt: &SelectStatement,
8686 cancel: CancelToken<'_>,
8687 emit: &mut F,
8688 ) -> Result<Option<usize>, EngineError>
8689 where
8690 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8691 {
8692 // Shape gates — keep the streamable surface narrow on
8693 // purpose. The fall-back path still handles everything else.
8694 let Some(from) = &stmt.from else {
8695 return Ok(None);
8696 };
8697 // v7.37 (round 830) — decline anything a row-security policy binds
8698 // for this session. Policies are injected in
8699 // `exec_bare_select_cancel`, below this path, so a statement claimed
8700 // here would read the table unfiltered: measured, `SELECT val FROM
8701 // sec` returned all three rows to a session whose policy allows two,
8702 // while `SELECT upper(val) FROM sec` — declined by the shape gates
8703 // and so materialised — returned the correct two.
8704 //
8705 // Declining sends it to the path that enforces. Teaching this one to
8706 // inject the predicate itself would keep the streaming benefit for
8707 // RLS tables and is the better end state; it is not what a
8708 // correctness fix should carry, and the fall-back is exactly as
8709 // correct, only slower.
8710 if self.select_reads_policy_subject_table(stmt) {
8711 return Ok(None);
8712 }
8713 // r1058 — a WITH list this path never materialises: the CTE
8714 // name would be resolved as a physical relation and error
8715 // ("relation \"big\" does not exist" over the extended
8716 // protocol, caught by the perm-runner's wire legs). The
8717 // materialising fallback owns CTE execution.
8718 if !stmt.ctes.is_empty() {
8719 return Ok(None);
8720 }
8721 // r1058 — rewritten system catalogs (`__spg_pg_stat_user_
8722 // tables` and kin) exist only as synth arms on the
8723 // materialising path; claiming one here errored "relation
8724 // does not exist" over the extended protocol for a query the
8725 // simple protocol answered. Prefix test only — a genuinely
8726 // missing relation must keep erroring in-path.
8727 if from.primary.name.starts_with("__spg_")
8728 || from
8729 .joins
8730 .iter()
8731 .any(|j| j.table.name.starts_with("__spg_"))
8732 {
8733 return Ok(None);
8734 }
8735 // r1058 — decline partitioned / inheritance parents, same
8736 // shape of bug as the RLS decline above: this path scans the
8737 // named table's own (empty) heap, so `SELECT id, region FROM
8738 // cust` on a partition parent streamed ZERO rows over the wire
8739 // while COUNT(*) — an aggregate, materialised below — said 3.
8740 // Caught by the perm-runner's server permutations; the
8741 // materialising fallback expands children correctly.
8742 if crate::partition::has_children(self.active_catalog(), &from.primary.name)
8743 || from
8744 .joins
8745 .iter()
8746 .any(|j| crate::partition::has_children(self.active_catalog(), &j.table.name))
8747 {
8748 return Ok(None);
8749 }
8750 // v7.39 (round 790) — single-table SELECTs stream too. This
8751 // gate said "joins only" because the path was written for
8752 // mailrs's joined PROJ shape; a plain `SELECT <cols> FROM t`
8753 // fell to the materialising fallback, which builds the whole
8754 // `Vec<Row<'static>>` and only then iterates it. Measured on
8755 // 300k rows: 181 MB single-table vs 70 MB for the SAME rows
8756 // reached through a one-row JOIN — 2.6x, purely for lacking a
8757 // join. The deferred-join structure handles one source as the
8758 // degenerate stride-1 case, so the walk below is unchanged.
8759 let _single_table = from.joins.is_empty();
8760 // An ORDER BY that the bounded sort can serve streams; everything
8761 // else still falls to the materialising fallback below.
8762 // r1025 — an ordering the index already holds needs no sort at all.
8763 // Tried before the spill sort, which is the path it replaces.
8764 if !stmt.order_by.is_empty()
8765 && from.joins.is_empty()
8766 && let Some(n) = self.try_index_order_stream(stmt, from, cancel, emit)?
8767 {
8768 return Ok(Some(n));
8769 }
8770 if !stmt.order_by.is_empty()
8771 && from.joins.is_empty()
8772 && let Some(n) = self.try_spill_sorted_stream(stmt, from, cancel, emit)?
8773 {
8774 return Ok(Some(n));
8775 }
8776 // r1031 — integer keys carried inline instead of an `OrderKey`
8777 // vector per row. Tried AFTER the spill sort on purpose: this lane
8778 // buffers the whole answer, so anything the spill path would take
8779 // must keep taking it rather than be turned back into an in-memory
8780 // sort that answers with a budget error.
8781 if !stmt.order_by.is_empty()
8782 && from.joins.is_empty()
8783 && let Some(n) = self.try_int_key_sorted_stream(stmt, from, cancel, emit)?
8784 {
8785 return Ok(Some(n));
8786 }
8787 if !stmt.order_by.is_empty()
8788 || stmt.limit.is_some()
8789 || stmt.offset.is_some()
8790 || stmt.having.is_some()
8791 || stmt.group_by.is_some()
8792 || stmt.distinct
8793 || !stmt.unions.is_empty()
8794 || stmt.limit_with_ties
8795 {
8796 return Ok(None);
8797 }
8798 if aggregate::uses_aggregate(stmt) {
8799 return Ok(None);
8800 }
8801 // No window / SRF on the streaming path.
8802 if select_has_window(stmt) {
8803 return Ok(None);
8804 }
8805 if stmt
8806 .items
8807 .iter()
8808 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
8809 {
8810 return Ok(None);
8811 }
8812 // v7.37 (round 831) — a joinless FROM over a plain stored table
8813 // never needs the deferred structure, and building one costs the
8814 // whole table. `materialise_table_ref_filtered` clones every row
8815 // into a `Vec<Row<'static>>` before anything is filtered or
8816 // projected, so peak cost tracks the TABLE, not the result:
8817 // measured over 300k rows of 200 bytes, `SELECT id FROM big` and
8818 // `SELECT pad FROM big` both cost +107 MB over baseline, the narrow
8819 // projection saving nothing, while an arithmetic projection — which
8820 // the shape gates decline, so it materialises through the ordinary
8821 // executor — cost +21 MB.
8822 //
8823 // Scanning in batches and releasing each one is what `cursor_fill`
8824 // already does for a lazy cursor, and it is the same walk: resume
8825 // from a slot, take visible rows, evaluate, hand them over, drop
8826 // them. Round 800's finding stands and is why this reads rows OUT
8827 // rather than seeding the join by index — touching the stored
8828 // `PersistentVec` in place makes the whole table resident, which is
8829 // worse than the copy. Each batch is copied, then freed.
8830 if from.joins.is_empty()
8831 && from.primary.unnest_expr.is_none()
8832 && from.primary.lateral_subquery.is_none()
8833 && from.primary.as_of_segment.is_none()
8834 && from.primary.generate_series_args.is_none()
8835 && let Some(n) = self.try_stream_single_table(stmt, from, cancel, emit)?
8836 {
8837 return Ok(Some(n));
8838 }
8839 // Build the deferred join under the regular byte budget.
8840 let mut budget = ByteBudget::new(self.max_query_bytes);
8841 let deferred = {
8842 let mut needed = alloc::collections::BTreeSet::new();
8843 let prunable = collect_qualified_refs(stmt, &mut needed).is_some();
8844 self.build_joined_filtered_rows(
8845 from,
8846 stmt.where_.as_ref(),
8847 cancel,
8848 if prunable { Some(&needed) } else { None },
8849 &mut budget,
8850 )?
8851 };
8852 let combined_schema = &deferred.combined_schema;
8853 // v7.39 (read01 round 53) — carry the catalog (see join.rs): a
8854 // `::regclass` / enum cast in a joined projection or HAVING needs it.
8855 // v7.39 (round 525) — and the session: a joined SELECT's WHERE is
8856 // the same predicate the unjoined shape carries.
8857 let joined_sess = self.dml_session();
8858 let ctx = EvalContext::new(combined_schema, None)
8859 .with_catalog(self.active_catalog())
8860 .with_session(&joined_sess);
8861 let projection =
8862 build_projection(&stmt.items, combined_schema, "", self.backslash_escapes)?;
8863 // Every projection item must be a bound qualified column —
8864 // anything that needs `eval_expr_with_correlated` keeps the
8865 // materialising path.
8866 let bound_pos = |e: &Expr| -> Option<usize> {
8867 match e {
8868 // v7.39 (round 822) — an UNQUALIFIED column resolves here
8869 // too. The `qualifier.is_some()` guard this replaces meant
8870 // `SELECT pad FROM big` — the commonest projection there is
8871 // — never reached the streaming walk: it fell out at this
8872 // gate and re-ran on the materialising path, after the
8873 // deferred join structure had already been built and paid
8874 // for. Measured (round 821, statement_timeout=120 over 400k
8875 // rows): `big.pad` and `b.pad` streamed and cancelled at
8876 // ~65k rows in 0.14 s, while bare `pad` ran to completion in
8877 // 0.80 s with the timeout never consulted. `find_column_pos`
8878 // has always handled the unqualified case (it falls through
8879 // to a by-name match), so the guard narrowed the gate for no
8880 // reason it recorded.
8881 Expr::Column(c) => eval::find_column_pos(c, &ctx),
8882 _ => None,
8883 }
8884 };
8885 let proj_decomposed: Vec<(usize, usize)> = {
8886 let mut out = Vec::with_capacity(projection.len());
8887 for p in &projection {
8888 let Some(abs) = bound_pos(&p.expr) else {
8889 return Ok(None);
8890 };
8891 let Some(k) = deferred
8892 .offsets
8893 .partition_point(|&o| o <= abs)
8894 .checked_sub(1)
8895 else {
8896 return Ok(None);
8897 };
8898 out.push((k, abs - deferred.offsets[k]));
8899 }
8900 out
8901 };
8902 // Emit columns once.
8903 let columns: Vec<ColumnSchema> = projection
8904 .iter()
8905 // v7.39 (read01 round 54) — keep the column's enum identity through
8906 // the projection (it lives outside the DataType lattice), or a
8907 // derived table / UNION / windowed result forgets it and any outer
8908 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
8909 .map(|p| {
8910 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8911 c.user_enum_type = p.user_enum_type.clone();
8912 c.mysql_fsp = p.mysql_fsp;
8913 c
8914 })
8915 .collect();
8916 emit(crate::StreamItem::Header(&columns))?;
8917 let sources_ref = &deferred.sources;
8918 let stride = deferred.stride;
8919 let survivors_ref = &deferred.survivors;
8920 let n_surv = if stride == 0 {
8921 0
8922 } else {
8923 survivors_ref.len() / stride
8924 };
8925 // Reused per-row cell-ref scratch — pushes are zero-alloc
8926 // after the first row.
8927 let null_value = Value::Null;
8928 let mut cell_refs: Vec<&Value> = Vec::with_capacity(projection.len());
8929 let mut count: usize = 0;
8930 for surv_i in 0..n_surv {
8931 if surv_i.is_multiple_of(256) {
8932 cancel.check()?;
8933 }
8934 let tuple = &survivors_ref[surv_i * stride..(surv_i + 1) * stride];
8935 cell_refs.clear();
8936 for &(k, col_in_src) in &proj_decomposed {
8937 let ri = tuple[k];
8938 let v: &Value = if ri == usize::MAX {
8939 &null_value
8940 } else {
8941 sources_ref[k]
8942 .get(ri)
8943 .and_then(|r| r.values.get(col_in_src))
8944 .unwrap_or(&null_value)
8945 };
8946 cell_refs.push(v);
8947 }
8948 emit(crate::StreamItem::Row(crate::RowCells::Refs(&cell_refs)))?;
8949 count += 1;
8950 }
8951 Ok(Some(count))
8952 }
8953
8954 fn exec_joined_select(
8955 &self,
8956 stmt: &SelectStatement,
8957 from: &FromClause,
8958 cancel: CancelToken<'_>,
8959 ) -> Result<QueryResult, EngineError> {
8960 // v7.37.x (docker-fair NOTEX attack) — short-circuit COUNT(*)
8961 // over a LEFT ANTI JOIN. The v7.37.27 NOT EXISTS pullup
8962 // rewrites `SELECT COUNT(*) FROM A WHERE NOT EXISTS (SELECT 1
8963 // FROM B WHERE B.k = A.k)` into
8964 // SELECT COUNT(*) FROM A LEFT JOIN B ON B.k = A.k
8965 // WHERE B.k IS NULL
8966 // The general join executor builds a hash, probes every outer
8967 // tuple, materialises (left_padded_with_null) for every miss,
8968 // then runs the aggregate over the result set. For COUNT(*) we
8969 // only need the count — skip the tuple materialisation. Build
8970 // a HashSet of B's unique join values, scan A's PK index, and
8971 // increment the counter on each miss. PG's Merge Anti-Join
8972 // does roughly this; ours becomes a simple HashSet probe.
8973 if let Some(out) = self.try_count_star_left_anti_join_fast(stmt, from)? {
8974 return Ok(out);
8975 }
8976 // v7.34.5 (mailrs prod #5) — walker-driven join + early stop.
8977 // When ORDER BY is on an indexed primary column, walking the
8978 // btree in the requested direction lets the streamer break
8979 // after `LIMIT + OFFSET` survivors without ever materialising
8980 // the rest of the join — the 80 ms `mailrs_prod_not_exists`
8981 // plateau is exactly this shape.
8982 if let Some(out) = self.try_streamed_inner_join_walk_topn(stmt, from, cancel)? {
8983 return Ok(out);
8984 }
8985 // v7.30.3 (mailrs round-26) — the bounded single-join path
8986 // first; peak memory scales with LIMIT instead of the table.
8987 if let Some(out) = self.try_streamed_inner_join_topn(stmt, from, cancel)? {
8988 return Ok(out);
8989 }
8990 // v7.17.0 Phase 3.P0-43 + P0-41 — delegate the join +
8991 // WHERE materialisation to the shared helper so the LATERAL
8992 // / UNNEST / regular-catalog paths route through one place.
8993 // (`build_joined_filtered_rows` carries LATERAL support as
8994 // of Phase 3.P0-41.) Downstream we still handle aggregate /
8995 // projection / ORDER BY / DISTINCT / LIMIT inline because
8996 // those depend on the SelectStatement's items list.
8997 let mut budget = ByteBudget::new(self.max_query_bytes);
8998 let deferred = {
8999 let mut needed = alloc::collections::BTreeSet::new();
9000 let prunable = collect_qualified_refs(stmt, &mut needed).is_some();
9001 self.build_joined_filtered_rows(
9002 from,
9003 stmt.where_.as_ref(),
9004 cancel,
9005 if prunable { Some(&needed) } else { None },
9006 &mut budget,
9007 )?
9008 };
9009 let combined_schema = &deferred.combined_schema;
9010 // v7.39 (read01 round 53) — carry the catalog (see join.rs): a
9011 // `::regclass` / enum cast in a joined projection or HAVING needs it.
9012 // v7.39 (round 525) — and the session: a joined SELECT's WHERE is
9013 // the same predicate the unjoined shape carries.
9014 let joined_sess = self.dml_session();
9015 let ctx = EvalContext::new(combined_schema, None)
9016 .with_catalog(self.active_catalog())
9017 .with_session(&joined_sess);
9018 // Aggregate path: handle GROUP BY / aggregate calls over the
9019 // joined+filtered rows.
9020 if aggregate::uses_aggregate(stmt) {
9021 // v7.32 (P4 borrow channel, increment 2) — borrow each
9022 // surviving join tuple as a RowRef::Tuple; the aggregate
9023 // engine reads source cells by reference (bound fast path =
9024 // zero clone) instead of consuming materialised combined
9025 // Rows. This is where the +211k materialise_tuple_vals
9026 // clones disappear for the join+aggregate shape.
9027 let refs = deferred.row_refs();
9028 // v7.29 — a per-query memo so correlated scalar
9029 // subqueries batch-evaluate once (group map) instead of
9030 // executing per group.
9031 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
9032 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
9033 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
9034 .map_err(|err| match err {
9035 EngineError::Eval(ev) => ev,
9036 other => eval::EvalError::TypeMismatch {
9037 detail: alloc::format!("{other}"),
9038 },
9039 })
9040 };
9041 let agg = aggregate::run(
9042 stmt,
9043 crate::join::AggRows::Refs(&refs),
9044 combined_schema,
9045 None,
9046 Some(&agg_correlated),
9047 self.parallel_runner.0.as_deref(),
9048 Some(self.active_catalog()),
9049 Some(self),
9050 )?;
9051 return self.finish_agg_result(agg, stmt, cancel);
9052 }
9053
9054 let projection =
9055 build_projection(&stmt.items, combined_schema, "", self.backslash_escapes)?;
9056 // v7.39 (round 734) — a set-returning projection over a JOIN.
9057 // This executor's projection loop treats every item as a scalar,
9058 // so `SELECT unnest(ARRAY[a.id, b.g]) FROM a JOIN b …` died with
9059 // "function unnest(integer[]) does not exist" where PG expands
9060 // it. The row-set executor already carries the full SRF pipeline
9061 // (lockstep expansion, ORDER-BY-on-expanded-rows, the round-733
9062 // sharding): materialise the joined survivors and hand over. The
9063 // WHERE is cleared — the join already applied it, and combined
9064 // columns resolve identically in both executors.
9065 if !self.srf_target_idxs(&projection).is_empty() {
9066 let refs = deferred.row_refs();
9067 let rows: Vec<Row<'static>> = refs.iter().map(|r| r.as_row().into_owned()).collect();
9068 let mut s2 = stmt.clone();
9069 s2.where_ = None;
9070 let schema = combined_schema.clone();
9071 return self.exec_select_over_rows(&s2, rows, schema, "", cancel);
9072 }
9073 // v7.33 (P4 borrow channel, increment 3) — project directly off
9074 // the deferred row-index tuples instead of materialising an
9075 // intermediate combined Row per survivor. A bound qualified
9076 // column is read by reference (`RowRef::get` → `tuple_value`) and
9077 // cloned ONCE into the output row; the old `materialise()` (a full
9078 // combined Row plus a source→intermediate clone per referenced
9079 // cell, for every survivor) is gone. A row materialises on demand
9080 // only when a projection or ORDER BY expression needs the eval
9081 // path (subquery / function / arithmetic / unqualified column).
9082 // Same bind-once classification the aggregate input fast path uses
9083 // (`accumulate_groups`), reading the same `tuple_value` mapping the
9084 // differential gate already covers.
9085 let refs = deferred.row_refs();
9086 let bound_pos = |e: &Expr| -> Option<usize> {
9087 match e {
9088 Expr::Column(c) if c.qualifier.is_some() => eval::find_column_pos(c, &ctx),
9089 _ => None,
9090 }
9091 };
9092 let proj_pos: Vec<Option<usize>> = projection.iter().map(|p| bound_pos(&p.expr)).collect();
9093 let all_proj_bound = proj_pos.iter().all(Option::is_some);
9094 // v7.36 (perf — mailrs Phase 1, PROJ SPGS 8.93 → ?) —
9095 // pre-decompose each bound projection position into
9096 // `(source_k, col_in_source)` so the per-row column read
9097 // skips the per-cell `tuple_value` partition_point + slice
9098 // walk. For PROJ_25k (5 cols × 25k rows = 125k tuple_value
9099 // calls) that walk dominated; this version reaches into
9100 // `pipe.sources[k].get(tuple[k])?.values[col]` directly.
9101 let proj_decomposed: Vec<Option<(usize, usize)>> = proj_pos
9102 .iter()
9103 .map(|p| {
9104 p.and_then(|abs| {
9105 let k = deferred
9106 .offsets
9107 .partition_point(|&o| o <= abs)
9108 .checked_sub(1)?;
9109 Some((k, abs - deferred.offsets[k]))
9110 })
9111 })
9112 .collect();
9113 // v7.39 (round 962) — which projection items are whole-row
9114 // references, and to which join source. The test is
9115 // `locate_column` declining the name, which is the SAME resolver
9116 // the evaluation path uses, so this cannot drift from it: a real
9117 // column carrying an alias's name resolves to a position and is
9118 // not reported here. The source index comes from the alias
9119 // prefix, the way the combined schema names its columns.
9120 let whole_row_src: Vec<Option<usize>> = projection
9121 .iter()
9122 .map(|p| {
9123 let Expr::Column(c) = &p.expr else {
9124 return None;
9125 };
9126 if !matches!(eval::locate_column(c, &ctx), Ok(None)) {
9127 return None;
9128 }
9129 let prefix = alloc::format!("{name}.", name = c.name);
9130 let abs = deferred
9131 .combined_schema
9132 .iter()
9133 .position(|s| s.name.starts_with(&prefix))?;
9134 deferred
9135 .offsets
9136 .partition_point(|&o| o <= abs)
9137 .checked_sub(1)
9138 })
9139 .collect();
9140 // ORDER BY (when present) still evaluates against a materialised
9141 // Row — keep the order-key encoder correct rather than fork it.
9142 let need_eval_row = !all_proj_bound || !stmt.order_by.is_empty();
9143 let mut tagged: Vec<(Vec<OrderKey>, Row<'static>)> = Vec::new();
9144 let mut proj_memo = memoize::MemoizeCache::default();
9145 let sources_ref = &deferred.sources;
9146 let stride = deferred.stride;
9147 let survivors_ref = &deferred.survivors;
9148 let n_surv = survivors_ref.len() / stride.max(1);
9149 // v7.38 (read01 B8) — streaming top-N budget (see the sibling
9150 // single-table path). Bounds this JOIN projection's accumulator
9151 // to O(keep) for `ORDER BY … LIMIT k`.
9152 let topk_stream: Option<(usize, Vec<bool>)> = if !stmt.order_by.is_empty()
9153 && !stmt.distinct
9154 && !stmt.limit_with_ties
9155 && !self.env_cfg().disable_topk
9156 {
9157 stmt.limit_literal().and_then(|l| {
9158 let keep = (l as usize).saturating_add(stmt.offset_literal().unwrap_or(0) as usize);
9159 (keep >= 1).then(|| (keep, stmt.order_by.iter().map(|o| o.desc).collect()))
9160 })
9161 } else {
9162 None
9163 };
9164 // v7.37.16 — streaming DISTINCT seen-set (see scan-path twin).
9165 let mut seen_distinct: hashbrown::HashMap<u64, crate::distinct::DistinctBucket> =
9166 hashbrown::HashMap::new();
9167 let distinct_hb = hashbrown::DefaultHashBuilder::default();
9168 for surv_i in 0..n_surv {
9169 let tuple = &survivors_ref[surv_i * stride..(surv_i + 1) * stride];
9170 let row = &refs[surv_i];
9171 let materialised: Option<Cow<'_, Row<'static>>> = if need_eval_row {
9172 Some(row.as_row())
9173 } else {
9174 None
9175 };
9176 let mut values = Vec::with_capacity(projection.len());
9177 for (i, p) in projection.iter().enumerate() {
9178 if let Some((k, col_in_src)) = proj_decomposed[i] {
9179 // v7.36 — direct (source_k, col) lookup, no
9180 // partition_point. tuple[k] is the row index in
9181 // sources[k]; LEFT-NULL slots are `usize::MAX`.
9182 let ri = tuple[k];
9183 let v: Value<'static> = if ri == usize::MAX {
9184 Value::Null
9185 } else {
9186 sources_ref[k]
9187 .get(ri)
9188 .and_then(|r| r.values.get(col_in_src))
9189 .cloned()
9190 .map(Value::into_owned)
9191 .unwrap_or(Value::Null)
9192 };
9193 values.push(v);
9194 } else if let Some(pos) = proj_pos[i] {
9195 // Bound but couldn't decompose (shouldn't normally
9196 // happen — keep as a safe path).
9197 values.push(
9198 row.get(pos)
9199 .cloned()
9200 .map(Value::into_owned)
9201 .unwrap_or(Value::Null),
9202 );
9203 } else if let Some(k) = whole_row_src[i]
9204 && tuple[k] == usize::MAX
9205 {
9206 // v7.39 (round 962) — a whole-row reference to a side
9207 // an OUTER join null-extended is NULL, not a
9208 // composite whose fields are all NULL. PG18.4 answers
9209 // `SELECT jb FROM wr LEFT JOIN jb ON <no match>` with
9210 // an empty cell; round 961 answered `(,)`.
9211 //
9212 // The evaluator below cannot tell the two apart: it
9213 // reads the MATERIALISED combined row, where a
9214 // null-extended side is indistinguishable from a real
9215 // row whose every column is NULL — and that row is
9216 // `(,)` in PG too, so guessing by "all fields NULL"
9217 // would trade one wrong answer for another. The
9218 // tuple, which is still in hand here, does know:
9219 // `usize::MAX` is the sentinel the join writes for
9220 // exactly this.
9221 values.push(Value::Null);
9222 } else {
9223 // Eval path — `materialised` is Some whenever any
9224 // projection item is non-bound (need_eval_row true).
9225 // v7.24 (round-16 B) — select-list subqueries under a
9226 // JOIN go through the correlated-aware evaluator too.
9227 let mrow = materialised.as_deref().expect("materialised for eval");
9228 values.push(self.eval_expr_with_correlated(
9229 &p.expr,
9230 mrow,
9231 &ctx,
9232 cancel,
9233 Some(&mut proj_memo),
9234 )?);
9235 }
9236 }
9237 let out_row = Row::new(values);
9238 // v7.37.16 — streaming DISTINCT (see the scan-path twin):
9239 // probe on the projected row; duplicates skip the
9240 // build_order_keys eval and never enter `tagged`.
9241 if stmt.distinct {
9242 let bucket = seen_distinct
9243 .entry(norm_hash_row(&out_row, &distinct_hb, ctx.mysql_dialect))
9244 .or_default();
9245 if bucket
9246 .iter()
9247 .any(|i| row_eq_norm(&tagged[i].1, &out_row, ctx.mysql_dialect))
9248 {
9249 continue;
9250 }
9251 bucket.push(tagged.len());
9252 }
9253 let order_keys = if stmt.order_by.is_empty() {
9254 Vec::new()
9255 } else {
9256 let mrow = materialised.as_deref().expect("materialised for order by");
9257 build_order_keys(&stmt.order_by, mrow, &ctx)?
9258 };
9259 budget.charge(approx_row_bytes(&out_row))?;
9260 tagged.push((order_keys, out_row));
9261 if let Some((k, descs)) = &topk_stream {
9262 topk_trim(&mut tagged, *k, descs);
9263 }
9264 }
9265 if !stmt.order_by.is_empty() {
9266 // v7.38 元机制 D acceptor — see other call site above.
9267 let keep = if self.env_cfg().disable_topk {
9268 None
9269 } else {
9270 stmt.limit_literal()
9271 .map(|l| l as usize + stmt.offset_literal().map_or(0, |o| o as usize))
9272 };
9273 let descs: Vec<bool> = stmt.order_by.iter().map(|o| o.desc).collect();
9274 // v7.39 (round 688) — the join's ORDER BY resolves its keys
9275 // against `ctx`, which is built from `build_combined_schema`, so
9276 // this is where a declared collation reaches the sort. There was
9277 // exactly ONE resolver call in the engine before this — the
9278 // single-table scan's — which is why every other shape sorted by
9279 // bytes no matter what the schemas carried.
9280 let colls = crate::orderby::order_by_collations(&stmt.order_by, &ctx)?;
9281 crate::orderby::partial_sort_tagged_in(&mut tagged, keep, &descs, &colls);
9282 }
9283 let mut output_rows: Vec<Row<'static>> = tagged.into_iter().map(|(_, r)| r).collect();
9284 apply_offset_and_limit(
9285 &mut output_rows,
9286 stmt.offset_literal(),
9287 stmt.limit_literal(),
9288 );
9289 let columns: Vec<ColumnSchema> = projection
9290 .into_iter()
9291 .map(|p| {
9292 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
9293 c.user_enum_type = p.user_enum_type;
9294 c.collation_name = p.collation_name;
9295 c.mysql_fsp = p.mysql_fsp;
9296 c
9297 })
9298 .collect();
9299 Ok(QueryResult::Rows {
9300 columns,
9301 rows: output_rows,
9302 })
9303 }
9304}
9305
9306impl Engine {
9307 /// v6.10.2 — cold-tier time-travel scan. Resolves the segment
9308 /// by id, decodes each row body against the table's current
9309 /// schema, applies the SELECT's projection + optional WHERE +
9310 /// optional LIMIT, returns a `Rows` result. JOINs / aggregates
9311 /// / ORDER BY are unsupported on this path (STABILITY carve-
9312 /// out); operators wanting them should restore the segment
9313 /// into a regular table first.
9314 fn exec_select_as_of_segment(
9315 &self,
9316 stmt: &SelectStatement,
9317 from: &spg_sql::ast::FromClause,
9318 segment_id: u32,
9319 ) -> Result<QueryResult, EngineError> {
9320 // v6.10.2 scope: no joins, no aggregates, no ORDER BY,
9321 // no GROUP BY / HAVING / UNION / OFFSET / DISTINCT.
9322 if !from.joins.is_empty()
9323 || stmt.group_by.is_some()
9324 || stmt.having.is_some()
9325 || !stmt.unions.is_empty()
9326 || !stmt.order_by.is_empty()
9327 || stmt.offset.is_some()
9328 || stmt.distinct
9329 || aggregate::uses_aggregate(stmt)
9330 {
9331 return Err(EngineError::Unsupported(
9332 "AS OF SEGMENT supports SELECT projection + WHERE + LIMIT only \
9333 (joins / aggregates / ORDER BY are STABILITY § \"Out of v6.10\")"
9334 .into(),
9335 ));
9336 }
9337 let table = self
9338 .active_catalog()
9339 .get(&from.primary.name)
9340 .ok_or_else(|| StorageError::TableNotFound {
9341 name: from.primary.name.clone(),
9342 })?;
9343 let schema = table.schema().clone();
9344 let schema_cols = &schema.columns;
9345 let alias = from
9346 .primary
9347 .alias
9348 .as_deref()
9349 .unwrap_or(from.primary.name.as_str());
9350 let ctx = self.ev_ctx(schema_cols, Some(alias));
9351 let seg = self
9352 .active_catalog()
9353 .cold_segment(segment_id)
9354 .ok_or_else(|| {
9355 EngineError::Unsupported(alloc::format!(
9356 "AS OF SEGMENT: cold segment {segment_id} not registered"
9357 ))
9358 })?;
9359 let mut out_rows: Vec<Row<'static>> = Vec::new();
9360 let mut limit_remaining: Option<usize> =
9361 stmt.limit_literal().and_then(|n| usize::try_from(n).ok());
9362 for (_key, body) in seg.scan() {
9363 let (row, _consumed) =
9364 spg_storage::decode_row_body_dense(&body, &schema, seg.codec_version())
9365 .map_err(EngineError::Storage)?;
9366 if let Some(where_expr) = &stmt.where_ {
9367 let cond = self.eval_expr_simple(where_expr, &row, &ctx)?;
9368 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
9369 continue;
9370 }
9371 }
9372 // Projection.
9373 let projected = self.project_row_simple(&row, &stmt.items, schema_cols, alias)?;
9374 out_rows.push(projected);
9375 if let Some(rem) = limit_remaining.as_mut() {
9376 if *rem == 0 {
9377 out_rows.pop();
9378 break;
9379 }
9380 *rem -= 1;
9381 }
9382 }
9383 // Output column schema: derive from SELECT items.
9384 let columns = self.derive_output_columns(&stmt.items, schema_cols, alias);
9385 Ok(QueryResult::Rows {
9386 columns,
9387 rows: out_rows,
9388 })
9389 }
9390
9391 /// v6.10.2 — simple-path WHERE eval that doesn't go through
9392 /// the correlated-subquery / Memoize machinery. AS OF SEGMENT
9393 /// scan paths predicate against a snapshot frozen segment, no
9394 /// cross-row state.
9395 fn eval_expr_simple(
9396 &self,
9397 expr: &Expr,
9398 row: &Row<'static>,
9399 ctx: &EvalContext,
9400 ) -> Result<Value<'static>, EngineError> {
9401 let cancel = CancelToken::none();
9402 self.eval_expr_with_correlated(expr, row, ctx, cancel, None)
9403 }
9404}
9405
9406// ---- SELECT result / projection / generate-series / SRF helpers (lib.rs split 12) ----
9407
9408/// One row-producing projection: an expression to evaluate, the resulting
9409/// column's user-visible name, its inferred type, and nullability.
9410#[derive(Debug, Clone)]
9411pub(crate) struct ProjectedItem {
9412 pub(crate) expr: Expr,
9413 pub(crate) output_name: String,
9414 pub(crate) ty: DataType,
9415 pub(crate) nullable: bool,
9416 /// v7.39 (read01 round 54) — a projected enum column keeps its enum
9417 /// identity. Enum-ness lives outside the DataType lattice (the value is a
9418 /// Text), so a projection that dropped this made the RESULT schema forget
9419 /// it — and a UNION's combined `ORDER BY <enum col>`, which sorts against
9420 /// that schema, silently fell back to TEXT order instead of member order.
9421 pub(crate) user_enum_type: Option<String>,
9422 /// v7.39 (round 425) — a projected MySQL temporal column keeps its
9423 /// declared fractional-seconds precision, so the renderer can pad to
9424 /// exactly that many digits (`DATETIME(3)` shows `.250`, and `.000` for
9425 /// a whole second). Like `user_enum_type` this lives outside the
9426 /// DataType lattice, so a projection that dropped it made the RESULT
9427 /// schema forget how wide the fraction should print.
9428 pub(crate) mysql_fsp: Option<u8>,
9429 /// v7.39 (round 688) — and its declared collation, the third thing to
9430 /// live outside the DataType lattice and the third to be lost the same
9431 /// way. Measured: `SELECT a.loc FROM a JOIN b … ORDER BY a.loc` over a
9432 /// column declared `COLLATE "en_US.utf8"` sorted by bytes, because the
9433 /// projection rebuilt the output column and the ORDER BY resolves
9434 /// against THAT schema.
9435 pub(crate) collation_name: Option<String>,
9436}
9437
9438/// Dedupe a row set, preserving first-seen order. `Row`'s `PartialEq` is
9439/// structural (`Vec<Value<'static>>` ⇒ pairwise `Value` equality), which gives SQL
9440/// `NULL = NULL → TRUE` and `NaN = NaN → FALSE`. The first agrees with
9441/// the spec's "two NULLs are not distinct"; the second is a tolerated
9442/// quirk for v1 (no NaN literals are reachable from the SQL surface).
9443/// v7.37 D.23 — is this expression a bare (non-window) aggregate call?
9444fn expr_is_aggregate_call(e: &Expr) -> bool {
9445 match e {
9446 Expr::FunctionCall { name, .. } => crate::aggregate::is_aggregate_name(name),
9447 Expr::AggregateOrdered { .. } => true,
9448 _ => false,
9449 }
9450}
9451
9452/// Collect distinct top-level aggregate call expressions (dedup by value). Does
9453/// not recurse into an aggregate's own args (it's hoisted whole). Reuses the same
9454/// pragmatic variant set as `rewrite_window_to_columns`; aggregates nested in
9455/// uncovered variants simply aren't hoisted (the query keeps erroring, no worse
9456/// than today — never a regression on a working query).
9457fn collect_agg_exprs(e: &Expr, out: &mut Vec<Expr>) {
9458 if expr_is_aggregate_call(e) {
9459 if !out.iter().any(|x| x == e) {
9460 out.push(e.clone());
9461 }
9462 return;
9463 }
9464 match e {
9465 Expr::Binary { lhs, rhs, .. } => {
9466 collect_agg_exprs(lhs, out);
9467 collect_agg_exprs(rhs, out);
9468 }
9469 Expr::Unary { expr, .. }
9470 | Expr::Cast { expr, .. }
9471 | Expr::IsNull { expr, .. }
9472 | Expr::BoolTest { expr, .. }
9473 | Expr::FieldAccess { base: expr, .. } => collect_agg_exprs(expr, out),
9474 Expr::FunctionCall { args, .. } => {
9475 for a in args {
9476 collect_agg_exprs(a, out);
9477 }
9478 }
9479 Expr::Like { expr, pattern, .. } => {
9480 collect_agg_exprs(expr, out);
9481 collect_agg_exprs(pattern, out);
9482 }
9483 Expr::Extract { source, .. } => collect_agg_exprs(source, out),
9484 Expr::WindowFunction {
9485 args,
9486 partition_by,
9487 order_by,
9488 ..
9489 } => {
9490 for a in args {
9491 collect_agg_exprs(a, out);
9492 }
9493 for p in partition_by {
9494 collect_agg_exprs(p, out);
9495 }
9496 for (o, _, _) in order_by {
9497 collect_agg_exprs(o, out);
9498 }
9499 }
9500 _ => {}
9501 }
9502}
9503
9504/// Replace each aggregate call in `aggs` with a `Column(__aggN)` reference.
9505fn replace_agg_exprs(e: &mut Expr, aggs: &[Expr]) {
9506 if expr_is_aggregate_call(e) {
9507 if let Some(idx) = aggs.iter().position(|x| x == e) {
9508 *e = Expr::Column(ColumnName {
9509 qualifier: None,
9510 name: alloc::format!("__agg{idx}"),
9511 });
9512 }
9513 return;
9514 }
9515 match e {
9516 Expr::Binary { lhs, rhs, .. } => {
9517 replace_agg_exprs(lhs, aggs);
9518 replace_agg_exprs(rhs, aggs);
9519 }
9520 Expr::Unary { expr, .. }
9521 | Expr::Cast { expr, .. }
9522 | Expr::IsNull { expr, .. }
9523 | Expr::BoolTest { expr, .. }
9524 | Expr::FieldAccess { base: expr, .. } => replace_agg_exprs(expr, aggs),
9525 Expr::FunctionCall { args, .. } => {
9526 for a in args {
9527 replace_agg_exprs(a, aggs);
9528 }
9529 }
9530 Expr::Like { expr, pattern, .. } => {
9531 replace_agg_exprs(expr, aggs);
9532 replace_agg_exprs(pattern, aggs);
9533 }
9534 Expr::Extract { source, .. } => replace_agg_exprs(source, aggs),
9535 Expr::WindowFunction {
9536 args,
9537 partition_by,
9538 order_by,
9539 ..
9540 } => {
9541 for a in args {
9542 replace_agg_exprs(a, aggs);
9543 }
9544 for p in partition_by {
9545 replace_agg_exprs(p, aggs);
9546 }
9547 for (o, _, _) in order_by {
9548 replace_agg_exprs(o, aggs);
9549 }
9550 }
9551 _ => {}
9552 }
9553}
9554
9555/// v7.37 D.23 — window functions run AFTER GROUP BY aggregation. Rewrite
9556/// `SELECT g, sum(v), rank() OVER (ORDER BY sum(v)) FROM t GROUP BY g` into an
9557/// aggregate derived subquery (`SELECT g, sum(v) AS __agg0 FROM t GROUP BY g`) +
9558/// an outer window query over it (`SELECT g, __agg0, rank() OVER (ORDER BY
9559/// __agg0) FROM (...) __aggwin`), which the window-over-derived path (D.13) runs.
9560/// Returns None outside the bounded subset (leaves current behaviour). Only fires
9561/// on the currently-erroring agg+window+GROUP BY shape → cannot regress working
9562/// window-only / aggregate-only queries.
9563fn rewrite_agg_before_window(stmt: &SelectStatement) -> Option<SelectStatement> {
9564 if !(crate::aggregate::uses_aggregate(stmt) || stmt.group_by.is_some()) {
9565 return None;
9566 }
9567 // Bounded subset: no set-ops; GROUP BY keys must be simple columns.
9568 if !stmt.unions.is_empty() {
9569 return None;
9570 }
9571 let group_cols: Vec<Expr> = stmt.group_by.clone().unwrap_or_default();
9572 if group_cols.iter().any(|g| !matches!(g, Expr::Column(_))) {
9573 return None;
9574 }
9575 stmt.from.as_ref()?;
9576 // Collect the aggregate calls to hoist from projection + outer ORDER BY.
9577 let mut aggs: Vec<Expr> = Vec::new();
9578 for item in &stmt.items {
9579 if let SelectItem::Expr { expr, .. } = item {
9580 collect_agg_exprs(expr, &mut aggs);
9581 }
9582 }
9583 for ob in &stmt.order_by {
9584 collect_agg_exprs(&ob.expr, &mut aggs);
9585 }
9586 // Inner aggregate subquery: group cols (by name) + each aggregate as __aggN.
9587 let mut inner_items: Vec<SelectItem> = Vec::new();
9588 for g in &group_cols {
9589 inner_items.push(SelectItem::Expr {
9590 expr: g.clone(),
9591 alias: None,
9592 });
9593 }
9594 for (i, a) in aggs.iter().enumerate() {
9595 inner_items.push(SelectItem::Expr {
9596 expr: a.clone(),
9597 alias: Some(alloc::format!("__agg{i}")),
9598 });
9599 }
9600 let inner = SelectStatement {
9601 items: inner_items,
9602 distinct: false,
9603 distinct_on: Vec::new(),
9604 unions: Vec::new(),
9605 order_by: Vec::new(),
9606 limit: None,
9607 offset: None,
9608 limit_with_ties: false,
9609 window_check_exprs: Vec::new(),
9610 ..stmt.clone()
9611 };
9612 let derived = TableRef {
9613 name: "__aggwin".into(),
9614 alias: Some("__aggwin".into()),
9615 only: false,
9616 as_of_segment: None,
9617 unnest_expr: None,
9618 unnest_column_aliases: Vec::new(),
9619 with_ordinality: false,
9620 generate_series_args: None,
9621 lateral_subquery: Some(alloc::boxed::Box::new(inner)),
9622 jsonb_each_text_arg: None,
9623 table_fn_call: None,
9624 rows_from: None,
9625 json_table: None,
9626 scalar_fn_item: false,
9627 };
9628 // Outer window query over the derived rows: aggregates → __aggN column refs.
9629 let mut outer_items = stmt.items.clone();
9630 for item in &mut outer_items {
9631 if let SelectItem::Expr { expr, alias } = item {
9632 // Preserve PG's column label for a bare aggregate projection.
9633 if alias.is_none()
9634 && let Expr::FunctionCall { name, .. } = expr
9635 && crate::aggregate::is_aggregate_name(name)
9636 {
9637 *alias = Some(name.to_ascii_lowercase());
9638 }
9639 replace_agg_exprs(expr, &aggs);
9640 }
9641 }
9642 let mut outer_order = stmt.order_by.clone();
9643 for ob in &mut outer_order {
9644 replace_agg_exprs(&mut ob.expr, &aggs);
9645 }
9646 let mut outer_distinct_on = stmt.distinct_on.clone();
9647 for e in &mut outer_distinct_on {
9648 replace_agg_exprs(e, &aggs);
9649 }
9650 Some(SelectStatement {
9651 locking: None,
9652 ctes: Vec::new(),
9653 distinct: stmt.distinct,
9654 distinct_on: outer_distinct_on,
9655 items: outer_items,
9656 from: Some(FromClause {
9657 primary: derived,
9658 joins: Vec::new(),
9659 }),
9660 where_: None,
9661 group_by: None,
9662 group_by_all: false,
9663 having: None,
9664 unions: Vec::new(),
9665 order_by: outer_order,
9666 limit: stmt.limit.clone(),
9667 offset: stmt.offset.clone(),
9668 limit_with_ties: stmt.limit_with_ties,
9669 window_check_exprs: Vec::new(),
9670 })
9671}
9672
9673/// v7.39 (round 591) — the right-hand side of a set operation, bucketed for
9674/// membership.
9675///
9676/// INTERSECT, EXCEPT and their ALL forms all ask "is this left row over
9677/// there?", and all four answered by scanning the whole right side once per
9678/// left row. The cost was (left rows x right rows), which is why
9679/// `500k INTERSECT 1000` took 1.67 s while the same two inputs the other way
9680/// round took 20 ms: a left row that MATCHES stops the scan early, and a left
9681/// row that does not pays for all of it. Over 100k left rows, raising the
9682/// right side from 100 to 10,000 took 35 ms to 2848.
9683///
9684/// This is the shape round 485 already solved for DISTINCT, and it reuses
9685/// that machinery: bucket by `norm_hash_row`, whose only guarantee is the one
9686/// needed here — rows `row_eq_norm` calls equal hash the same — and settle
9687/// every bucket with the exact comparator, so a collision costs time and
9688/// never an answer.
9689struct PeerIndex<'r> {
9690 bh: hashbrown::DefaultHashBuilder,
9691 buckets: hashbrown::HashMap<u64, Vec<usize>>,
9692 rows: &'r [Row<'static>],
9693 mysql: bool,
9694}
9695
9696impl<'r> PeerIndex<'r> {
9697 fn build(rows: &'r [Row<'static>], mysql: bool) -> Self {
9698 // ONE hasher for the whole pass: the default builder is seeded per
9699 // instance, so a fresh one per row would put equal rows in different
9700 // buckets.
9701 let bh = hashbrown::DefaultHashBuilder::default();
9702 let mut buckets: hashbrown::HashMap<u64, Vec<usize>> =
9703 hashbrown::HashMap::with_capacity(rows.len());
9704 for (i, r) in rows.iter().enumerate() {
9705 buckets
9706 .entry(norm_hash_row(r, &bh, mysql))
9707 .or_default()
9708 .push(i);
9709 }
9710 Self {
9711 bh,
9712 buckets,
9713 rows,
9714 mysql,
9715 }
9716 }
9717
9718 fn contains(&self, r: &Row<'static>) -> bool {
9719 let h = norm_hash_row(r, &self.bh, self.mysql);
9720 self.buckets
9721 .get(&h)
9722 .is_some_and(|b| b.iter().any(|&i| row_eq_norm(&self.rows[i], r, self.mysql)))
9723 }
9724
9725 /// Remove ONE occurrence, so the multiset forms cancel row for row the
9726 /// way the pool they replaced did.
9727 fn take_one(&mut self, r: &Row<'static>) -> bool {
9728 let h = norm_hash_row(r, &self.bh, self.mysql);
9729 let Some(b) = self.buckets.get_mut(&h) else {
9730 return false;
9731 };
9732 let Some(pos) = b
9733 .iter()
9734 .position(|&i| row_eq_norm(&self.rows[i], r, self.mysql))
9735 else {
9736 return false;
9737 };
9738 b.swap_remove(pos);
9739 true
9740 }
9741}
9742
9743pub(crate) fn dedup_rows(rows: Vec<Row<'static>>, mysql: bool) -> Vec<Row<'static>> {
9744 dedup_by_row(rows, |r| r, mysql)
9745}
9746
9747/// v7.37.16 — hash-bucketed DISTINCT. The old `out.iter().any(row_eq_norm)`
9748/// was O(n·u) — `SELECT DISTINCT v` over 50 k rows with ~39 k unique values
9749/// ran 4 SECONDS (80 µs/row) vs PG's ~5 ms. Bucket rows by `norm_hash_row`
9750/// and run the exact `row_eq_norm` only within a bucket: first-occurrence
9751/// order is preserved, and correctness needs only the one-way guarantee
9752/// "row_eq_norm-Equal ⇒ equal hash" (collisions are re-checked exactly).
9753/// Small inputs keep the linear scan — no hasher setup for a 10-row page.
9754fn dedup_by_row<T>(items: Vec<T>, row_of: impl Fn(&T) -> &Row<'static>, mysql: bool) -> Vec<T> {
9755 if items.len() <= 32 {
9756 let mut out: Vec<T> = Vec::with_capacity(items.len());
9757 for it in items {
9758 if !out
9759 .iter()
9760 .any(|seen| row_eq_norm(row_of(seen), row_of(&it), mysql))
9761 {
9762 out.push(it);
9763 }
9764 }
9765 return out;
9766 }
9767 // ONE BuildHasher instance for the whole pass — the default builder
9768 // is randomly seeded PER INSTANCE, so a fresh one per row would give
9769 // equal rows different hashes and never dedup.
9770 let bh = hashbrown::DefaultHashBuilder::default();
9771 let mut out: Vec<T> = Vec::with_capacity(items.len().min(1024));
9772 let mut buckets: hashbrown::HashMap<u64, crate::distinct::DistinctBucket> =
9773 hashbrown::HashMap::with_capacity(items.len());
9774 for it in items {
9775 let h = norm_hash_row(row_of(&it), &bh, mysql);
9776 let bucket = buckets.entry(h).or_default();
9777 if !bucket
9778 .iter()
9779 .any(|i| row_eq_norm(row_of(&out[i]), row_of(&it), mysql))
9780 {
9781 bucket.push(out.len());
9782 out.push(it);
9783 }
9784 }
9785 out
9786}
9787
9788/// Hash companion to [`row_eq_norm`]. Guarantees only the direction dedup
9789/// needs: rows that `row_eq_norm` deems Equal hash identically; DISTINCT
9790/// rows may collide (buckets are re-checked with the exact comparator).
9791///
9792/// Domain design mirrors `value_cmp`'s equivalence classes:
9793/// - The numeric family (SmallInt/Int/BigInt/Float/Numeric/NumericBig)
9794/// shares one domain: a value that is an integer fitting i64 hashes the
9795/// i64 (so `Int(1)`, `BigInt(1)`, `Float(1.0)`, `Numeric(1.00)` agree);
9796/// anything else hashes the f64 approximation computed by THE SAME
9797/// formula the value_cmp float arms use (`numeric_to_f64`), so
9798/// `Numeric(0.5) == Float(0.5)` agree bit-for-bit. NaN (any family)
9799/// hashes a constant; ±Inf hash their f64 bits; -0.0 folds into 0.0.
9800/// Known un-closable corner: an integer in [2^53, 2^63) can compare
9801/// Equal to a float via value_cmp's lossy f64 arm while hashing in the
9802/// exact-i64 domain — mixed int/float rows at that magnitude may miss a
9803/// dedup (PG itself compares int8↔float8 in the lossy float8 domain).
9804/// - Text and BpChar share a trailing-blank-trimmed byte domain (value_cmp
9805/// compares them blank-insensitively; plain Text pairs that differ only
9806/// in trailing blanks merely collide and are separated exactly).
9807/// - Families value_cmp compares exactly (Bool/Date/Time/Timestamp/…)
9808/// hash their fields under a distinct tag.
9809/// - Everything value_cmp falls back to debug-format ordering for
9810/// (Json, arrays, vectors, geometry, ranges, …) shares one constant
9811/// bucket — degrades to the exact linear scan, never wrong.
9812fn norm_hash_row(row: &Row<'static>, bh: &hashbrown::DefaultHashBuilder, mysql: bool) -> u64 {
9813 norm_hash_values(&row.values, bh, mysql)
9814}
9815
9816/// v7.39 (round 485) — the same hash over a bare value slice, so the
9817/// DISTINCT probe can run against a reused buffer instead of demanding a
9818/// `Row` that has to be allocated first (see `values_eq_norm`).
9819fn norm_hash_values(
9820 values: &[Value<'static>],
9821 bh: &hashbrown::DefaultHashBuilder,
9822 mysql: bool,
9823) -> u64 {
9824 use core::hash::{BuildHasher, Hash, Hasher};
9825 let mut h = bh.build_hasher();
9826 for v in values {
9827 // v7.39 (round 410) — hash the folded key when the MySQL collation
9828 // deduplicates a text value, so `row_eq_norm`-equal rows (`'a'` vs
9829 // `'A'` vs `'a '`) share a hash bucket.
9830 if mysql {
9831 if let Some(folded) = mysql_dedup_fold(v) {
9832 folded.hash(&mut h);
9833 continue;
9834 }
9835 }
9836 norm_hash_value(v, &mut h);
9837 }
9838 h.finish()
9839}
9840
9841/// r1044 — `10^p` as an `i128`, or `None` past what one holds.
9842///
9843/// `i128::MAX` is about 1.7e38, so 10^38 is the last power that fits.
9844const fn pow10_i128(p: u16) -> Option<i128> {
9845 const P: [i128; 39] = {
9846 let mut t = [1i128; 39];
9847 let mut i = 1;
9848 while i < 39 {
9849 t[i] = t[i - 1] * 10;
9850 i += 1;
9851 }
9852 t
9853 };
9854 if (p as usize) < P.len() {
9855 Some(P[p as usize])
9856 } else {
9857 None
9858 }
9859}
9860
9861fn norm_hash_value<H: core::hash::Hasher>(v: &Value<'static>, h: &mut H) {
9862 const TAG_NULL: u8 = 0;
9863 const TAG_BOOL: u8 = 1;
9864 const TAG_NUM_I64: u8 = 2;
9865 const TAG_NUM_F64: u8 = 3;
9866 const TAG_TEXT: u8 = 4;
9867 const TAG_DATE: u8 = 6;
9868 const TAG_TIME: u8 = 7;
9869 const TAG_TIMESTAMP: u8 = 8;
9870 const TAG_TIMETZ: u8 = 10;
9871 const TAG_UUID: u8 = 11;
9872 const TAG_MONEY: u8 = 12;
9873 const TAG_BYTES: u8 = 13;
9874 const TAG_INTERVAL: u8 = 14;
9875 const TAG_CHAR1: u8 = 15;
9876 const TAG_OPAQUE: u8 = 255;
9877 // One shared writer for the numeric family: an integer value
9878 // representable as i64 goes exact (round-trip probe — no_std, so no
9879 // f64::trunc); otherwise the f64 approximation. -0.0 round-trips
9880 // through 0i64, folding it into 0.0 as value_cmp requires.
9881 let num_f64 = |h: &mut H, x: f64| {
9882 if x.is_nan() {
9883 h.write_u8(TAG_NUM_F64);
9884 h.write_u64(0x7ff8_dead_beef_0001); // one bucket for every NaN
9885 return;
9886 }
9887 const TWO63: f64 = 9_223_372_036_854_775_808.0;
9888 if (-TWO63..TWO63).contains(&x) {
9889 #[allow(clippy::cast_possible_truncation)]
9890 let n = x as i64;
9891 #[allow(clippy::cast_precision_loss)]
9892 if (n as f64) == x {
9893 h.write_u8(TAG_NUM_I64);
9894 h.write_i64(n);
9895 return;
9896 }
9897 }
9898 h.write_u8(TAG_NUM_F64);
9899 h.write_u64(x.to_bits());
9900 };
9901 match v {
9902 Value::Null => h.write_u8(TAG_NULL),
9903 Value::Bool(b) => {
9904 h.write_u8(TAG_BOOL);
9905 h.write_u8(u8::from(*b));
9906 }
9907 Value::SmallInt(n) => {
9908 h.write_u8(TAG_NUM_I64);
9909 h.write_i64(i64::from(*n));
9910 }
9911 Value::Int(n) => {
9912 h.write_u8(TAG_NUM_I64);
9913 h.write_i64(i64::from(*n));
9914 }
9915 Value::BigInt(n) => {
9916 h.write_u8(TAG_NUM_I64);
9917 h.write_i64(*n);
9918 }
9919 Value::Float(x) => num_f64(h, *x),
9920 Value::Numeric {
9921 scaled,
9922 scale,
9923 kind,
9924 } => match kind {
9925 spg_storage::NumericKind::NaN => num_f64(h, f64::NAN),
9926 spg_storage::NumericKind::PosInf => num_f64(h, f64::INFINITY),
9927 spg_storage::NumericKind::NegInf => num_f64(h, f64::NEG_INFINITY),
9928 spg_storage::NumericKind::Finite => {
9929 // Reduce trailing fractional zeros so 1.50 and 1.5 share a
9930 // representation, then: exact integers fitting i64 go to the
9931 // i64 domain; everything else uses numeric_to_f64 — the SAME
9932 // formula value_cmp's Numeric↔Float arm compares with.
9933 // r1044 — the reduction is required (`1.5` and `1.50` are
9934 // one value and must land in one bucket) and it used to
9935 // walk one digit at a time. That is O(scale), and scale
9936 // is not small in practice: `n / 100` on a NUMERIC
9937 // column stores `9.1900000000000000`, scale 16, so the
9938 // loop ran fourteen times PER ROW.
9939 //
9940 // Priced by ablation rather than guessed at — removing
9941 // the loop entirely took `SELECT DISTINCT n FROM t ORDER
9942 // BY n` over 400,000 rows from 52 ms to 14.8, against
9943 // PostgreSQL's 12.2-13.8. Two `pow10` lookup tables
9944 // tried first moved it not at all, which is why this one
9945 // was measured before it was written.
9946 //
9947 // Binary search over the same powers finds the whole
9948 // run of trailing zeros in at most six tests and one
9949 // division, instead of one test and one division per
9950 // digit.
9951 let (mut s, mut sc) = (*scaled, *scale);
9952 if sc > 0 && s != 0 {
9953 let mut lo: u16 = 0;
9954 let mut hi: u16 = sc;
9955 while lo < hi {
9956 let mid = (lo + hi).div_ceil(2);
9957 match pow10_i128(mid) {
9958 Some(p) if s % p == 0 => lo = mid,
9959 _ => hi = mid - 1,
9960 }
9961 }
9962 if lo > 0 {
9963 if let Some(p) = pow10_i128(lo) {
9964 s /= p;
9965 sc -= lo;
9966 }
9967 }
9968 }
9969 if sc == 0 {
9970 if let Ok(n) = i64::try_from(s) {
9971 h.write_u8(TAG_NUM_I64);
9972 h.write_i64(n);
9973 } else {
9974 num_f64(h, crate::orderby::numeric_to_f64(s, 0));
9975 }
9976 } else {
9977 num_f64(h, crate::orderby::numeric_to_f64(s, sc));
9978 }
9979 }
9980 },
9981 // Beyond-i128 NUMERIC compares exactly via numeric_bignum_cmp; a
9982 // value that also fits i128 reuses the Numeric path above so
9983 // Big(5) and Numeric(5) agree. A genuinely huge one can't equal
9984 // any i128-representable value — constant bucket is safe.
9985 Value::NumericBig(b) => match b.to_i128() {
9986 Some(s) => norm_hash_value(
9987 &Value::Numeric {
9988 scaled: s,
9989 scale: b.scale(),
9990 kind: spg_storage::NumericKind::Finite,
9991 },
9992 h,
9993 ),
9994 None => h.write_u8(TAG_OPAQUE),
9995 },
9996 // value_cmp compares Text↔BpChar blank-insensitively (both sides
9997 // trimmed), so both hash the trimmed bytes. Text pairs differing
9998 // only in trailing blanks collide and are split exactly in-bucket.
9999 Value::Text(s) | Value::BpChar(s) => {
10000 h.write_u8(TAG_TEXT);
10001 h.write(s.trim_end_matches(' ').as_bytes());
10002 }
10003 Value::Char1(c) => {
10004 h.write_u8(TAG_CHAR1);
10005 h.write_u8(*c);
10006 }
10007 Value::Date(d) => {
10008 h.write_u8(TAG_DATE);
10009 h.write_i32(*d);
10010 }
10011 Value::Time(t) => {
10012 h.write_u8(TAG_TIME);
10013 h.write_i64(*t);
10014 }
10015 Value::Timestamp(t) => {
10016 h.write_u8(TAG_TIMESTAMP);
10017 h.write_i64(*t);
10018 }
10019 Value::TimeTz { us, offset_secs } => {
10020 h.write_u8(TAG_TIMETZ);
10021 h.write_i64(*us);
10022 h.write_i32(*offset_secs);
10023 }
10024 Value::Uuid(u) => {
10025 h.write_u8(TAG_UUID);
10026 h.write(u);
10027 }
10028 Value::Money(c) => {
10029 h.write_u8(TAG_MONEY);
10030 h.write_i64(*c);
10031 }
10032 Value::Bytes(b) => {
10033 h.write_u8(TAG_BYTES);
10034 h.write(b.as_ref());
10035 }
10036 Value::Interval {
10037 months,
10038 days,
10039 micros,
10040 } => {
10041 h.write_u8(TAG_INTERVAL);
10042 h.write_i32(*months);
10043 h.write_i32(*days);
10044 h.write_i64(*micros);
10045 }
10046 // v7.37.16 — REAL joined the numeric value_cmp family (widened
10047 // to f64, same formulas as the arms), so it hashes in the shared
10048 // numeric domain: Real(1.5) must agree with Float(1.5)/Int/…
10049 // f32→f64 is exact, so equal-under-cmp implies equal bits here.
10050 Value::Real(x) => num_f64(h, f64::from(*x)),
10051 // Json (structural equality), vector families (float rendering),
10052 // arrays / geometry / net / ranges / composites (debug-format
10053 // fallback): one constant bucket — exact linear within.
10054 _ => h.write_u8(TAG_OPAQUE),
10055 }
10056}
10057
10058/// v7.38 (read01) — row equality for DISTINCT / UNION / INTERSECT / EXCEPT that
10059/// treats numerically-equal exact values as one regardless of type or scale
10060/// (`1 = 1.0 = 1.00`), matching PG (and GROUP BY). Uses the scale-aware
10061/// `orderby::value_cmp`, so `Int(1)` and `Numeric{10,1}` compare Equal; plain
10062/// `Row` `==` would keep them distinct.
10063/// v7.39 (round 410) — under the MySQL dialect a set operation / DISTINCT
10064/// deduplicates by the session collation (`utf8mb4_uca1400_ai_ci`, which is
10065/// case- and accent-insensitive and PAD SPACE): `'a'`, `'A'`, and `'a '`
10066/// collapse to one row, exactly as GROUP BY already folds its keys. Returns
10067/// the folded comparison key for a text value, None for anything else (which
10068/// keeps the byte-exact `value_cmp` path).
10069fn mysql_dedup_fold(v: &Value) -> Option<String> {
10070 match v {
10071 Value::Text(s) | Value::BpChar(s) => {
10072 Some(spg_storage::mysql_ci_fold(s.trim_end_matches(' ')))
10073 }
10074 _ => None,
10075 }
10076}
10077
10078/// v7.39 (round 485) — how many projected rows the single-table scan
10079/// builds, and how many of those the DISTINCT probe throws away again.
10080///
10081/// The round-485 profile of `SELECT DISTINCT g FROM h ORDER BY g` put
10082/// 21 % of all samples in malloc/free called straight from the scan
10083/// closure. The closure's one per-row allocation is the projected
10084/// `Vec<Value>`, and under DISTINCT most of those are discarded a few
10085/// instructions later — but "most" is a guess until it is a number, so
10086/// these count it. (Round 480 was spent acting on an inference about a
10087/// branch that turned out never to run.)
10088/// v7.39 (round 488) — reachability counters for round 487's projection
10089/// binding. The interleaved panel says round 487 costs `group_500k` 13 %,
10090/// and a never-called-function probe rules out code layout — so the
10091/// question is whether that shape reaches this code at all, which is a
10092/// number, not an inference.
10093pub static SCAN_PATH_ENTERED: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
10094pub static PROJ_DIRECT_FIRE: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
10095
10096pub static PROJ_ROW_BUILT: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
10097pub static DISTINCT_DUP_DROPPED: core::sync::atomic::AtomicU64 =
10098 core::sync::atomic::AtomicU64::new(0);
10099
10100pub(crate) fn row_eq_norm(a: &Row<'static>, b: &Row<'static>, mysql: bool) -> bool {
10101 values_eq_norm(&a.values, &b.values, mysql)
10102}
10103
10104/// v7.39 (round 485) — `row_eq_norm` over bare value slices, so the
10105/// DISTINCT probe can compare a reused projection buffer against a kept
10106/// row without building a `Row` for it.
10107pub(crate) fn values_eq_norm(a: &[Value<'static>], b: &[Value<'static>], mysql: bool) -> bool {
10108 a.len() == b.len()
10109 && a.iter().zip(b).all(|(x, y)| {
10110 if mysql {
10111 if let (Some(fx), Some(fy)) = (mysql_dedup_fold(x), mysql_dedup_fold(y)) {
10112 return fx == fy;
10113 }
10114 }
10115 crate::orderby::value_cmp(x, y) == core::cmp::Ordering::Equal
10116 })
10117}
10118
10119/// Coerce a `Value` to an `f64` sort key for ORDER BY. Numbers map directly;
10120/// NULL sorts last (treated as `+∞`); booleans are 0.0 / 1.0; text uses lex
10121/// order via the byte values; vectors are not sortable.
10122pub(crate) fn value_to_order_key(v: &Value) -> Result<OrderKey, EngineError> {
10123 // v7.37.16 — TEXT rides a FULL-precision key: carry the whole string
10124 // so values sharing a ≥6-byte common prefix (`product_001` vs
10125 // `product_002`, ISO timestamps stored as text, prefixed IDs / SKUs)
10126 // order by their exact bytes instead of the old lossy f64 coarse key.
10127 // Comparison is byte-lexicographic (see `order_key_elem_cmp`), which
10128 // matches PG's default C / binary text collation. Every other type
10129 // keeps the lossless-enough `f64` fast path below.
10130 if let Value::Text(s) = v {
10131 return Ok(OrderKey::Text(s.as_ref().into()));
10132 }
10133 // v7.39 (bpchar epic) — bpchar sorts by its blank-stripped form then
10134 // byte order (PG bpcharcmp under C collation), so mixed-pad values of
10135 // the same logical string order equal.
10136 if let Value::BpChar(s) = v {
10137 return Ok(OrderKey::Text(s.trim_end_matches(' ').into()));
10138 }
10139 // v7.38 (read01 P6.24) — jsonb sorts by PG's type-aware total order, so
10140 // carry the parsed value and compare it structurally (see
10141 // `order_key_elem_cmp`). Unparseable text falls back to a Text key.
10142 if let Value::Json(s) = v {
10143 return Ok(match crate::json::parse(s) {
10144 Ok(jv) => OrderKey::Json(jv),
10145 Err(_) => OrderKey::Text(s.as_ref().into()),
10146 });
10147 }
10148 // v7.37 — byte-orderable types PG sorts byte-wise but that have no
10149 // meaningful f64 projection. bytea/uuid/macaddr sort by their raw bytes;
10150 // inet/cidr by `[family, addr.., bits]` (family, then address, then mask),
10151 // matching PG's network ordering.
10152 match v {
10153 Value::Bytes(b) => return Ok(OrderKey::Bytes(b.as_ref().to_vec())),
10154 // v7.38 (read01, T3.C3) — arbitrary-precision NUMERIC sorts by exact value.
10155 Value::NumericBig(b) => {
10156 return Ok(OrderKey::Numeric(alloc::boxed::Box::new(
10157 spg_storage::NumericKey::from_big(b),
10158 )));
10159 }
10160 Value::Uuid(u) => return Ok(OrderKey::Bytes(u.to_vec())),
10161 Value::Macaddr(m) => return Ok(OrderKey::Bytes(m.to_vec())),
10162 Value::Macaddr8(m) => return Ok(OrderKey::Bytes(m.to_vec())),
10163 Value::PgLsn(l) => return Ok(OrderKey::Bytes(l.to_be_bytes().to_vec())),
10164 Value::Inet { family, bits, addr } | Value::Cidr { family, bits, addr } => {
10165 let mut key = alloc::vec::Vec::with_capacity(18);
10166 key.push(*family);
10167 key.extend_from_slice(addr);
10168 key.push(*bits);
10169 return Ok(OrderKey::Bytes(key));
10170 }
10171 _ => {}
10172 }
10173 // v7.38 (read01, U16) — one-dimensional arrays sort element-wise, then
10174 // shorter-first (PG: `{1} < {1,2} < {2} < {10}`). Each element carries its
10175 // own OrderKey so integer arrays sort numerically; a NULL element rides to
10176 // the end via the +INF sentinel.
10177 let inf = || OrderKey::NullBig;
10178 let arr = match v {
10179 Value::IntArray(a) => Some(
10180 a.iter()
10181 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10182 .collect(),
10183 ),
10184 Value::SmallIntArray(a) => Some(
10185 a.iter()
10186 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10187 .collect(),
10188 ),
10189 Value::BigIntArray(a) => Some(
10190 a.iter()
10191 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10192 .collect(),
10193 ),
10194 Value::BoolArray(a) => Some(
10195 a.iter()
10196 .map(|o| o.map_or_else(inf, |b| OrderKey::Int(i128::from(b))))
10197 .collect(),
10198 ),
10199 Value::TextArray(a) => Some(
10200 a.iter()
10201 .map(|o| o.as_ref().map_or_else(inf, |s| OrderKey::Text(s.clone())))
10202 .collect(),
10203 ),
10204 #[allow(clippy::cast_precision_loss)]
10205 Value::FloatArray(a) => Some(
10206 a.iter()
10207 .map(|o| o.map_or(OrderKey::NullBig, OrderKey::Num))
10208 .collect(),
10209 ),
10210 // r1040 — array elements take the same exact key their scalar
10211 // form does; an f64 projection here would order `{0.1}` against
10212 // `{0.1000000000000000001}` by luck.
10213 Value::NumericArray(a) => Some(
10214 a.iter()
10215 .map(|o| {
10216 o.map_or_else(inf, |(m, s)| {
10217 OrderKey::Numeric(alloc::boxed::Box::new(
10218 spg_storage::NumericKey::from_numeric(
10219 m,
10220 s,
10221 spg_storage::NumericKind::Finite,
10222 ),
10223 ))
10224 })
10225 })
10226 .collect(),
10227 ),
10228 Value::DateArray(a) => Some(
10229 a.iter()
10230 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10231 .collect(),
10232 ),
10233 _ => None,
10234 };
10235 if let Some(elements) = arr {
10236 return Ok(OrderKey::Array(elements));
10237 }
10238 // v7.39 (read01 round 56) — a COMPOSITE sorts field by field, left to
10239 // right, which is exactly the lexicographic element order an Array key
10240 // already gives: `(2,'b') < (9,'a')` because the leading field decides.
10241 if let Value::Composite(fields) = v {
10242 let elements = fields
10243 .iter()
10244 .map(|(_, fv)| value_to_order_key(fv))
10245 .collect::<Result<alloc::vec::Vec<_>, _>>()?;
10246 return Ok(OrderKey::Array(elements));
10247 }
10248 // v7.38 (read01 U31) — the integer-valued types carry an EXACT i128 key.
10249 // Projecting these to f64 (the historic path) silently collapses BigInt /
10250 // Timestamp / Time / TimeTz / Money values past 2^53, so `ORDER BY` gave
10251 // the wrong order for large ids and microsecond timestamps.
10252 match v {
10253 Value::SmallInt(n) => return Ok(OrderKey::Int(i128::from(*n))),
10254 Value::Int(n) => return Ok(OrderKey::Int(i128::from(*n))),
10255 Value::BigInt(n) => return Ok(OrderKey::Int(i128::from(*n))),
10256 // PG TIME/TIMESTAMP/DATE/MONEY/YEAR are ordered by their underlying
10257 // integer (days / micros / cents / calendar year); TIMETZ by the
10258 // UTC-equivalent micros (local wall - offset) so the same physical
10259 // instant in different zones sorts equal.
10260 Value::Date(d) => return Ok(OrderKey::Int(i128::from(*d))),
10261 Value::Timestamp(t) => return Ok(OrderKey::Int(i128::from(*t))),
10262 Value::Time(us) => return Ok(OrderKey::Int(i128::from(*us))),
10263 Value::Year(y) => return Ok(OrderKey::Int(i128::from(*y))),
10264 Value::TimeTz { us, offset_secs } => {
10265 return Ok(OrderKey::Int(
10266 i128::from(*us) - i128::from(*offset_secs) * 1_000_000,
10267 ));
10268 }
10269 Value::Money(c) => return Ok(OrderKey::Int(i128::from(*c))),
10270 _ => {}
10271 }
10272 let num = match v {
10273 // Callers without NULLS FIRST/LAST context (array elements,
10274 // histogram sampling) put NULL last, as before.
10275 Value::Null => return Ok(OrderKey::NullBig),
10276 // v7.17.0 Phase 3.P0-38 — range ordering is not supported
10277 // in v7.17.0 (needs lex-then-inclusivity tiebreak).
10278 Value::Range { .. } => {
10279 return Err(EngineError::Unsupported(
10280 "ORDER BY of a range value is not supported in v7.17.0".into(),
10281 ));
10282 }
10283 // v7.17.0 Phase 3.P0-39 — hstore is not orderable.
10284 Value::Hstore(_) => {
10285 return Err(EngineError::Unsupported(
10286 "ORDER BY of a hstore value is not supported".into(),
10287 ));
10288 }
10289 // v7.17.0 Phase 3.P0-40 — 2D arrays not orderable.
10290 Value::IntArray2D(_) | Value::BigIntArray2D(_) | Value::TextArray2D(_) => {
10291 return Err(EngineError::Unsupported(
10292 "ORDER BY of a 2D array is not supported in v7.17.0".into(),
10293 ));
10294 }
10295 // r1039/r1040 — the exact canonical key, not an f64 projection.
10296 //
10297 // r1039 fixed the three specials, which carry a canonical zero in
10298 // `scaled` and so all sorted as the number 0. The projection
10299 // itself was the rest of the defect: "precision losses here only
10300 // matter for tie-breaks well past 15 significant digits" was the
10301 // comment, and the measurement disagreed — f64 called
10302 // `0.1` and `0.1000000000000000001` Equal, and a stable sort then
10303 // returned them in insertion order. Three of ten values came back
10304 // in the wrong place against PG18.4.
10305 Value::Numeric {
10306 scaled,
10307 scale,
10308 kind,
10309 } => {
10310 return Ok(OrderKey::Numeric(alloc::boxed::Box::new(
10311 spg_storage::NumericKey::from_numeric(*scaled, *scale, *kind),
10312 )));
10313 }
10314 Value::Float(x) => *x,
10315 // v7.37.16 — REAL sorts by its exact f64 widening (it had no
10316 // arm and fell through to the unsupported error).
10317 Value::Real(x) => f64::from(*x),
10318 Value::Bool(b) => {
10319 if *b {
10320 1.0
10321 } else {
10322 0.0
10323 }
10324 }
10325 Value::Vector(_) | Value::Sq8Vector(_) | Value::HalfVector(_) => {
10326 return Err(EngineError::Unsupported(
10327 "ORDER BY of a raw vector column is not meaningful — use `<->`".into(),
10328 ));
10329 }
10330 // v7.37 — PG orders INTERVAL by its total time, treating a month as
10331 // 30 days (`1 hour < 90 min < 1 day < 1 mon`). Project to total micros;
10332 // f64 is exact for any interval under ~285 years, and only ORDER BY
10333 // tie-breaks past that magnitude lose precision. Matches the
10334 // min/max(interval) comparator in aggregate.rs.
10335 #[allow(clippy::cast_precision_loss)]
10336 Value::Interval {
10337 months,
10338 days,
10339 micros,
10340 } => {
10341 let total = i128::from(*months) * 30 * 86_400_000_000
10342 + i128::from(*days) * 86_400_000_000
10343 + i128::from(*micros);
10344 total as f64
10345 }
10346 Value::Json(_) => {
10347 return Err(EngineError::Unsupported(
10348 "ORDER BY of a JSON value is not supported — cast the document to text first"
10349 .into(),
10350 ));
10351 }
10352 // v7.5.0 — Value is #[non_exhaustive]; future variants need
10353 // an explicit ORDER BY mapping. Surface as Unsupported until
10354 // engine support is added.
10355 _ => {
10356 return Err(EngineError::Unsupported(
10357 "ORDER BY of this value type is not supported".into(),
10358 ));
10359 }
10360 };
10361 Ok(OrderKey::Num(num))
10362}
10363
10364/// Find the schema entry that a SELECT-list `Expr::Column` refers to.
10365/// Mirrors `resolve_column` in `eval.rs`, but returns a proper
10366/// `EngineError` so the projection-build path keeps `UnknownQualifier`
10367/// vs `ColumnNotFound` distinct.
10368/// PG's name for the physical row identity. It is reserved there — no table
10369/// can have a column called this — which is what lets `*` skip it by name.
10370pub(crate) const CTID_COLUMN: &str = "ctid";
10371
10372/// v7.39 (round 512) — PG's system columns, in the order they are appended.
10373/// All six are reserved names there, which is what lets `*` skip them and
10374/// lets a scan tell them from a user column without a flag.
10375pub(crate) const SYSTEM_COLUMNS: [&str; 6] = ["ctid", "xmin", "xmax", "cmin", "cmax", "tableoid"];
10376
10377/// Is this name one of them?
10378pub(crate) fn is_system_column(name: &str) -> bool {
10379 SYSTEM_COLUMNS.iter().any(|s| name.eq_ignore_ascii_case(s))
10380}
10381
10382/// Where the scan's appended system columns begin, if this schema carries
10383/// them: the trailing six, named in order. A catalog view with a column of
10384/// its own called `xmin` does not match, which is the point.
10385fn system_column_tail_start(cols: &[ColumnSchema]) -> Option<usize> {
10386 let start = cols.len().checked_sub(SYSTEM_COLUMNS.len())?;
10387 cols[start..]
10388 .iter()
10389 .zip(SYSTEM_COLUMNS)
10390 .all(|(c, name)| c.name.eq_ignore_ascii_case(name))
10391 .then_some(start)
10392}
10393
10394/// v7.39 (round 540) — which positions `*` must skip.
10395///
10396/// The rule stays round 512's — the synthetic columns are the trailing
10397/// six of a relation's block, matched by POSITION so a genuine `xmin`
10398/// column is not lost — but a JOINED schema names its columns
10399/// `alias.column` and lays the peers out end to end, so a peer's six sit
10400/// in the MIDDLE of the whole list. Grouping by qualifier first puts the
10401/// "trailing six" test back on the block it was written for.
10402fn synthetic_system_positions(cols: &[ColumnSchema]) -> alloc::vec::Vec<bool> {
10403 let mut skip = alloc::vec![false; cols.len()];
10404 fn qualifier(n: &str) -> Option<&str> {
10405 n.rsplit_once('.').map(|(q, _)| q)
10406 }
10407 fn bare(n: &str) -> &str {
10408 n.rsplit('.').next().unwrap_or(n)
10409 }
10410 let mut i = 0;
10411 while i < cols.len() {
10412 let q = qualifier(&cols[i].name);
10413 let mut end = i;
10414 while end < cols.len() && qualifier(&cols[end].name) == q {
10415 end += 1;
10416 }
10417 if let Some(start) = (end - i)
10418 .checked_sub(SYSTEM_COLUMNS.len())
10419 .map(|off| i + off)
10420 && cols[start..end]
10421 .iter()
10422 .zip(SYSTEM_COLUMNS)
10423 .all(|(c, name)| bare(&c.name).eq_ignore_ascii_case(name))
10424 {
10425 for s in skip.iter_mut().take(end).skip(start) {
10426 *s = true;
10427 }
10428 }
10429 i = end;
10430 }
10431 skip
10432}
10433
10434/// v7.39 (round 511) — does this statement name `ctid` anywhere it would be
10435/// read? Only then is the column materialised.
10436pub(crate) fn expr_references_ctid(e: &Expr) -> bool {
10437 let mut found = false;
10438 crate::expr_analysis::visit_expr_columns_and_subqueries(
10439 e,
10440 &mut |c| {
10441 if is_system_column(&c.name) {
10442 found = true;
10443 }
10444 },
10445 &mut |_| {},
10446 );
10447 found
10448}
10449
10450fn references_ctid(stmt: &SelectStatement) -> bool {
10451 let in_expr = expr_references_ctid;
10452 stmt.items.iter().any(|i| match i {
10453 SelectItem::Expr { expr, .. } => in_expr(expr),
10454 _ => false,
10455 }) || stmt.where_.as_ref().is_some_and(in_expr)
10456 || stmt.order_by.iter().any(|o| in_expr(&o.expr))
10457 || stmt
10458 .group_by
10459 .as_ref()
10460 .is_some_and(|g| g.iter().any(in_expr))
10461 || stmt.having.as_ref().is_some_and(in_expr)
10462}
10463
10464/// v7.39 (round 961) — the whole-row schema for `SELECT t FROM t`, which
10465/// is a name the projection has to TYPE before any row exists.
10466///
10467/// Evaluation has answered this since round T9 (`resolve_column` builds a
10468/// `Value::Composite` of every column), but the typing side below had no
10469/// such branch and raised `column "t" does not exist` first — so the
10470/// feature was unreachable through a projection. Measured against PG18.4:
10471/// `SELECT wr FROM wr` answers `(7,z)` there and errored here.
10472///
10473/// The type is `Jsonb` + a composite marker, which is exactly how a
10474/// column DECLARED as a composite type is described (`ddl.rs`, round 56):
10475/// the value travels as a `Value::Composite` and renders in the canonical
10476/// `(7,z)` form. SPG has no catalog entry for a table's implicit row type,
10477/// so the marker names the alias and no rehydration keys off it — the
10478/// value arrives already built.
10479fn whole_row_projection_schema(alias: &str) -> ColumnSchema {
10480 let mut s = ColumnSchema::new(
10481 alloc::string::String::from(alias),
10482 spg_storage::DataType::Jsonb,
10483 true,
10484 );
10485 s.user_composite_type = Some(alloc::string::String::from(alias));
10486 s
10487}
10488
10489pub(crate) fn resolve_projection_column<'a>(
10490 c: &ColumnName,
10491 schema_cols: &'a [ColumnSchema],
10492 table_alias: &str,
10493) -> Result<Cow<'a, ColumnSchema>, EngineError> {
10494 if let Some(q) = &c.qualifier {
10495 let composite = alloc::format!("{q}.{name}", name = c.name);
10496 if let Some(s) = schema_cols.iter().find(|s| s.name == composite) {
10497 return Ok(Cow::Borrowed(s));
10498 }
10499 // Single-table case: the qualifier may equal the active alias —
10500 // then look for the bare column name.
10501 if q == table_alias
10502 && let Some(s) = schema_cols.iter().find(|s| s.name == c.name)
10503 {
10504 return Ok(Cow::Borrowed(s));
10505 }
10506 // For multi-table schemas the qualifier is unknown only if no
10507 // column bears the "<q>." prefix. For single-table, the alias
10508 // mismatch alone is enough.
10509 let prefix = alloc::format!("{q}.");
10510 let qualifier_known =
10511 q == table_alias || schema_cols.iter().any(|s| s.name.starts_with(&prefix));
10512 if !qualifier_known {
10513 return Err(EngineError::Eval(EvalError::UnknownQualifier {
10514 qualifier: q.clone(),
10515 }));
10516 }
10517 return Err(EngineError::Eval(EvalError::ColumnNotFound {
10518 name: c.name.clone(),
10519 }));
10520 }
10521 if let Some(s) = schema_cols.iter().find(|s| s.name == c.name) {
10522 return Ok(Cow::Borrowed(s));
10523 }
10524 let suffix = alloc::format!(".{name}", name = c.name);
10525 let mut matches = schema_cols.iter().filter(|s| s.name.ends_with(&suffix));
10526 let first = matches.next();
10527 let extra = matches.next();
10528 match (first, extra) {
10529 (Some(s), None) => Ok(Cow::Borrowed(s)),
10530 (Some(_), Some(_)) => Err(EngineError::Eval(EvalError::TypeMismatch {
10531 detail: alloc::format!("column reference \"{}\" is ambiguous", c.name),
10532 })),
10533 // The whole-row reference, checked LAST so a real column carrying
10534 // the alias's name still wins — the same precedence
10535 // `resolve_column` applies on the evaluation side.
10536 //
10537 // Two schema shapes reach here. A single-table (or subquery, or
10538 // CTE) scan carries its alias and bare column names, so the name
10539 // has to equal the alias. A JOIN's combined schema carries no
10540 // alias at all and qualifies every column `alias.col`, so the
10541 // alias is identified by the prefix instead — which is exactly
10542 // how `whole_row_composite` picks the fields out on the
10543 // evaluation side. Measured: `SELECT wr FROM wr JOIN jb ON …`
10544 // answers `(7,z)` on PG18.4 and errored here until this arm
10545 // covered the joined shape too.
10546 _ if !table_alias.is_empty() && c.name == table_alias => {
10547 Ok(Cow::Owned(whole_row_projection_schema(table_alias)))
10548 }
10549 _ if table_alias.is_empty() && {
10550 let prefix = alloc::format!("{name}.", name = c.name);
10551 schema_cols.iter().any(|s| s.name.starts_with(&prefix))
10552 } =>
10553 {
10554 Ok(Cow::Owned(whole_row_projection_schema(&c.name)))
10555 }
10556 _ => Err(EngineError::Eval(EvalError::ColumnNotFound {
10557 name: c.name.clone(),
10558 })),
10559 }
10560}
10561
10562/// v7.39 (round 135) — drop the synthetic `__grp_ord_*` columns injected by the
10563/// parser to carry per-branch GROUPING() masks into a grouping-set query's
10564/// ORDER BY. They must never reach the output. No-op unless such a column is
10565/// present, so the common path is untouched.
10566/// v7.39 (round 529) — the LIMIT / OFFSET that DISTINCT ON deferred.
10567///
10568/// PG limits what the dedup LEFT, not what fed it; SPG limited first, so
10569/// a `LIMIT 2` that should have answered two groups answered one.
10570fn apply_deferred_limit(
10571 rows: alloc::vec::Vec<Row<'static>>,
10572 deferred: &(
10573 Option<spg_sql::ast::LimitExpr>,
10574 Option<spg_sql::ast::LimitExpr>,
10575 ),
10576) -> alloc::vec::Vec<Row<'static>> {
10577 let count = |e: &Option<spg_sql::ast::LimitExpr>| match e {
10578 Some(spg_sql::ast::LimitExpr::Literal(n)) => Some(*n as usize),
10579 _ => None,
10580 };
10581 let mut rows = rows;
10582 if let Some(off) = count(&deferred.1) {
10583 rows = rows.split_off(off.min(rows.len()));
10584 }
10585 if let Some(lim) = count(&deferred.0) {
10586 rows.truncate(lim);
10587 }
10588 rows
10589}
10590
10591fn strip_synthetic_order_cols(result: QueryResult) -> QueryResult {
10592 let QueryResult::Rows { columns, rows } = result else {
10593 return result;
10594 };
10595 if !columns.iter().any(|c| c.name.starts_with("__grp_ord_")) {
10596 return QueryResult::Rows { columns, rows };
10597 }
10598 let keep: Vec<usize> = columns
10599 .iter()
10600 .enumerate()
10601 .filter(|(_, c)| !c.name.starts_with("__grp_ord_"))
10602 .map(|(i, _)| i)
10603 .collect();
10604 let new_cols: Vec<ColumnSchema> = keep.iter().map(|&i| columns[i].clone()).collect();
10605 let new_rows: Vec<Row<'static>> = rows
10606 .into_iter()
10607 .map(|r| Row::new(keep.iter().map(|&i| r.values[i].clone()).collect()))
10608 .collect();
10609 QueryResult::Rows {
10610 columns: new_cols,
10611 rows: new_rows,
10612 }
10613}
10614
10615/// v7.39 (round 487) — bind every projection item that is a bare column
10616/// reference to its position, once per query.
10617///
10618/// `#[inline(never)]` and out of line on purpose. Round 486 established
10619/// that adding code inside these scan bodies moves neighbouring hot
10620/// functions around under fat LTO: the first version of this had the loop
10621/// inline in `run_single_table_scan` and four aggregate shapes that never
10622/// touch that function — `full_agg`, `join_agg`, `group_500k`,
10623/// `filter_agg` — went up ~5 %, reproduced against the parent commit on
10624/// the same machine. Keeping it out of line kept them still.
10625#[inline(never)]
10626fn bind_direct_columns(
10627 projection: &[ProjectedItem],
10628 ctx: &eval::EvalContext<'_>,
10629) -> Vec<Option<usize>> {
10630 projection
10631 .iter()
10632 .map(|p| match &p.expr {
10633 Expr::Column(c) => eval::compile_column_pos(c, ctx).filter(|pos| {
10634 // Same exclusion `compile_into` makes: a composite column
10635 // has to be rehydrated from stored JSON, which is not a
10636 // cell read.
10637 ctx.columns
10638 .get(*pos)
10639 .is_none_or(|sc| sc.user_composite_type.is_none())
10640 }),
10641 _ => None,
10642 })
10643 .collect()
10644}
10645
10646/// v7.39 (round 505) — the name an un-aliased projected expression reports.
10647///
10648/// PG18 names a call for its function and everything else `?column?`;
10649/// measured with `\gdesc`. SPG used to print the parsed expression back
10650/// out for both dialects, so `SELECT upper(s)` reported `upper(s)` and
10651/// name-keyed row access found nothing under `upper`.
10652///
10653/// The MySQL half is NOT this rule and is deliberately left alone here:
10654/// MariaDB echoes the item's SOURCE TEXT verbatim (`a+b`, spacing and all),
10655/// which needs the parser to hand over spans the AST does not carry yet.
10656/// Until it does, a MySQL session keeps the printed form — closer to what
10657/// MariaDB answers than `?column?` would be.
10658pub(crate) fn default_output_name(expr: &Expr, mysql: bool) -> String {
10659 if mysql {
10660 return expr.to_string();
10661 }
10662 spg_sql::ast::figure_column_name(expr).unwrap_or_else(|| "?column?".to_string())
10663}
10664
10665pub(crate) fn build_projection(
10666 items: &[SelectItem],
10667 schema_cols: &[ColumnSchema],
10668 table_alias: &str,
10669 mysql: bool,
10670) -> Result<Vec<ProjectedItem>, EngineError> {
10671 build_projection_hiding_tail(items, schema_cols, table_alias, mysql, 0)
10672}
10673
10674/// v7.39 (round 592) — `build_projection` with the last `hidden_tail` columns
10675/// invisible to `*`.
10676///
10677/// The windowed-SELECT path appends a synthetic `__win_N` column per window
10678/// function so the rewritten projection can reference the computed values as
10679/// ordinary columns. `*` then expanded them too, and
10680/// `SELECT wr.*, row_number() OVER (ORDER BY id) FROM wr` came back with an
10681/// EXTRA column — the internal name's value, repeated. A wrong answer, and a
10682/// silent one: the row simply had one more field than the client asked for.
10683///
10684/// Hidden by POSITION rather than by name, for the reason round 512 recorded
10685/// about the system columns: a name test looks safe until a real column
10686/// happens to carry the name. These are appended last, so the count is what
10687/// identifies them.
10688pub(crate) fn build_projection_hiding_tail(
10689 items: &[SelectItem],
10690 schema_cols: &[ColumnSchema],
10691 table_alias: &str,
10692 mysql: bool,
10693 hidden_tail: usize,
10694) -> Result<Vec<ProjectedItem>, EngineError> {
10695 let visible = schema_cols.len().saturating_sub(hidden_tail);
10696 // v7.39 (round 462) — a join's combined schema qualifies every column
10697 // `alias.col` so the deferred-join cell lookups resolve by composite
10698 // name. That is an internal convention, and `*` was handing it to the
10699 // client: PG18 answers `SELECT * FROM a JOIN b` with the BARE names
10700 // (`id, g, id, h` — duplicates and all), SPG answered `a.id, a.g,
10701 // b.id, b.h`, so name-keyed row access found nothing. Round 128 had
10702 // already learned this for `q.*`; plain `*` never got the same rule.
10703 //
10704 // The signal is the schema itself, not the call site: only a combined
10705 // join schema arrives with no table alias AND every column qualified.
10706 // A single-table schema carries its alias, an empty schema has nothing
10707 // to strip, and a synthetic schema's names carry no dot.
10708 let joined_schema = table_alias.is_empty()
10709 && !schema_cols.is_empty()
10710 && schema_cols.iter().all(|c| c.name.contains('.'));
10711 let bare_name = |name: &str| -> String {
10712 if !joined_schema {
10713 return name.to_string();
10714 }
10715 match name.split_once('.') {
10716 Some((_, rest)) if !rest.is_empty() => rest.to_string(),
10717 _ => name.to_string(),
10718 }
10719 };
10720 let mut out = Vec::new();
10721 for item in items {
10722 match item {
10723 SelectItem::Wildcard => {
10724 // v7.39 (round 511) — `*` never expands a system column, as
10725 // PG's does not. They join the schema only when the statement
10726 // asked for them, so this matters for the mixed shape
10727 // `SELECT *, ctid FROM t`.
10728 //
10729 // v7.39 (round 512) — by POSITION, not by name. Matching on
10730 // the name alone looked safe because PG reserves them, and it
10731 // is not: `pg_replication_slots` genuinely has a column called
10732 // `xmin`, and `SELECT * FROM pg_replication_slots` lost it.
10733 // Only the trailing six, in the order the scan appends them,
10734 // are the synthetic ones.
10735 let sys_skip = synthetic_system_positions(schema_cols);
10736 for (idx, col) in schema_cols.iter().enumerate() {
10737 if sys_skip[idx] || idx >= visible {
10738 continue;
10739 }
10740 out.push(ProjectedItem {
10741 expr: Expr::Column(ColumnName {
10742 qualifier: None,
10743 name: col.name.clone(),
10744 }),
10745 output_name: bare_name(&col.name),
10746 ty: col.ty,
10747 nullable: col.nullable,
10748 user_enum_type: col.user_enum_type.clone(),
10749 mysql_fsp: col.mysql_fsp,
10750 collation_name: col.collation_name.clone(),
10751 });
10752 }
10753 }
10754 // v7.39 (round 128) — `q.*` expands to every column belonging to
10755 // the qualifier `q`. Single-table schemas carry bare column names
10756 // reachable via `table_alias`; a join's combined schema carries
10757 // `alias.col` names, so a column belongs to `q` when its name has
10758 // the `q.` prefix. PG labels the expanded columns by their bare
10759 // name, so the `alias.` prefix is stripped from the output name.
10760 SelectItem::QualifiedWildcard(q) => {
10761 let prefix = alloc::format!("{q}.");
10762 let single_table = !table_alias.is_empty() && q == table_alias;
10763 let mut matched = 0usize;
10764 for col in &schema_cols[..visible] {
10765 let belongs =
10766 col.name.starts_with(&prefix) || (single_table && !col.name.contains('.'));
10767 if !belongs {
10768 continue;
10769 }
10770 matched += 1;
10771 let output_name = col
10772 .name
10773 .strip_prefix(&prefix)
10774 .unwrap_or(&col.name)
10775 .to_string();
10776 out.push(ProjectedItem {
10777 expr: Expr::Column(ColumnName {
10778 qualifier: None,
10779 name: col.name.clone(),
10780 }),
10781 output_name,
10782 ty: col.ty,
10783 nullable: col.nullable,
10784 user_enum_type: col.user_enum_type.clone(),
10785 mysql_fsp: col.mysql_fsp,
10786 collation_name: col.collation_name.clone(),
10787 });
10788 }
10789 if matched == 0 {
10790 return Err(EngineError::Eval(EvalError::UnknownQualifier {
10791 qualifier: q.clone(),
10792 }));
10793 }
10794 }
10795 SelectItem::Expr { expr, alias } => {
10796 // Plain column ref keeps full schema info (real type +
10797 // nullability). For compound expressions try the
10798 // describe-side function-return-type table first
10799 // (e.g. `SELECT now()` → Timestamptz, `SELECT
10800 // concat(…)` → Text). Falls back to nullable Text
10801 // for shapes the describe path can't resolve.
10802 if let Expr::Column(c) = expr {
10803 let sch = resolve_projection_column(c, schema_cols, table_alias)?;
10804 let output_name = alias.clone().unwrap_or_else(|| c.name.clone());
10805 out.push(ProjectedItem {
10806 expr: expr.clone(),
10807 output_name,
10808 ty: sch.ty,
10809 nullable: sch.nullable,
10810 // v7.39 (read01 round 54) — a bare enum column keeps
10811 // its enum identity through the projection.
10812 user_enum_type: sch.user_enum_type.clone(),
10813 mysql_fsp: sch.mysql_fsp,
10814 collation_name: sch.collation_name.clone(),
10815 });
10816 } else if let Some(shape) = describe::describe_expr(expr, schema_cols) {
10817 let output_name = alias
10818 .clone()
10819 .unwrap_or_else(|| default_output_name(expr, mysql));
10820 out.push(ProjectedItem {
10821 expr: expr.clone(),
10822 output_name,
10823 ty: shape.ty,
10824 // v7.39 (round 258) — a projected EXPRESSION keeps its
10825 // enum identity too, not just a bare column. `FROM
10826 // (VALUES ('happy'::mood), …) t(m)` lowers to constant
10827 // SELECTs, so the derived column arrived here as a cast
10828 // and lost the enum — making the outer ORDER BY / min /
10829 // max / array_agg sort by the label's TEXT.
10830 nullable: shape.nullable,
10831 user_enum_type: None,
10832 mysql_fsp: crate::eval::expr_mysql_fsp(expr, schema_cols),
10833 // A bare column reference keeps its collation; any
10834 // other expression produces a new value and has none.
10835 collation_name: match expr {
10836 Expr::Column(c) => schema_cols
10837 .iter()
10838 .find(|sc| sc.name.eq_ignore_ascii_case(&c.name))
10839 .and_then(|sc| sc.collation_name.clone()),
10840 _ => None,
10841 },
10842 });
10843 } else {
10844 let output_name = alias
10845 .clone()
10846 .unwrap_or_else(|| default_output_name(expr, mysql));
10847 out.push(ProjectedItem {
10848 expr: expr.clone(),
10849 output_name,
10850 // A user ENUM has no DataType of its own, so
10851 // `describe_expr` cannot type `'ok'::mood` and the
10852 // item lands HERE, defaulting to text — which is why
10853 // pg_typeof answered `text` and a derived table sorted
10854 // enum values by their label.
10855 ty: DataType::Text,
10856 nullable: true,
10857 user_enum_type: crate::eval::expr_enum_type_name_pub(expr, schema_cols)
10858 .map(alloc::string::String::from),
10859 mysql_fsp: crate::eval::expr_mysql_fsp(expr, schema_cols),
10860 collation_name: match expr {
10861 Expr::Column(c) => schema_cols
10862 .iter()
10863 .find(|sc| sc.name.eq_ignore_ascii_case(&c.name))
10864 .and_then(|sc| sc.collation_name.clone()),
10865 _ => None,
10866 },
10867 });
10868 }
10869 }
10870 }
10871 }
10872 Ok(out)
10873}
10874
10875// ---- v4.12 window-function helpers ----
10876// The (partition-key, order-key, original-index) tuple shape used
10877// across these helpers is intrinsic to the planner. Factoring it
10878// into a typedef adds indirection without making the code clearer,
10879// so several lints are allowed inline on the affected functions
10880// rather than module-wide.
10881
10882/// v4.22: pick more specific column types from observed rows when
10883/// the projection builder defaulted to Text (the v1.x behavior for
10884/// non-column expressions). Lets `WITH t(n) AS (SELECT 1 ...)`
10885/// land an Int column in the CTE storage table rather than failing
10886/// the insert with "expected TEXT, got INT".
10887pub(crate) fn infer_column_types(
10888 columns: &[ColumnSchema],
10889 rows: &[Row<'static>],
10890) -> Vec<ColumnSchema> {
10891 let mut out = columns.to_vec();
10892 for (col_idx, col) in out.iter_mut().enumerate() {
10893 if col.ty != DataType::Text {
10894 continue;
10895 }
10896 let mut inferred: Option<DataType> = None;
10897 let mut all_null = true;
10898 for row in rows {
10899 let Some(v) = row.values.get(col_idx) else {
10900 continue;
10901 };
10902 let ty = match v {
10903 Value::Null => continue,
10904 Value::SmallInt(_) => DataType::SmallInt,
10905 Value::Int(_) => DataType::Int,
10906 Value::BigInt(_) => DataType::BigInt,
10907 Value::Float(_) => DataType::Float,
10908 Value::Bool(_) => DataType::Bool,
10909 Value::Vector(_) => DataType::Vector {
10910 dim: 0,
10911 encoding: VecEncoding::F32,
10912 },
10913 // v7.38 (read01 U16) — carry array values through with an
10914 // array type so a recursive CTE that projects an array
10915 // (e.g. a SEARCH/CYCLE ord / path column) types the working
10916 // column as an array, not Text.
10917 Value::TextArray(_) => DataType::TextArray,
10918 Value::IntArray(_) => DataType::IntArray,
10919 Value::BigIntArray(_) => DataType::BigIntArray,
10920 Value::SmallIntArray(_) => DataType::SmallIntArray,
10921 Value::FloatArray(_) => DataType::FloatArray,
10922 Value::BoolArray(_) => DataType::BoolArray,
10923 // v7.39 (GUC knife 2) — an interval projection describes
10924 // as INTERVAL (typed drivers read the RowDescription OID).
10925 Value::Interval { .. } => DataType::Interval,
10926 _ => DataType::Text,
10927 };
10928 all_null = false;
10929 inferred = Some(match inferred {
10930 None => ty,
10931 Some(prev) if prev == ty => prev,
10932 Some(_) => DataType::Text,
10933 });
10934 }
10935 if let Some(t) = inferred {
10936 col.ty = t;
10937 col.nullable = true;
10938 } else if all_null {
10939 col.nullable = true;
10940 }
10941 }
10942 out
10943}
10944
10945/// Numeric widening rank for UNION type resolution (higher = wider).
10946fn numeric_rank(t: DataType) -> Option<u8> {
10947 match t {
10948 DataType::SmallInt => Some(1),
10949 DataType::Int => Some(2),
10950 DataType::BigInt => Some(3),
10951 DataType::Numeric { .. } => Some(4),
10952 DataType::Float => Some(5),
10953 _ => None,
10954 }
10955}
10956
10957/// Resolve the common result type for a UNION / VALUES column from the
10958/// set of concrete (non-NULL) branch types, following the safe subset
10959/// of PG's type resolution:
10960/// * all-numeric → the widest numeric (int ∪ bigint → bigint, … ∪
10961/// numeric → numeric, … ∪ float → float);
10962/// * DATE ∪ TIMESTAMP → TIMESTAMP;
10963/// * exactly one concrete non-TEXT type mixed with TEXT literals →
10964/// that concrete type (the TEXT cells get parsed into it).
10965/// Returns `None` for anything ambiguous, so the caller leaves the
10966/// column untouched rather than risk a wrong or failing coercion.
10967fn resolve_union_common_type(types: &[DataType]) -> Option<DataType> {
10968 // NB: types are collected from RUNTIME values, which are coarser
10969 // than the schema (e.g. a timestamptz cell is Value::Timestamp), so
10970 // a single-concrete-type fast path must NOT overwrite the column
10971 // type — it would downgrade tstz to ts. NULL-only unification (PG:
10972 // `VALUES (NULL),(1.5)` types the column numeric even on the NULL
10973 // row's pg_typeof) needs schema-level resolution — recorded, not
10974 // attempted here.
10975 if types.len() < 2 {
10976 return None;
10977 }
10978 if types.iter().all(|t| numeric_rank(*t).is_some()) {
10979 return types
10980 .iter()
10981 .max_by_key(|t| numeric_rank(**t).unwrap_or(0))
10982 .copied();
10983 }
10984 let non_text: Vec<&DataType> = types
10985 .iter()
10986 .filter(|t| !matches!(t, DataType::Text))
10987 .collect();
10988 // v7.38 (T-tstz Phase 1) — temporal common type, per PG18.4: if any branch
10989 // is timestamptz the result is timestamptz (tstz ∪ ts, tstz ∪ date), else
10990 // if any is timestamp the result is timestamp (ts ∪ date). All values are
10991 // the same UTC-micros instant, so widening date/ts to tstz is lossless.
10992 if non_text.iter().all(|t| {
10993 matches!(
10994 t,
10995 DataType::Date | DataType::Timestamp | DataType::Timestamptz
10996 )
10997 }) && non_text
10998 .iter()
10999 .any(|t| matches!(t, DataType::Timestamp | DataType::Timestamptz))
11000 {
11001 if non_text.iter().any(|t| matches!(t, DataType::Timestamptz)) {
11002 return Some(DataType::Timestamptz);
11003 }
11004 return Some(DataType::Timestamp);
11005 }
11006 // A single concrete non-TEXT type mixed with TEXT literals.
11007 if non_text.len() == 1 {
11008 return Some(*non_text[0]);
11009 }
11010 // v7.37.16 — SEVERAL concrete types mixed with TEXT literals
11011 // (`VALUES ('NaN'::float8),(1.0),('NaN')` → float8 ∪ numeric ∪
11012 // text): resolve the concrete set first (PG treats the unknown-
11013 // typed string literals as castable to whatever the knowns
11014 // resolve to), then the TEXT cells parse into that target — the
11015 // caller's coercion dry-run still abandons the column if any
11016 // literal doesn't parse.
11017 if !non_text.is_empty() && non_text.len() < types.len() {
11018 let concrete: Vec<DataType> = non_text.iter().map(|t| **t).collect();
11019 return resolve_union_common_type(&concrete);
11020 }
11021 None
11022}
11023
11024/// Coerce every cell of a UNION / VALUES result column to one common
11025/// type (see [`resolve_union_common_type`]). Conservative: a column
11026/// whose branches already agree, or whose types don't resolve, or where
11027/// any cell fails to coerce, is left exactly as it was — this never
11028/// turns a previously-working query into an error.
11029fn unify_union_columns(columns: &mut [ColumnSchema], rows: &mut [Row<'static>]) {
11030 for col_idx in 0..columns.len() {
11031 let mut seen: Vec<DataType> = Vec::new();
11032 for row in rows.iter() {
11033 if let Some(dt) = row.values.get(col_idx).and_then(Value::data_type) {
11034 if !seen.contains(&dt) {
11035 seen.push(dt);
11036 }
11037 }
11038 }
11039 // v7.37.16 — a single concrete runtime type under a TEXT-typed
11040 // column means the column type came off a NULL (or unknown-text)
11041 // branch: NULL literals describe as TEXT (`L::Null → Text`), so
11042 // `VALUES (NULL),(1.5)` left the column "text" while every
11043 // non-NULL cell is numeric. Adopt the concrete type — schema
11044 // only, no cell changes. tstz-safe by construction: a real
11045 // timestamptz column's schema type is Timestamptz, not Text, so
11046 // the coarser runtime type (Value::Timestamp) can't downgrade it
11047 // through this arm; and a real text column's non-NULL cells are
11048 // Text, which keeps seen == [Text] and skips it.
11049 if seen.len() == 1
11050 && matches!(columns[col_idx].ty, DataType::Text)
11051 && !matches!(seen[0], DataType::Text)
11052 {
11053 columns[col_idx].ty = seen[0];
11054 continue;
11055 }
11056 let Some(target) = resolve_union_common_type(&seen) else {
11057 continue;
11058 };
11059 // v7.38 (read01) — an unconstrained NUMERIC result column keeps each
11060 // value's own scale in PG (`VALUES (1.0),(1.00)` renders `1.0` / `1.00`,
11061 // not `1.00` / `1.00`). So when the common type is NUMERIC, leave an
11062 // existing numeric cell untouched and only promote integers (to scale 0)
11063 // rather than rescaling everything to the widest scale.
11064 let scale_preserving_numeric = matches!(target, DataType::Numeric { .. });
11065 // Dry-run the coercion; abandon the whole column if any fails.
11066 let mut coerced: Vec<Option<Value<'static>>> = Vec::with_capacity(rows.len());
11067 let mut ok = true;
11068 for row in rows.iter() {
11069 match row.values.get(col_idx) {
11070 Some(Value::Numeric { .. }) if scale_preserving_numeric => {
11071 coerced.push(Some(row.values[col_idx].clone()));
11072 }
11073 Some(v) => {
11074 let cell_target = if scale_preserving_numeric {
11075 DataType::Numeric {
11076 precision: 0,
11077 scale: 0,
11078 }
11079 } else {
11080 target
11081 };
11082 match crate::conversions::coerce_value(
11083 v.clone(),
11084 cell_target,
11085 &columns[col_idx].name,
11086 col_idx,
11087 ) {
11088 Ok(cv) => coerced.push(Some(cv)),
11089 Err(_) => {
11090 ok = false;
11091 break;
11092 }
11093 }
11094 }
11095 None => coerced.push(None),
11096 }
11097 }
11098 if !ok {
11099 continue;
11100 }
11101 for (row, cv) in rows.iter_mut().zip(coerced) {
11102 if let (Some(slot), Some(nv)) = (row.values.get_mut(col_idx), cv) {
11103 *slot = nv;
11104 }
11105 }
11106 columns[col_idx].ty = target;
11107 }
11108}
11109
11110/// v4.22: encode a Row to a comparable byte key for UNION-DISTINCT
11111/// dedup inside the recursive iteration. Crude but deterministic
11112/// — Debug prints embed type discriminants so NULL ≠ "" ≠ 0.
11113fn encode_row_key(row: &Row<'static>) -> Vec<u8> {
11114 let mut out = Vec::new();
11115 for v in &row.values {
11116 // v7.38 (read01) — UNION / DISTINCT dedup must treat numerically-equal
11117 // exact values as one, regardless of type or scale (`1 = 1.0 = 1.00`),
11118 // like PG (and like GROUP BY, which already normalizes). The old
11119 // `{v:?}` key made `Numeric{10,1}` differ from `Numeric{100,2}`. Encode
11120 // the exact-decimal family through one scale-stripped canonical form.
11121 match v {
11122 Value::SmallInt(n) => encode_numeric_key(&mut out, i128::from(*n), 0),
11123 Value::Int(n) => encode_numeric_key(&mut out, i128::from(*n), 0),
11124 Value::BigInt(n) => encode_numeric_key(&mut out, i128::from(*n), 0),
11125 Value::Numeric { scaled, scale, .. } => encode_numeric_key(&mut out, *scaled, *scale),
11126 other => {
11127 let s = alloc::format!("{other:?}|");
11128 out.extend_from_slice(s.as_bytes());
11129 }
11130 }
11131 }
11132 out
11133}
11134
11135/// Append a scale-independent canonical key for an exact-decimal value: strip
11136/// trailing fractional zeros so `1`, `1.0`, `1.00` all key the same. The `\x01`
11137/// tag keeps a numeric key from colliding with a text value's `{v:?}` form.
11138fn encode_numeric_key(out: &mut Vec<u8>, mut scaled: i128, mut scale: u16) {
11139 while scale > 0 && scaled % 10 == 0 {
11140 scaled /= 10;
11141 scale -= 1;
11142 }
11143 let s = alloc::format!("\u{1}{scaled}e-{scale}|");
11144 out.extend_from_slice(s.as_bytes());
11145}
11146
11147/// Multi-arg `unnest(a, b, …)` — evaluate each array argument
11148/// (uncorrelated; outer refs were substituted upstream), then zip
11149/// them in parallel, NULL-padding shorter arrays to the longest
11150/// (PG's ROWS FROM shorthand). Shared by the primary-position
11151/// executor and the join-position materialiser, which both detect
11152/// the parser's `__unnest_zip` marker call.
11153pub(crate) fn unnest_zip_rows(
11154 args: &[Expr],
11155) -> Result<(alloc::vec::Vec<DataType>, alloc::vec::Vec<Row<'static>>), EngineError> {
11156 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
11157 let ctx = EvalContext::new(&empty_schema, None);
11158 let dummy_row = Row::new(alloc::vec::Vec::new());
11159 let mut dtypes: alloc::vec::Vec<DataType> = alloc::vec::Vec::with_capacity(args.len());
11160 let mut columns: alloc::vec::Vec<alloc::vec::Vec<Value<'static>>> =
11161 alloc::vec::Vec::with_capacity(args.len());
11162 for a in args {
11163 let v = eval::eval_expr(a, &dummy_row, &ctx).map_err(EngineError::Eval)?;
11164 let (dt, items): (DataType, alloc::vec::Vec<Value<'static>>) = match v {
11165 Value::Null => (DataType::Text, alloc::vec::Vec::new()),
11166 Value::TextArray(xs) => (
11167 DataType::Text,
11168 xs.into_iter()
11169 .map(|x| x.map(Value::text).unwrap_or(Value::Null))
11170 .collect(),
11171 ),
11172 Value::IntArray(xs) => (
11173 DataType::Int,
11174 xs.into_iter()
11175 .map(|x| x.map(Value::Int).unwrap_or(Value::Null))
11176 .collect(),
11177 ),
11178 Value::BigIntArray(xs) => (
11179 DataType::BigInt,
11180 xs.into_iter()
11181 .map(|x| x.map(Value::BigInt).unwrap_or(Value::Null))
11182 .collect(),
11183 ),
11184 other => {
11185 return Err(EngineError::Unsupported(alloc::format!(
11186 "unnest() expects array arguments, got {}",
11187 crate::conversions::pg_type_name_for_error_opt(other.data_type())
11188 )));
11189 }
11190 };
11191 dtypes.push(dt);
11192 columns.push(items);
11193 }
11194 let max_len = columns.iter().map(|c| c.len()).max().unwrap_or(0);
11195 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::with_capacity(max_len);
11196 for i in 0..max_len {
11197 let vals: alloc::vec::Vec<Value<'static>> = columns
11198 .iter()
11199 .map(|c| c.get(i).cloned().unwrap_or(Value::Null))
11200 .collect();
11201 rows.push(Row::new(vals));
11202 }
11203 Ok((dtypes, rows))
11204}
11205
11206/// Detect the parser's multi-arg unnest marker on an unnest_expr.
11207pub(crate) fn unnest_zip_args(expr: &Expr) -> Option<&[Expr]> {
11208 match expr {
11209 Expr::FunctionCall { name, args } if name == "__unnest_zip" => Some(args.as_slice()),
11210 _ => None,
11211 }
11212}
11213
11214/// Evaluate generate_series arguments (uncorrelated — outer refs
11215/// were substituted upstream where applicable) and build the row
11216/// stream. Dispatches on the start value's shape and rejects
11217/// mixed-shape calls early (e.g. start = timestamp, stop =
11218/// integer) so the caller gets a clean error rather than a panic.
11219/// Shared by the primary-position executor and the join-position
11220/// materialiser.
11221pub(crate) fn generate_series_rows(
11222 args: &[Expr],
11223 cancel: &CancelToken<'_>,
11224) -> Result<(DataType, alloc::vec::Vec<Row<'static>>), EngineError> {
11225 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
11226 let ctx = EvalContext::new(&empty_schema, None);
11227 let dummy_row = Row::new(alloc::vec::Vec::new());
11228 let mut arg_values: alloc::vec::Vec<Value<'static>> =
11229 alloc::vec::Vec::with_capacity(args.len());
11230 for a in args {
11231 arg_values.push(eval::eval_expr(a, &dummy_row, &ctx).map_err(EngineError::Eval)?);
11232 }
11233 generate_series_from_values(arg_values, args, cancel)
11234}
11235
11236/// v7.39 (read01 round 96) — the value-producing core of `generate_series`,
11237/// split out so the SELECT-list SRF path (`top_level_srf_output`) shares the
11238/// full integer / numeric / timestamp overload set with the FROM-clause path.
11239/// Before this split the target-list arm reimplemented only the integer case,
11240/// so `SELECT generate_series(1,2), generate_series(ts, ts, interval)` yielded
11241/// NULL for the timestamp column instead of the series. `arg_values` are the
11242/// already-evaluated arguments; `args` is kept only for the timestamptz-vs-
11243/// timestamp type resolution (it inspects the argument expressions' types).
11244pub(crate) fn generate_series_from_values(
11245 mut arg_values: alloc::vec::Vec<Value<'static>>,
11246 args: &[Expr],
11247 cancel: &CancelToken<'_>,
11248) -> Result<(DataType, alloc::vec::Vec<Row<'static>>), EngineError> {
11249 // PG: a NULL bound or step yields zero rows (also keeps the
11250 // NULL-padded lateral probe alive — schema without data).
11251 if arg_values.iter().any(|v| matches!(v, Value::Null)) {
11252 return Ok((DataType::BigInt, alloc::vec::Vec::new()));
11253 }
11254 // PG resolves `generate_series(date, date, interval)` to the
11255 // timestamp/timestamptz overload by implicitly casting each date
11256 // bound up to a timestamp at midnight (verified vs live PG18.4:
11257 // date args yield rows anchored at 00:00:00). SPG's TZ-naive
11258 // timestamp model renders the same instants, so fold any Date
11259 // bound to its midnight Timestamp (canonical `days *
11260 // 86_400_000_000`, matching cast.rs `cast_to_timestamp`) before
11261 // the shape match so the existing timestamp arm drives the walk.
11262 // v7.39 (read01 round 76) — WHICH timestamp overload PG picks matters:
11263 // `generate_series(date, date, interval)` has no date overload, and among
11264 // the two candidates PG prefers the timestamptz one (timestamptz is the
11265 // preferred type of the datetime category), so the column comes back
11266 // `timestamp with time zone` — the rows render with a `+00` offset. A
11267 // timestamptz bound obviously lands there too. Only genuinely
11268 // timestamp-typed bounds keep the TZ-naive result type.
11269 let empty_cols: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
11270 let tz = arg_values.iter().any(|v| matches!(v, Value::Date(_)))
11271 || args.iter().any(|a| {
11272 crate::describe::describe_expr(a, &empty_cols)
11273 .is_some_and(|s| matches!(s.ty, DataType::Timestamptz))
11274 });
11275 for v in &mut arg_values {
11276 if let Value::Date(d) = *v {
11277 *v = Value::Timestamp(crate::conversions::date_days_to_micros(d));
11278 }
11279 }
11280 match arg_values.as_slice() {
11281 [Value::Timestamp(start), Value::Timestamp(stop), step] => {
11282 let interval_step = match step {
11283 Value::Interval { .. } => step.clone(),
11284 // v7.38 (read01) — PG resolves an unknown-type string step
11285 // (`generate_series(date, date, '2 days')`) to INTERVAL; accept
11286 // a bare text step by parsing it the same way `::interval` does.
11287 Value::Text(s) => crate::conversions::coerce_value(
11288 Value::text(s.as_ref()),
11289 DataType::Interval,
11290 "",
11291 0,
11292 )
11293 .map_err(|_| {
11294 EngineError::Unsupported(alloc::format!(
11295 "generate_series(timestamp, timestamp, …): \
11296 could not parse step {s:?} as INTERVAL"
11297 ))
11298 })?,
11299 other => {
11300 return Err(EngineError::Unsupported(alloc::format!(
11301 "generate_series(timestamp, timestamp, …): \
11302 step must be INTERVAL, got {}",
11303 crate::conversions::pg_type_name_for_error_opt(other.data_type())
11304 )));
11305 }
11306 };
11307 let rows = generate_series_timestamps(*start, *stop, interval_step, cancel)?;
11308 Ok((
11309 if tz {
11310 DataType::Timestamptz
11311 } else {
11312 DataType::Timestamp
11313 },
11314 rows,
11315 ))
11316 }
11317 [start, stop, step]
11318 if value_is_integer(start) && value_is_integer(stop) && value_is_integer(step) =>
11319 {
11320 let s = value_to_i64(start);
11321 let e = value_to_i64(stop);
11322 let st = value_to_i64(step);
11323 // PG types the series by the argument type: int4 args → int4
11324 // elements, int8 (bigint) args → int8. Any BigInt operand widens.
11325 let wide = value_is_bigint(start) || value_is_bigint(stop) || value_is_bigint(step);
11326 let rows = generate_series_integers(s, e, st, wide, cancel)?;
11327 Ok((
11328 if wide {
11329 DataType::BigInt
11330 } else {
11331 DataType::Int
11332 },
11333 rows,
11334 ))
11335 }
11336 [start, stop] if value_is_integer(start) && value_is_integer(stop) => {
11337 let s = value_to_i64(start);
11338 let e = value_to_i64(stop);
11339 let wide = value_is_bigint(start) || value_is_bigint(stop);
11340 let rows = generate_series_integers(s, e, 1, wide, cancel)?;
11341 Ok((
11342 if wide {
11343 DataType::BigInt
11344 } else {
11345 DataType::Int
11346 },
11347 rows,
11348 ))
11349 }
11350 // v7.39 (read01 numeric.c) — the NUMERIC overload. PG walks the
11351 // series in exact numeric arithmetic; NaN / infinity bounds and a
11352 // zero step get dedicated wordings, and a mixed int/numeric call
11353 // resolves here via the implicit int→numeric cast.
11354 [_, _] | [_, _, _]
11355 if arg_values
11356 .iter()
11357 .any(|v| matches!(v, Value::Numeric { .. } | Value::NumericBig(_)))
11358 && arg_values.iter().all(|v| {
11359 matches!(v, Value::Numeric { .. } | Value::NumericBig(_)) || value_is_integer(v)
11360 }) =>
11361 {
11362 use spg_storage::NumericKind as K;
11363 let words: [(&str, &str); 3] = [
11364 (
11365 "start value cannot be NaN",
11366 "start value cannot be infinity",
11367 ),
11368 ("stop value cannot be NaN", "stop value cannot be infinity"),
11369 ("step size cannot be NaN", "step size cannot be infinity"),
11370 ];
11371 for (i, v) in arg_values.iter().enumerate() {
11372 if let Value::Numeric { kind, .. } = v {
11373 if *kind != K::Finite {
11374 let (nan_w, inf_w) = words[i];
11375 return Err(EngineError::Unsupported(
11376 if *kind == K::NaN { nan_w } else { inf_w }.into(),
11377 ));
11378 }
11379 }
11380 }
11381 let big =
11382 |v: &Value<'_>| eval::binop::value_to_bignum(v).expect("finite numeric or integer");
11383 let start = big(&arg_values[0]);
11384 let stop = big(&arg_values[1]);
11385 let step = if arg_values.len() == 3 {
11386 big(&arg_values[2])
11387 } else {
11388 spg_storage::bignum::BigNumeric::from_i128(1, 0)
11389 };
11390 if step.is_zero() {
11391 return Err(EngineError::Unsupported(
11392 "step size cannot equal zero".into(),
11393 ));
11394 }
11395 let descending = step.parts().0;
11396 let mut rows = alloc::vec::Vec::new();
11397 let mut cur = start;
11398 const MAX_ROWS: usize = 10_000_000;
11399 loop {
11400 cancel.check()?;
11401 let c = cur.cmp(&stop);
11402 if descending {
11403 if c == core::cmp::Ordering::Less {
11404 break;
11405 }
11406 } else if c == core::cmp::Ordering::Greater {
11407 break;
11408 }
11409 if rows.len() >= MAX_ROWS {
11410 return Err(EngineError::Unsupported(alloc::format!(
11411 "generate_series() result exceeds {MAX_ROWS} rows"
11412 )));
11413 }
11414 rows.push(Row::new(alloc::vec![eval::binop::bignum_to_value(
11415 cur.clone()
11416 )]));
11417 cur = cur.add(&step);
11418 }
11419 Ok((
11420 DataType::Numeric {
11421 precision: 0,
11422 scale: 0,
11423 },
11424 rows,
11425 ))
11426 }
11427 _ => Err(EngineError::Unsupported(alloc::format!(
11428 "generate_series(): v7.17 supports integer or (timestamp, timestamp, interval) \
11429 argument shapes; got {}",
11430 arg_values
11431 .iter()
11432 .map(|v| crate::conversions::pg_type_name_for_error_opt(v.data_type()))
11433 .collect::<alloc::vec::Vec<_>>()
11434 .join(", ")
11435 ))),
11436 }
11437}
11438
11439/// v7.17.0 Phase 3.10 — integer-mode generate_series materialiser.
11440/// Step direction follows the sign: positive step iterates upward
11441/// (stops when current > stop); negative iterates downward; zero
11442/// errors. Caller-facing row stream is `BigInt`-typed so a single
11443/// projection schema covers SmallInt / Int / BigInt callers.
11444fn generate_series_integers(
11445 start: i64,
11446 stop: i64,
11447 step: i64,
11448 wide: bool,
11449 cancel: &CancelToken<'_>,
11450) -> Result<alloc::vec::Vec<Row<'static>>, EngineError> {
11451 if step == 0 {
11452 return Err(EngineError::Unsupported(
11453 "step size cannot equal zero".into(),
11454 ));
11455 }
11456 let mut out = alloc::vec::Vec::new();
11457 let mut cur = start;
11458 // Hard cap to keep a runaway call from eating all memory. PG
11459 // has no such cap but does honour query timeout; SPG's cancel
11460 // token will fire too — this is a defense-in-depth backstop.
11461 const MAX_ROWS: usize = 10_000_000;
11462 loop {
11463 cancel.check()?;
11464 if step > 0 && cur > stop {
11465 break;
11466 }
11467 if step < 0 && cur < stop {
11468 break;
11469 }
11470 out.push(Row::new(alloc::vec![if wide {
11471 Value::BigInt(cur)
11472 } else {
11473 Value::Int(cur as i32)
11474 }]));
11475 if out.len() > MAX_ROWS {
11476 return Err(EngineError::Unsupported(alloc::format!(
11477 "generate_series(): exceeded {MAX_ROWS} rows; \
11478 narrow start/stop or use a larger step"
11479 )));
11480 }
11481 cur = match cur.checked_add(step) {
11482 Some(n) => n,
11483 None => break,
11484 };
11485 }
11486 Ok(out)
11487}
11488
11489/// v7.17.0 Phase 3.10 — timestamp-mode generate_series. step is a
11490/// `Value::Interval { months, micros }` per the caller's guard;
11491/// each iteration adds the interval via `apply_binary_interval`
11492/// so month-shifting handles short-month rollover (PG semantics).
11493fn generate_series_timestamps(
11494 start: i64,
11495 stop: i64,
11496 step: Value,
11497 cancel: &CancelToken<'_>,
11498) -> Result<alloc::vec::Vec<Row<'static>>, EngineError> {
11499 let (months, days, micros) = match &step {
11500 Value::Interval {
11501 months,
11502 days,
11503 micros,
11504 } => (*months, *days, *micros),
11505 _ => unreachable!("caller guards step.is_interval"),
11506 };
11507 if months == 0 && days == 0 && micros == 0 {
11508 return Err(EngineError::Unsupported(
11509 "generate_series(): INTERVAL step cannot be zero".into(),
11510 ));
11511 }
11512 let ascending = months > 0 || days > 0 || micros > 0;
11513 let mut out = alloc::vec::Vec::new();
11514 let mut cur = Value::Timestamp(start);
11515 const MAX_ROWS: usize = 10_000_000;
11516 loop {
11517 cancel.check()?;
11518 let cur_t = match cur {
11519 Value::Timestamp(t) => t,
11520 _ => unreachable!("loop invariant: cur is Timestamp"),
11521 };
11522 if ascending && cur_t > stop {
11523 break;
11524 }
11525 if !ascending && cur_t < stop {
11526 break;
11527 }
11528 out.push(Row::new(alloc::vec![Value::Timestamp(cur_t)]));
11529 if out.len() > MAX_ROWS {
11530 return Err(EngineError::Unsupported(alloc::format!(
11531 "generate_series(): exceeded {MAX_ROWS} rows; \
11532 narrow start/stop or use a larger step"
11533 )));
11534 }
11535 let next = eval::apply_binary_interval(
11536 spg_sql::ast::BinOp::Add,
11537 &cur,
11538 &Value::Interval {
11539 months,
11540 days,
11541 micros,
11542 },
11543 )
11544 .map_err(EngineError::Eval)?;
11545 cur = match next {
11546 Some(v) => v,
11547 None => break,
11548 };
11549 }
11550 Ok(out)
11551}
11552
11553/// v7.17.0 Phase 3.P0-49 — PG-canonical: `FETCH FIRST <n> ROWS
11554/// WITH TIES` requires an `ORDER BY`. Without one, there's no
11555/// way to identify "ties" deterministically, so PG errors at
11556/// plan time. SPG mirrors that surface so the same DDL / app
11557/// behaviour holds on cutover.
11558fn check_with_ties_requires_order_by(stmt: &SelectStatement) -> Result<(), EngineError> {
11559 if stmt.limit_with_ties && stmt.order_by.is_empty() {
11560 return Err(EngineError::Unsupported(alloc::string::String::from(
11561 "WITH TIES cannot be specified without ORDER BY clause",
11562 )));
11563 }
11564 Ok(())
11565}
11566
11567/// v7.19 P5 — true iff `expr` is `unnest(arg)` at the top level
11568/// (case-insensitive). Used by `exec_select_cancel`'s
11569/// projection loop to detect Set-Returning-Function rows that
11570/// need per-row expansion. Only the top-level call counts —
11571/// `coalesce(unnest(arr), 'x')` is NOT a SRF row from the
11572/// projection's perspective; it would surface as an "unknown
11573/// function" mismatch downstream, which is what we want
11574/// (multi-SRF / nested SRF is documented carve-out for v7.19).
11575fn is_top_level_unnest(expr: &spg_sql::ast::Expr) -> bool {
11576 top_level_srf_kind(expr).is_some()
11577}
11578
11579/// v7.38 (read01, T15) — which set-returning function a top-level SELECT-list
11580/// call is, if any. Matching is allocation-free (`eq_ignore_ascii_case`, no
11581/// `to_ascii_lowercase`) because `top_level_srf_output` classifies once per
11582/// source row.
11583#[derive(Clone, Copy, PartialEq, Eq)]
11584pub(crate) enum SrfKind {
11585 Unnest,
11586 /// v7.39 (read01 round 67) — `generate_series(a, b[, step])` in the target
11587 /// list. It used to be handled ONLY by the parser's lift into FROM, so a
11588 /// second one in the same list came back as "unknown function".
11589 GenerateSeries,
11590 GenerateSubscripts,
11591 /// `_text` variants unwrap scalars to their lexeme; the plain forms render
11592 /// every value as compact JSON text.
11593 ArrayElements {
11594 as_text: bool,
11595 },
11596 PathQuery,
11597 RegexpMatches,
11598 Each {
11599 as_text: bool,
11600 },
11601 ObjectKeys,
11602}
11603
11604/// Case-insensitive match against any of `names`.
11605fn name_is(name: &str, names: &[&str]) -> bool {
11606 names.iter().any(|n| name.eq_ignore_ascii_case(n))
11607}
11608
11609pub(crate) fn top_level_srf_kind(expr: &spg_sql::ast::Expr) -> Option<SrfKind> {
11610 let spg_sql::ast::Expr::FunctionCall { name, args } = expr else {
11611 return None;
11612 };
11613 let n = args.len();
11614 // v7.38 (read01) — generate_subscripts(arr, dim) is set-returning in the
11615 // SELECT list (it returned an array there before) and shares the unnest
11616 // expansion machinery.
11617 if n == 1 && name.eq_ignore_ascii_case("unnest") {
11618 return Some(SrfKind::Unnest);
11619 }
11620 if (2..=3).contains(&n) && name.eq_ignore_ascii_case("generate_series") {
11621 return Some(SrfKind::GenerateSeries);
11622 }
11623 if n == 2 && name.eq_ignore_ascii_case("generate_subscripts") {
11624 return Some(SrfKind::GenerateSubscripts);
11625 }
11626 // v7.38 (read01, T15) — the jsonb/json SRF family and regexp_matches expand
11627 // per element / match in the SELECT list; they collapsed to a single row
11628 // (a TextArray, or an "unknown function" error for `each`) before.
11629 if n == 1 && name_is(name, &["jsonb_array_elements", "json_array_elements"]) {
11630 return Some(SrfKind::ArrayElements { as_text: false });
11631 }
11632 if n == 1
11633 && name_is(
11634 name,
11635 &["jsonb_array_elements_text", "json_array_elements_text"],
11636 )
11637 {
11638 return Some(SrfKind::ArrayElements { as_text: true });
11639 }
11640 // v7.39 (jsonpath depth) — 3rd arg = vars, 4th = silent.
11641 if (2..=4).contains(&n) && name_is(name, &["jsonb_path_query", "json_path_query"]) {
11642 return Some(SrfKind::PathQuery);
11643 }
11644 if (2..=3).contains(&n) && name.eq_ignore_ascii_case("regexp_matches") {
11645 return Some(SrfKind::RegexpMatches);
11646 }
11647 if n == 1 && name_is(name, &["jsonb_each", "json_each"]) {
11648 return Some(SrfKind::Each { as_text: false });
11649 }
11650 if n == 1 && name_is(name, &["jsonb_each_text", "json_each_text"]) {
11651 return Some(SrfKind::Each { as_text: true });
11652 }
11653 if n == 1 && name_is(name, &["jsonb_object_keys", "json_object_keys"]) {
11654 return Some(SrfKind::ObjectKeys);
11655 }
11656 None
11657}
11658
11659/// v7.38 (read01) — the row-set a top-level SELECT-list SRF emits: the elements
11660/// for `unnest(arr)`, or the 1-based subscripts `1..=length` for
11661/// `generate_subscripts(arr, 1)` (a non-1 dimension over a 1-D array yields no
11662/// rows, as in PG).
11663pub(crate) fn top_level_srf_output(
11664 expr: &spg_sql::ast::Expr,
11665 row: &Row<'static>,
11666 ctx: &EvalContext<'_>,
11667) -> Result<Vec<Value<'static>>, EngineError> {
11668 let (Some(kind), spg_sql::ast::Expr::FunctionCall { name, args }) =
11669 (top_level_srf_kind(expr), expr)
11670 else {
11671 return Err(EngineError::Unsupported(
11672 "expected a SELECT-list SRF call".into(),
11673 ));
11674 };
11675 match kind {
11676 SrfKind::Unnest => {
11677 // v7.39 (round 743) — `unnest(ARRAY[e1, …, ek])` evaluates
11678 // the elements DIRECTLY: the old path built the whole
11679 // Value::Array (one eval + a clone per element) only for
11680 // array_value_to_elements to clone every element back out.
11681 // Any other argument shape (a column, a function result)
11682 // keeps the build-then-split path.
11683 if let spg_sql::ast::Expr::Array(items) = &args[0] {
11684 return items
11685 .iter()
11686 .map(|e| eval::eval_expr(e, row, ctx).map_err(EngineError::Eval))
11687 .collect();
11688 }
11689 let arr = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11690 array_value_to_elements(&arr)
11691 }
11692 SrfKind::GenerateSeries => {
11693 // v7.39 (read01 round 96) — evaluate the args against the actual
11694 // row, then hand off to the shared core so the numeric and
11695 // timestamp/timestamptz overloads work here too (this arm used to
11696 // handle only integers, silently NULLing a temporal/numeric series
11697 // when it shared a target list with another SRF).
11698 let mut arg_values: Vec<Value<'static>> = Vec::with_capacity(args.len());
11699 for a in args {
11700 arg_values.push(eval::eval_expr(a, row, ctx).map_err(EngineError::Eval)?);
11701 }
11702 let (_, rows) = generate_series_from_values(arg_values, args, &CancelToken::none())?;
11703 Ok(rows
11704 .into_iter()
11705 .map(|r| r.values.into_iter().next().unwrap_or(Value::Null))
11706 .collect())
11707 }
11708 SrfKind::GenerateSubscripts => {
11709 let arr = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11710 let dim = eval::eval_expr(&args[1], row, ctx).map_err(EngineError::Eval)?;
11711 if !matches!(dim, Value::Int(1) | Value::BigInt(1) | Value::SmallInt(1)) {
11712 return Ok(Vec::new());
11713 }
11714 let len = array_value_to_elements(&arr)?.len();
11715 Ok((1..=len).map(|i| Value::Int(i as i32)).collect())
11716 }
11717 // One Value per array element (`_text` → text / SQL NULL, plain → the
11718 // element's compact JSON text) — the element list the FROM-clause form
11719 // materialises.
11720 SrfKind::ArrayElements { as_text } => {
11721 let arg = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11722 if matches!(arg, Value::Null) {
11723 return Ok(Vec::new());
11724 }
11725 let items =
11726 crate::json::array_element_rows(&arg, as_text, name).map_err(EngineError::Eval)?;
11727 Ok(items
11728 .into_iter()
11729 .map(|opt| opt.map(Value::text).unwrap_or(Value::Null))
11730 .collect())
11731 }
11732 // The scalar form already yields a TextArray of the keys (or errors on
11733 // a non-object, like PG); expand it into rows.
11734 SrfKind::ObjectKeys => {
11735 let v = eval::eval_expr(expr, row, ctx).map_err(EngineError::Eval)?;
11736 array_value_to_elements(&v)
11737 }
11738 // One row per match, each a text[] of the pattern's capture groups.
11739 SrfKind::RegexpMatches => {
11740 let vals: Vec<Value<'static>> = args
11741 .iter()
11742 .map(|a| eval::eval_expr(a, row, ctx).map_err(EngineError::Eval))
11743 .collect::<Result<_, _>>()?;
11744 crate::eval::regexp_matches_rows(&vals).map_err(EngineError::Eval)
11745 }
11746 // One composite `(key, value)` row per object member (plain → jsonb
11747 // value, `_text` → text / SQL NULL).
11748 SrfKind::Each { as_text } => {
11749 let arg = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11750 if matches!(arg, Value::Null) {
11751 return Ok(Vec::new());
11752 }
11753 let pairs = crate::json::each_rows(&arg, as_text, name).map_err(EngineError::Eval)?;
11754 Ok(pairs
11755 .into_iter()
11756 .map(|(k, v)| {
11757 let val = if as_text {
11758 v.map(Value::text).unwrap_or(Value::Null)
11759 } else {
11760 v.map(Value::json).unwrap_or(Value::Null)
11761 };
11762 Value::Composite(alloc::vec![
11763 ("key".to_string(), Value::text(k)),
11764 ("value".to_string(), val),
11765 ])
11766 })
11767 .collect())
11768 }
11769 // One Value per matched JSON value.
11770 SrfKind::PathQuery => {
11771 let doc = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11772 let path = eval::eval_expr(&args[1], row, ctx).map_err(EngineError::Eval)?;
11773 // v7.39 — optional vars document (3rd arg).
11774 let vars = match args.get(2) {
11775 Some(a) => {
11776 let v = eval::eval_expr(a, row, ctx).map_err(EngineError::Eval)?;
11777 crate::json::parse_path_vars(&v).map_err(EngineError::Eval)?
11778 }
11779 None => None,
11780 };
11781 match crate::json::path_query_vars(&doc, &path, vars.as_ref())
11782 .map_err(EngineError::Eval)?
11783 {
11784 Value::Null => Ok(Vec::new()),
11785 Value::TextArray(items) => Ok(items
11786 .into_iter()
11787 .map(|opt| opt.map(Value::text).unwrap_or(Value::Null))
11788 .collect()),
11789 other => Ok(alloc::vec![other]),
11790 }
11791 }
11792 }
11793}
11794
11795/// v7.19 P5 — turn an array-typed `Value` into the element list
11796/// `unnest()` projection emits. NULL → empty list (PG: `unnest(NULL)
11797/// = (no rows)`). Non-array values fall through to a type-mismatch
11798/// error.
11799pub(crate) fn array_value_to_elements(v: &Value) -> Result<Vec<Value<'static>>, EngineError> {
11800 // v7.39 (round 236) — PG unnests a multidimensional array into its
11801 // elements in row-major order (`unnest(ARRAY[[1,2],[3,4]])` is four
11802 // rows). SPG stores 2-D arrays as their own variants, which fell
11803 // through to the type-mismatch arm below.
11804 if let Some(flat) = crate::eval::values::flatten_2d(v) {
11805 return array_value_to_elements(&flat);
11806 }
11807 match v {
11808 Value::Null => Ok(Vec::new()),
11809 Value::TextArray(items) => Ok(items
11810 .iter()
11811 .map(|opt| {
11812 opt.as_ref()
11813 .map(|s| Value::text(s.clone()))
11814 .unwrap_or(Value::Null)
11815 })
11816 .collect()),
11817 Value::IntArray(items) => Ok(items
11818 .iter()
11819 .map(|opt| opt.map(Value::Int).unwrap_or(Value::Null))
11820 .collect()),
11821 Value::BigIntArray(items) => Ok(items
11822 .iter()
11823 .map(|opt| opt.map(Value::BigInt).unwrap_or(Value::Null))
11824 .collect()),
11825 // v7.39 (read01 multirangetypes.c) — unnest(anymultirange): one
11826 // range per canonical span.
11827 Value::Multirange { kind, ranges } => Ok(ranges
11828 .iter()
11829 .map(|s| Value::Range {
11830 kind: *kind,
11831 lower: s.lower.clone(),
11832 upper: s.upper.clone(),
11833 lower_inc: s.lower_inc,
11834 upper_inc: s.upper_inc,
11835 empty: false,
11836 })
11837 .collect()),
11838 other => Err(EngineError::Eval(EvalError::TypeMismatch {
11839 detail: alloc::format!(
11840 "unnest() expects an array argument, got {}",
11841 crate::conversions::pg_type_name_for_error_opt(other.data_type())
11842 ),
11843 })),
11844 }
11845}
11846
11847impl Engine {
11848 /// v7.17.0 Phase 1.2 — find every catalog VIEW referenced in
11849 /// the SELECT's FROM / JOIN graph, re-parse each view's body
11850 /// source, and prepend it as a synthetic CTE on the
11851 /// returned SelectStatement. Returns `None` when no view
11852 /// references are found (caller proceeds with the original
11853 /// statement); returns `Some(rewritten)` otherwise (caller
11854 /// re-runs exec_select_cancel on the rewritten form so the
11855 /// regular CTE materialiser handles it).
11856 fn expand_views_in_select(
11857 &self,
11858 stmt: &SelectStatement,
11859 ) -> Result<Option<SelectStatement>, EngineError> {
11860 let cat = self.active_catalog();
11861 let mut referenced: Vec<String> = Vec::new();
11862 if let Some(from) = &stmt.from {
11863 collect_view_refs(&from.primary, cat, &mut referenced);
11864 for j in &from.joins {
11865 collect_view_refs(&j.table, cat, &mut referenced);
11866 }
11867 }
11868 // Don't expand a view name that's already shadowed by a
11869 // CTE on the same SELECT — the CTE wins per PG.
11870 referenced.retain(|n| !stmt.ctes.iter().any(|c| c.name == *n));
11871 if referenced.is_empty() {
11872 return Ok(None);
11873 }
11874 let mut new_ctes: Vec<spg_sql::ast::Cte> = Vec::with_capacity(referenced.len());
11875 for name in &referenced {
11876 let view = cat.view(name).ok_or_else(|| {
11877 EngineError::Storage(spg_storage::StorageError::Corrupt(alloc::format!(
11878 "view {name:?} disappeared mid-expansion"
11879 )))
11880 })?;
11881 let parsed = spg_sql::parser::parse_statement(&view.body).map_err(|e| {
11882 EngineError::Unsupported(alloc::format!("view {name:?} body re-parse failed: {e}"))
11883 })?;
11884 let Statement::Select(body) = parsed else {
11885 return Err(EngineError::Unsupported(alloc::format!(
11886 "view {name:?} body is not a SELECT (catalog corruption)"
11887 )));
11888 };
11889 new_ctes.push(spg_sql::ast::Cte {
11890 name: name.clone(),
11891 body: spg_sql::ast::CteBody::Select(body),
11892 recursive: false,
11893 column_overrides: view.columns.clone(),
11894 search: None,
11895 cycle: None,
11896 });
11897 }
11898 let mut out = stmt.clone();
11899 // Prepend so view CTEs are visible to caller-supplied CTEs.
11900 new_ctes.extend(out.ctes);
11901 out.ctes = new_ctes;
11902 Ok(Some(out))
11903 }
11904
11905 /// v7.37.6-B(sentori Epic 2 P0)— if `stmt`'s FROM-clause references
11906 /// any partition-parent table, rewrite the SELECT so each parent
11907 /// reference resolves to a CTE whose body is a `UNION ALL` over the
11908 /// children that pass the WHERE-derived partition-key range. Returns
11909 /// `None`(no rewrite needed)when no parent is referenced or all
11910 /// references are shadowed by a same-name CTE.
11911 ///
11912 /// Pruning vocabulary at v7.37.6-B:
11913 /// * Flat `AND` chain over `<key> {>= | > | < | <= | =} literal`
11914 /// and `<key> BETWEEN literal AND literal`.
11915 /// * Anything outside that(OR / nested IN / function call on the
11916 /// key)defaults to "no pruning" — every child + DEFAULT lands
11917 /// in the UNION. Correctness is preserved; only the plan size
11918 /// widens.
11919 fn expand_partition_parents_in_select(
11920 &self,
11921 stmt: &SelectStatement,
11922 ) -> Result<Option<SelectStatement>, EngineError> {
11923 let cat = self.active_catalog();
11924 let Some(from) = &stmt.from else {
11925 return Ok(None);
11926 };
11927 let mut parent_refs: Vec<String> = Vec::new();
11928 collect_partition_parent_refs(&from.primary, cat, &mut parent_refs);
11929 for j in &from.joins {
11930 collect_partition_parent_refs(&j.table, cat, &mut parent_refs);
11931 }
11932 // Drop names shadowed by a CTE on the same SELECT(PG semantics
11933 // — same as view expansion above).
11934 parent_refs.retain(|n| !stmt.ctes.iter().any(|c| c.name.eq_ignore_ascii_case(n)));
11935 if parent_refs.is_empty() {
11936 return Ok(None);
11937 }
11938 // Synthesise a CTE name per parent so the existing
11939 // "CTE shadows a real table" guard doesn't fire (the parent
11940 // IS a real table in the catalog, unlike VIEW expansion's
11941 // case). The FROM-clause TableRef walker below rewrites
11942 // every parent reference to point at the synthetic CTE.
11943 let synth_name = |p: &str| alloc::format!("__spg_partition_{p}");
11944 let mut new_ctes: Vec<spg_sql::ast::Cte> = Vec::with_capacity(parent_refs.len());
11945 let mut expanded_parents: Vec<alloc::string::String> = Vec::new();
11946 for parent_name in &parent_refs {
11947 // No children = no rewrite. The parent itself is a real
11948 // (empty-rows) table — the regular FROM-resolution path
11949 // will scan it and return 0 rows, matching the
11950 // "partition parent with no children" plan. Skipping the
11951 // CTE here also avoids `SELECT * FROM parent` re-entering
11952 // this rewrite on the synthetic body (infinite recursion).
11953 let Some(body) = self.build_partition_parent_union_body(parent_name, stmt)? else {
11954 continue;
11955 };
11956 new_ctes.push(spg_sql::ast::Cte {
11957 name: synth_name(parent_name),
11958 body: spg_sql::ast::CteBody::Select(body),
11959 recursive: false,
11960 column_overrides: Vec::new(),
11961 search: None,
11962 cycle: None,
11963 });
11964 expanded_parents.push(parent_name.clone());
11965 }
11966 if expanded_parents.is_empty() {
11967 return Ok(None);
11968 }
11969 let mut out = stmt.clone();
11970 if let Some(from) = out.from.as_mut() {
11971 rewrite_partition_parent_table_ref(&mut from.primary, &expanded_parents, &synth_name);
11972 for j in &mut from.joins {
11973 rewrite_partition_parent_table_ref(&mut j.table, &expanded_parents, &synth_name);
11974 }
11975 }
11976 new_ctes.extend(out.ctes);
11977 out.ctes = new_ctes;
11978 Ok(Some(out))
11979 }
11980
11981 /// Build the `SELECT * FROM child1 UNION ALL …` body for one parent.
11982 /// Children include every overlap-hit `Range` plus(always)the
11983 /// `Default` child(if any). Returns `Ok(None)` when no children
11984 /// would survive — caller skips the CTE injection and lets the
11985 /// parent fall through to the regular(empty-rows)scan path,
11986 /// avoiding the infinite recursion that an empty-body CTE
11987 /// referencing the parent name would trigger.
11988 /// v7.37.16 (16.10) — public helper invoked from explain.rs to
11989 /// surface "which children survive the WHERE-clause prune" in
11990 /// EXPLAIN output. Returns `None` when `parent_name` isn't
11991 /// actually a partition parent; otherwise returns the list of
11992 /// children the planner would scan (same algorithm as
11993 /// [`Self::build_partition_parent_union_body`] but without the
11994 /// SQL re-parse).
11995 /// v7.39 (round 224) — the kept-children prune keyed off a bare WHERE
11996 /// expression (the PG-shaped EXPLAIN's scan builder has no full
11997 /// SelectStatement in hand). Wraps the original by synthesising a
11998 /// minimal statement carrying just the predicate.
11999 pub(crate) fn explain_partition_kept_children_by_where(
12000 &self,
12001 parent_name: &str,
12002 where_: Option<&spg_sql::ast::Expr>,
12003 ) -> Option<Vec<alloc::string::String>> {
12004 let mut synth = SelectStatement::default();
12005 synth.where_ = where_.cloned();
12006 self.explain_partition_kept_children(parent_name, &synth)
12007 }
12008
12009 pub(crate) fn explain_partition_kept_children(
12010 &self,
12011 parent_name: &str,
12012 outer: &SelectStatement,
12013 ) -> Option<Vec<alloc::string::String>> {
12014 use spg_storage::PartitionRole;
12015 let cat = self.active_catalog();
12016 let parent = cat.get(parent_name)?;
12017 let (key_position, parent_kind) = match &parent.schema().partition_role {
12018 Some(PartitionRole::Parent {
12019 key_column_positions,
12020 kind,
12021 ..
12022 }) => (*key_column_positions.first().unwrap_or(&0), *kind),
12023 _ => return None,
12024 };
12025 let key_col_name = parent.schema().columns[key_position].name.clone();
12026 let (lo_bound, hi_bound) = match outer.where_.as_ref() {
12027 Some(expr) => extract_key_range(expr, &key_col_name),
12028 None => (None, None),
12029 };
12030 let eq_value: Option<spg_storage::Value<'static>> = match outer.where_.as_ref() {
12031 Some(expr) => extract_key_eq_value(expr, &key_col_name),
12032 None => None,
12033 };
12034 let children = crate::partition::children_of_parent(cat, parent_name);
12035 let mut kept: Vec<alloc::string::String> = Vec::new();
12036 let mut default_child: Option<alloc::string::String> = None;
12037 for child_name in &children {
12038 let Some(child) = cat.get(child_name) else {
12039 continue;
12040 };
12041 match &child.schema().partition_role {
12042 Some(PartitionRole::Range { lower, upper, .. }) => {
12043 if range_satisfies_filter(lower, upper, lo_bound.as_ref(), hi_bound.as_ref()) {
12044 kept.push(child_name.clone());
12045 }
12046 }
12047 Some(PartitionRole::List { values, .. }) => match &eq_value {
12048 Some(v) => {
12049 if values.iter().any(|b| b.equals_value(v)) {
12050 kept.push(child_name.clone());
12051 }
12052 }
12053 None => kept.push(child_name.clone()),
12054 },
12055 Some(PartitionRole::Hash {
12056 modulus, remainder, ..
12057 }) => match &eq_value {
12058 Some(v) => {
12059 let h = crate::partition::pg_compatible_hash(v);
12060 if h.rem_euclid(u64::from(*modulus)) == u64::from(*remainder) {
12061 kept.push(child_name.clone());
12062 }
12063 }
12064 None => kept.push(child_name.clone()),
12065 },
12066 Some(PartitionRole::Default { .. }) => {
12067 default_child = Some(child_name.clone());
12068 }
12069 _ => {}
12070 }
12071 }
12072 let _ = parent_kind;
12073 if let Some(d) = default_child {
12074 if kept.is_empty() || eq_value.is_none() {
12075 kept.push(d);
12076 }
12077 }
12078 Some(kept)
12079 }
12080
12081 fn build_partition_parent_union_body(
12082 &self,
12083 parent_name: &str,
12084 outer: &SelectStatement,
12085 ) -> Result<Option<SelectStatement>, EngineError> {
12086 use spg_storage::PartitionRole;
12087 let cat = self.active_catalog();
12088 let parent = cat.get(parent_name).ok_or_else(|| {
12089 EngineError::Storage(spg_storage::StorageError::Corrupt(alloc::format!(
12090 "partition parent {parent_name:?} disappeared mid-expansion"
12091 )))
12092 })?;
12093 let (key_position, parent_kind) = match &parent.schema().partition_role {
12094 Some(PartitionRole::Parent {
12095 key_column_positions,
12096 kind,
12097 ..
12098 }) => (*key_column_positions.first().unwrap_or(&0), *kind),
12099 // v7.39 (round 645) — an INHERITANCE parent, which has no
12100 // role of its own: the relationship is recorded only in the
12101 // children. Three things differ from a partition parent and
12102 // all three are in this body.
12103 //
12104 // * The parent HOLDS ROWS, so it is a term of the union —
12105 // `FROM ONLY`, or expanding it would recurse.
12106 // * There is no partition key, so there is nothing to
12107 // prune: every child is a term.
12108 // * A child may declare columns of its own, so the terms
12109 // name the PARENT's columns rather than `*`. PG's
12110 // `SELECT * FROM parent` returns the parent's shape.
12111 //
12112 // Answered from this match rather than a branch before it —
12113 // round 644 measured what an extra early return beside an
12114 // existing test costs in this file.
12115 _ if crate::partition::has_inheritance_children(cat, parent_name) => {
12116 let cols = parent
12117 .schema()
12118 .columns
12119 .iter()
12120 .map(|c| quote_ident_for_sql(&c.name))
12121 .collect::<Vec<_>>()
12122 .join(", ");
12123 let carry_sys = references_ctid(outer);
12124 let sys = if carry_sys {
12125 let mut t = alloc::string::String::new();
12126 for s in SYSTEM_COLUMNS {
12127 t.push_str(", ");
12128 t.push_str(s);
12129 }
12130 t
12131 } else {
12132 alloc::string::String::new()
12133 };
12134 let mut body = alloc::format!(
12135 "SELECT {cols}{sys} FROM ONLY {}",
12136 quote_ident_for_sql(parent_name)
12137 );
12138 for child in crate::partition::children_of_parent(cat, parent_name) {
12139 body.push_str(&alloc::format!(
12140 " UNION ALL SELECT {cols}{sys} FROM {}",
12141 quote_ident_for_sql(&child)
12142 ));
12143 }
12144 return parse_select_or_corrupt(&body).map(Some);
12145 }
12146 _ => {
12147 return Err(EngineError::Unsupported(alloc::format!(
12148 "partition expansion: {parent_name:?} is not a parent"
12149 )));
12150 }
12151 };
12152 let key_col_name = parent.schema().columns[key_position].name.clone();
12153 // v7.37.16 (16.7) — for RANGE we extract a (lo, hi) interval
12154 // off the WHERE; for LIST / HASH we extract a single `=`
12155 // literal (and the rest of the planner falls back to "keep
12156 // every child" — same conservative path as 16.1/16.2).
12157 let (lo_bound, hi_bound) = match outer.where_.as_ref() {
12158 Some(expr) => extract_key_range(expr, &key_col_name),
12159 None => (None, None),
12160 };
12161 let eq_value: Option<spg_storage::Value<'static>> = match outer.where_.as_ref() {
12162 Some(expr) => extract_key_eq_value(expr, &key_col_name),
12163 None => None,
12164 };
12165 let children = crate::partition::children_of_parent(cat, parent_name);
12166 let mut kept: Vec<String> = Vec::new();
12167 let mut default_child: Option<String> = None;
12168 // First pass — apply per-strategy gates, defer DEFAULT until
12169 // we know whether some non-DEFAULT child matched.
12170 for child_name in &children {
12171 let Some(child) = cat.get(child_name) else {
12172 continue;
12173 };
12174 match &child.schema().partition_role {
12175 Some(PartitionRole::Range { lower, upper, .. }) => {
12176 if range_satisfies_filter(lower, upper, lo_bound.as_ref(), hi_bound.as_ref()) {
12177 kept.push(child_name.clone());
12178 }
12179 }
12180 // v7.37.16 (16.7) — LIST pruning: if WHERE has `key
12181 // = <lit>`, only the child whose values contain that
12182 // literal survives. Otherwise (no equality predicate
12183 // or planner couldn't extract one) keep the child
12184 // conservatively.
12185 Some(PartitionRole::List { values, .. }) => match &eq_value {
12186 Some(v) => {
12187 if values.iter().any(|b| b.equals_value(v)) {
12188 kept.push(child_name.clone());
12189 }
12190 }
12191 None => kept.push(child_name.clone()),
12192 },
12193 // v7.37.16 (16.7) — HASH pruning: with `key = <lit>`
12194 // we know the residue class deterministically, so
12195 // only the matching REMAINDER child survives.
12196 Some(PartitionRole::Hash {
12197 modulus, remainder, ..
12198 }) => match &eq_value {
12199 Some(v) => {
12200 let h = crate::partition::pg_compatible_hash(v);
12201 if h.rem_euclid(u64::from(*modulus)) == u64::from(*remainder) {
12202 kept.push(child_name.clone());
12203 }
12204 }
12205 None => kept.push(child_name.clone()),
12206 },
12207 Some(PartitionRole::Default { .. }) => {
12208 default_child = Some(child_name.clone());
12209 }
12210 _ => {}
12211 }
12212 }
12213 // PG-style DEFAULT semantics: the DEFAULT child must be
12214 // scanned iff some row could fall outside every concrete
12215 // child's bound predicate. We approximate that as "no
12216 // concrete child matched" (== full prune) — strictly
12217 // conservative for LIST / HASH (DEFAULT also catches rows
12218 // outside the union of value-sets / residues), and matches
12219 // PG for the equality case where we *do* know the routing
12220 // outcome.
12221 let _ = parent_kind; // used to silence dead-code lint while 16.8-9 lands.
12222 if let Some(d) = default_child {
12223 if kept.is_empty() {
12224 kept.push(d);
12225 } else if eq_value.is_none() {
12226 // Without an equality literal, the DEFAULT child may
12227 // still hold matching rows (e.g. LIKE on TEXT keys
12228 // for which a LIST partition exists). Keep it.
12229 kept.push(d);
12230 }
12231 }
12232 // Build the UNION ALL body text and re-parse — keeps the
12233 // rewrite expressible in surface SQL so the engine's existing
12234 // parser path handles the AST shape uniformly.
12235 if kept.is_empty() {
12236 // No children survive — caller falls back to scanning the
12237 // (empty) parent table. Returning None here is what
12238 // prevents the synthetic CTE from referring back to the
12239 // parent name and re-entering this rewrite pass.
12240 let _ = parent_name;
12241 return Ok(None);
12242 }
12243 // v7.39 (round 622, S05a) — the system columns of the CHILD the row
12244 // actually lives in.
12245 //
12246 // The parent is read through a synthetic CTE, so a `tableoid` on it
12247 // resolved against that CTE: every row of every child reported
12248 // `__spg_partition_pm`, an internal name no user ever typed, where
12249 // PG reports `pm_a` / `pm_b`. That is not only a leak — it silently
12250 // empties `WHERE tableoid::regclass::TEXT = 'pm_a'`, which is how
12251 // one asks "which partition is this row in", answering 0 rows where
12252 // PG answers 1. `ctid` had the same shape: it numbered the CTE's
12253 // output, so rows in different children got distinct ctids instead
12254 // of each child's own physical position.
12255 //
12256 // Naming them in the term is what carries them: the child scan
12257 // materialises its own six because the statement now references
12258 // them, and they land in SYSTEM_COLUMNS order right after the user
12259 // columns — the exact layout the positional `*` skip already
12260 // expects. Only done when the outer statement asks for one, so a
12261 // plain `SELECT * FROM parent` scans exactly what it scanned.
12262 let carry_sys = references_ctid(outer);
12263 let mut body = alloc::string::String::new();
12264 for (i, child_name) in kept.iter().enumerate() {
12265 if i > 0 {
12266 body.push_str(" UNION ALL ");
12267 }
12268 body.push_str("SELECT *");
12269 if carry_sys {
12270 for sys in SYSTEM_COLUMNS {
12271 body.push_str(", ");
12272 body.push_str(sys);
12273 }
12274 }
12275 body.push_str(" FROM ");
12276 body.push_str("e_ident_for_sql(child_name));
12277 }
12278 parse_select_or_corrupt(&body).map(Some)
12279 }
12280}
12281
12282/// Rewrite a `TableRef` pointing at a partition parent so it
12283/// references the synthetic CTE created by the expansion. If the
12284/// original ref had no alias, preserve the parent name as an alias
12285/// so column references like `events_partitioned.received_at`
12286/// keep resolving.
12287fn rewrite_partition_parent_table_ref(
12288 t: &mut spg_sql::ast::TableRef,
12289 parents: &[alloc::string::String],
12290 synth_name: &impl Fn(&str) -> alloc::string::String,
12291) {
12292 if t.lateral_subquery.is_some() || t.unnest_expr.is_some() || t.generate_series_args.is_some() {
12293 return;
12294 }
12295 // v7.39 (round 644) — an ONLY reference stays pointed at the parent
12296 // itself. The rewrite is keyed on the NAME, so in
12297 // `FROM ONLY po a JOIN po b` the un-qualified `b` put `po` on the
12298 // parent list and this then rewrote BOTH — including the one that
12299 // asked not to descend. PG answers 0 for that join; SPG answered 2.
12300 // Folded into the existing test — see the note in
12301 // `collect_partition_parent_refs` for what a separate one cost.
12302 if t.only || !parents.iter().any(|p| p == &t.name) {
12303 return;
12304 }
12305 if t.alias.is_none() {
12306 t.alias = Some(t.name.clone());
12307 }
12308 t.name = synth_name(&t.name);
12309}
12310
12311/// Walk a `TableRef` and push its `name` if it resolves to a partition
12312/// parent in `cat`. Skips `lateral_subquery` / `unnest_expr` /
12313/// `generate_series_args` references — those aren't catalog tables.
12314fn collect_partition_parent_refs(
12315 t: &spg_sql::ast::TableRef,
12316 cat: &spg_storage::Catalog,
12317 out: &mut Vec<alloc::string::String>,
12318) {
12319 if t.lateral_subquery.is_some() || t.unnest_expr.is_some() || t.generate_series_args.is_some() {
12320 return;
12321 }
12322 // v7.39 (round 644) — `FROM ONLY <parent>` scans the parent alone.
12323 // The keyword used to be absorbed at parse time, so this fanned out
12324 // anyway and `SELECT count(*) FROM ONLY <partitioned parent>`
12325 // answered 2 where PG answers 0.
12326 //
12327 // Folded into the existing test rather than given an early return of
12328 // its own: as two extra lines in this function's body it cost
12329 // `WHERE g BETWEEN 10 AND 20` **26x**, 5.9 ms to 155 ms, measured
12330 // outside the panel. Rounds 641 and 643 met the same wall from the
12331 // other two directions — adding to a hot function and taking away
12332 // from a cold one. What goes in a body near the row loop is a
12333 // codegen decision whatever its shape.
12334 if !t.only && crate::partition::has_children(cat, &t.name) {
12335 out.push(t.name.clone());
12336 }
12337}
12338
12339/// v7.37.6-B partition-key range derived from a WHERE expression.
12340/// `i64` microseconds since epoch with the same sign convention as
12341/// `Value::Timestamp`. Inclusive bool: `true` ⇒ inclusive(`>=` / `<=`
12342/// / `=`),`false` ⇒ exclusive(`>` / `<`).
12343#[derive(Debug, Clone, Copy)]
12344pub(crate) struct PartitionFilterBound {
12345 pub micros: i64,
12346 pub inclusive: bool,
12347}
12348
12349/// Walk a flat AND chain looking for `<key> <op> <timestamptz-literal>`
12350/// shapes; tighten the running lo / hi as we go. Anything outside that
12351/// (OR / nested calls / non-key columns)is ignored — caller treats
12352/// `None` as "no constraint on that side."
12353fn extract_key_range(
12354 expr: &spg_sql::ast::Expr,
12355 key_col: &str,
12356) -> (Option<PartitionFilterBound>, Option<PartitionFilterBound>) {
12357 let mut lo: Option<PartitionFilterBound> = None;
12358 let mut hi: Option<PartitionFilterBound> = None;
12359 let mut stack: Vec<&spg_sql::ast::Expr> = alloc::vec![expr];
12360 while let Some(e) = stack.pop() {
12361 match e {
12362 spg_sql::ast::Expr::Binary {
12363 lhs,
12364 op: spg_sql::ast::BinOp::And,
12365 rhs,
12366 } => {
12367 stack.push(lhs);
12368 stack.push(rhs);
12369 }
12370 // BETWEEN is desugared at parse time into `lhs >= low AND
12371 // lhs <= high`, so it lands here as two regular Binary
12372 // arms via the AND walker above.
12373 spg_sql::ast::Expr::Binary { lhs, op, rhs } => {
12374 let (col_ref, lit_side, swapped) = if is_column_ref(lhs, key_col) {
12375 (Some(lhs.as_ref()), rhs.as_ref(), false)
12376 } else if is_column_ref(rhs, key_col) {
12377 (Some(rhs.as_ref()), lhs.as_ref(), true)
12378 } else {
12379 (None, lhs.as_ref(), false)
12380 };
12381 if col_ref.is_none() {
12382 continue;
12383 }
12384 let Some(lit) = literal_to_micros(lit_side) else {
12385 continue;
12386 };
12387 use spg_sql::ast::BinOp::{Eq, Gt, GtEq, Lt, LtEq};
12388 let effective_op = if swapped {
12389 match op {
12390 Lt => Gt,
12391 LtEq => GtEq,
12392 Gt => Lt,
12393 GtEq => LtEq,
12394 other => *other,
12395 }
12396 } else {
12397 *op
12398 };
12399 match effective_op {
12400 Eq => {
12401 tighten_lo(
12402 &mut lo,
12403 PartitionFilterBound {
12404 micros: lit,
12405 inclusive: true,
12406 },
12407 );
12408 tighten_hi(
12409 &mut hi,
12410 PartitionFilterBound {
12411 micros: lit,
12412 inclusive: true,
12413 },
12414 );
12415 }
12416 GtEq => {
12417 tighten_lo(
12418 &mut lo,
12419 PartitionFilterBound {
12420 micros: lit,
12421 inclusive: true,
12422 },
12423 );
12424 }
12425 Gt => {
12426 tighten_lo(
12427 &mut lo,
12428 PartitionFilterBound {
12429 micros: lit,
12430 inclusive: false,
12431 },
12432 );
12433 }
12434 LtEq => {
12435 tighten_hi(
12436 &mut hi,
12437 PartitionFilterBound {
12438 micros: lit,
12439 inclusive: true,
12440 },
12441 );
12442 }
12443 Lt => {
12444 tighten_hi(
12445 &mut hi,
12446 PartitionFilterBound {
12447 micros: lit,
12448 inclusive: false,
12449 },
12450 );
12451 }
12452 _ => {}
12453 }
12454 }
12455 _ => {}
12456 }
12457 }
12458 (lo, hi)
12459}
12460
12461fn tighten_lo(slot: &mut Option<PartitionFilterBound>, new: PartitionFilterBound) {
12462 match slot {
12463 None => *slot = Some(new),
12464 Some(cur) => {
12465 if new.micros > cur.micros
12466 || (new.micros == cur.micros && !new.inclusive && cur.inclusive)
12467 {
12468 *slot = Some(new);
12469 }
12470 }
12471 }
12472}
12473
12474fn tighten_hi(slot: &mut Option<PartitionFilterBound>, new: PartitionFilterBound) {
12475 match slot {
12476 None => *slot = Some(new),
12477 Some(cur) => {
12478 if new.micros < cur.micros
12479 || (new.micros == cur.micros && !new.inclusive && cur.inclusive)
12480 {
12481 *slot = Some(new);
12482 }
12483 }
12484 }
12485}
12486
12487fn is_column_ref(e: &spg_sql::ast::Expr, key_col: &str) -> bool {
12488 if let spg_sql::ast::Expr::Column(c) = e {
12489 c.name.eq_ignore_ascii_case(key_col)
12490 } else {
12491 false
12492 }
12493}
12494
12495/// v7.37.16 (16.7) — walk an AND-chain WHERE and pull a single
12496/// `key_col = <literal>` predicate out for LIST/HASH partition
12497/// pruning. Returns `None` when no equality literal can be lifted
12498/// (planner then keeps every child — correctness preserved). The
12499/// returned `Value<'static>` is an owned coercion so the caller can
12500/// outlive any AST node it was extracted from.
12501pub(crate) fn extract_key_eq_value(
12502 expr: &spg_sql::ast::Expr,
12503 key_col: &str,
12504) -> Option<spg_storage::Value<'static>> {
12505 let mut stack: Vec<&spg_sql::ast::Expr> = alloc::vec![expr];
12506 while let Some(e) = stack.pop() {
12507 match e {
12508 spg_sql::ast::Expr::Binary {
12509 lhs,
12510 op: spg_sql::ast::BinOp::And,
12511 rhs,
12512 } => {
12513 stack.push(lhs);
12514 stack.push(rhs);
12515 }
12516 spg_sql::ast::Expr::Binary {
12517 lhs,
12518 op: spg_sql::ast::BinOp::Eq,
12519 rhs,
12520 } => {
12521 let lit_side = if is_column_ref(lhs, key_col) {
12522 rhs.as_ref()
12523 } else if is_column_ref(rhs, key_col) {
12524 lhs.as_ref()
12525 } else {
12526 continue;
12527 };
12528 let cloned = lit_side.clone();
12529 let Ok(v) = crate::conversions::literal_expr_to_value(cloned) else {
12530 continue;
12531 };
12532 // Coerce to an owned Value<'static> so the caller
12533 // can hold it past the WHERE expression's lifetime.
12534 let owned: spg_storage::Value<'static> = match v {
12535 spg_storage::Value::Text(s) => {
12536 spg_storage::Value::Text(alloc::borrow::Cow::Owned(s.into_owned()))
12537 }
12538 spg_storage::Value::SmallInt(n) => spg_storage::Value::SmallInt(n),
12539 spg_storage::Value::Int(n) => spg_storage::Value::Int(n),
12540 spg_storage::Value::BigInt(n) => spg_storage::Value::BigInt(n),
12541 spg_storage::Value::Date(d) => spg_storage::Value::Date(d),
12542 spg_storage::Value::Timestamp(t) => spg_storage::Value::Timestamp(t),
12543 spg_storage::Value::Bool(b) => spg_storage::Value::Bool(b),
12544 spg_storage::Value::Null => spg_storage::Value::Null,
12545 // Anything else (Vector / Json / Bytes / Numeric /
12546 // arrays / interval / …) isn't a current partition
12547 // key type; skip without pruning.
12548 _ => continue,
12549 };
12550 return Some(owned);
12551 }
12552 _ => {}
12553 }
12554 }
12555 None
12556}
12557
12558/// Coerce a literal Expr(after the parser folded sequence calls etc.)
12559/// to i64 microseconds. Mirrors `evaluate_partition_bound`'s shape so
12560/// pruning and routing agree on the literal vocabulary. Returns
12561/// `None` when the literal isn't recognised(planner then skips
12562/// pruning on that branch — correctness preserved).
12563fn literal_to_micros(e: &spg_sql::ast::Expr) -> Option<i64> {
12564 let cloned = e.clone();
12565 let value = crate::conversions::literal_expr_to_value(cloned).ok()?;
12566 match value {
12567 spg_storage::Value::Timestamp(m) => Some(m),
12568 spg_storage::Value::Date(days) => Some(i64::from(days) * 86_400i64 * 1_000_000i64),
12569 spg_storage::Value::Text(s) => crate::eval::parse_timestamp_literal(&s),
12570 _ => None,
12571 }
12572}
12573
12574/// `[range_lo, range_hi)` of a child is kept iff it can hold any row
12575/// satisfying the WHERE-derived filter range. PG-style half-open:
12576/// child upper exclusive. Filter inclusivity is honoured per-bound.
12577fn range_satisfies_filter(
12578 range_lo: &spg_storage::PartitionBound,
12579 range_hi: &spg_storage::PartitionBound,
12580 filter_lo: Option<&PartitionFilterBound>,
12581 filter_hi: Option<&PartitionFilterBound>,
12582) -> bool {
12583 use spg_storage::PartitionBound;
12584 // For each filter side, reject children that can't host any row
12585 // matching the predicate.
12586 if let Some(lo) = filter_lo {
12587 // child upper bound vs filter lower:
12588 // if filter is x >= L, child rejects iff child.hi <= L
12589 // if filter is x > L, child rejects iff child.hi <= L
12590 // (child.hi exclusive, so equality with L still rejects)
12591 match range_hi {
12592 PartitionBound::MinValue => return false,
12593 PartitionBound::MaxValue => {}
12594 PartitionBound::TimestampTz(hi) => {
12595 if *hi <= lo.micros {
12596 return false;
12597 }
12598 }
12599 // v7.37.16 (16.6) — non-TIMESTAMPTZ bounds aren't
12600 // matched against TIMESTAMPTZ filters here; keep child
12601 // (conservative: don't prune).
12602 PartitionBound::BigInt(_)
12603 | PartitionBound::Int(_)
12604 | PartitionBound::SmallInt(_)
12605 | PartitionBound::Date(_)
12606 | PartitionBound::Text(_) => {}
12607 }
12608 }
12609 if let Some(hi) = filter_hi {
12610 // child lower bound vs filter upper:
12611 // if filter is x <= U, child rejects iff child.lo > U
12612 // if filter is x < U, child rejects iff child.lo >= U
12613 match range_lo {
12614 PartitionBound::MaxValue => return false,
12615 PartitionBound::MinValue => {}
12616 PartitionBound::TimestampTz(lo) => {
12617 let rejects = if hi.inclusive {
12618 *lo > hi.micros
12619 } else {
12620 *lo >= hi.micros
12621 };
12622 if rejects {
12623 return false;
12624 }
12625 }
12626 PartitionBound::BigInt(_)
12627 | PartitionBound::Int(_)
12628 | PartitionBound::SmallInt(_)
12629 | PartitionBound::Date(_)
12630 | PartitionBound::Text(_) => {}
12631 }
12632 }
12633 true
12634}
12635
12636fn quote_ident_for_sql(name: &str) -> alloc::string::String {
12637 // Match spg-sql's quoting rule(unquoted when ASCII-lowercase
12638 // identifier, otherwise quoted). Conservative: always quote so
12639 // children with reserved names round-trip safely through the
12640 // CTE-body parse.
12641 let mut out = alloc::string::String::with_capacity(name.len() + 2);
12642 out.push('"');
12643 for c in name.chars() {
12644 if c == '"' {
12645 out.push('"');
12646 }
12647 out.push(c);
12648 }
12649 out.push('"');
12650 out
12651}
12652
12653fn parse_select_or_corrupt(sql: &str) -> Result<SelectStatement, EngineError> {
12654 let parsed = spg_sql::parser::parse_statement(sql).map_err(|e| {
12655 EngineError::Unsupported(alloc::format!(
12656 "partition expansion: generated SQL {sql:?} failed to re-parse: {e}"
12657 ))
12658 })?;
12659 let Statement::Select(body) = parsed else {
12660 return Err(EngineError::Unsupported(alloc::format!(
12661 "partition expansion: generated SQL {sql:?} is not a SELECT"
12662 )));
12663 };
12664 Ok(body)
12665}
12666
12667/// v7.39 (read01 round 65/66) — the column shape a set-returning function
12668/// exposes. `RETURNS TABLE(id int, v text)` names them; a `SETOF <scalar>`
12669/// yields ONE column named after the call's alias when there is one (`FROM
12670/// odds() AS x` → `x`), else after the function. Get this wrong and the alias
12671/// resolves to the whole ROW: `SELECT x::text FROM odds() AS x` renders `(1)`.
12672fn setof_column_shape_from(
12673 declared: &str,
12674 name: &str,
12675 alias: Option<&str>,
12676 got: &[ColumnSchema],
12677) -> alloc::vec::Vec<ColumnSchema> {
12678 let upper = declared.to_ascii_uppercase();
12679 if upper.starts_with("TABLE(") {
12680 let raw = &declared["TABLE(".len()..declared.len() - 1];
12681 return raw
12682 .split(',')
12683 .zip(got.iter())
12684 .map(|(decl, g)| {
12685 let cname = decl.split_whitespace().next().unwrap_or(g.name.as_str());
12686 ColumnSchema::new(cname.to_string(), g.ty, true)
12687 })
12688 .collect();
12689 }
12690 let cname = alias.unwrap_or(name);
12691 got.first()
12692 .map(|c| alloc::vec![ColumnSchema::new(cname.to_string(), c.ty, true)])
12693 .unwrap_or_default()
12694}
12695
12696/// The plpgsql twin: the interpreter hands back raw value rows, so the types
12697/// come off the first row.
12698fn setof_column_shape(
12699 declared: &str,
12700 name: &str,
12701 alias: Option<&str>,
12702 first_row: Option<&alloc::vec::Vec<Value<'static>>>,
12703) -> alloc::vec::Vec<ColumnSchema> {
12704 let got: alloc::vec::Vec<ColumnSchema> = first_row
12705 .map(|r| {
12706 r.iter()
12707 .enumerate()
12708 .map(|(i, v)| {
12709 ColumnSchema::new(
12710 alloc::format!("col{i}"),
12711 v.data_type().unwrap_or(DataType::Text),
12712 true,
12713 )
12714 })
12715 .collect()
12716 })
12717 .unwrap_or_default();
12718 setof_column_shape_from(declared, name, alias, &got)
12719}
12720
12721/// v7.39 (read01 round 67) — expand every set-returning call in a target list
12722/// for ONE input row, PG's ProjectSet semantics.
12723///
12724/// Several SRFs in one list run in **LOCKSTEP**, not as a cross product: the
12725/// output has as many rows as the LONGEST of them, and a shorter one is padded
12726/// with NULLs. (`SELECT generate_series(1,3), generate_series(10,11)` →
12727/// `1/10, 2/11, 3/NULL`.) A single SRF is the degenerate case of that, and an
12728/// SRF that yields no rows at all contributes none — `SELECT unnest('{}'::int[])`
12729/// is zero rows, not one NULL row.
12730///
12731/// Non-SRF items repeat, evaluated once per output row from the same input row.
12732/// v7.39 (read01 round 79) — where an aggregate may NOT appear. Both of these
12733/// used to reach the scalar function dispatcher, which reported the aggregate as
12734/// an *unknown function* — the same "symptom two layers above the cause" shape
12735/// round 78 found with SRFs. Neither can be diagnosed down there: the dispatcher
12736/// sees a call, not the clause it came from. The statement knows.
12737/// v7.39 (round 294, E3 Phase 1b) — PG's rules on WHERE a row-locking
12738/// clause may appear.
12739///
12740/// PG rejects `FOR UPDATE` on exactly the shapes that have no
12741/// identifiable base row to lock, each with its own wording. SPG
12742/// accepted all of them and locked nothing, so a query that PG refuses
12743/// outright came back looking like it had taken locks.
12744///
12745/// Every wording read off live PG 18.4.
12746fn validate_locking_clause(stmt: &SelectStatement) -> Result<(), EngineError> {
12747 let Some(lock) = &stmt.locking else {
12748 return Ok(());
12749 };
12750 let verb = lock_clause_verb(lock.strength);
12751 let refuse = |what: &str| {
12752 Err(EngineError::Unsupported(alloc::format!(
12753 "{verb} is not allowed with {what}"
12754 )))
12755 };
12756 if !stmt.unions.is_empty() {
12757 return refuse("UNION/INTERSECT/EXCEPT");
12758 }
12759 if stmt.distinct || !stmt.distinct_on.is_empty() {
12760 return refuse("DISTINCT clause");
12761 }
12762 if stmt.group_by.is_some() || stmt.group_by_all {
12763 return refuse("GROUP BY clause");
12764 }
12765 let has_agg = stmt.items.iter().any(|it| match it {
12766 spg_sql::ast::SelectItem::Expr { expr, .. } => crate::aggregate::contains_aggregate(expr),
12767 _ => false,
12768 });
12769 if has_agg {
12770 return refuse("aggregate functions");
12771 }
12772 // `FOR UPDATE OF t` must name a relation that is actually in FROM.
12773 for want in &lock.of_tables {
12774 if !locking_from_names(stmt)
12775 .iter()
12776 .any(|n| n.eq_ignore_ascii_case(want))
12777 {
12778 return Err(EngineError::Unsupported(alloc::format!(
12779 "relation \"{want}\" in {verb} clause not found in FROM clause"
12780 )));
12781 }
12782 }
12783 Ok(())
12784}
12785
12786/// How PG names the clause in its diagnostics.
12787const fn lock_clause_verb(s: spg_sql::ast::LockStrength) -> &'static str {
12788 use spg_sql::ast::LockStrength as LS;
12789 match s {
12790 LS::Update => "FOR UPDATE",
12791 LS::NoKeyUpdate => "FOR NO KEY UPDATE",
12792 LS::Share => "FOR SHARE",
12793 LS::KeyShare => "FOR KEY SHARE",
12794 }
12795}
12796
12797/// Every relation name (or alias) the FROM clause exposes.
12798fn locking_from_names(stmt: &SelectStatement) -> alloc::vec::Vec<String> {
12799 let mut out = alloc::vec::Vec::new();
12800 if let Some(f) = &stmt.from {
12801 let mut push = |t: &spg_sql::ast::TableRef| {
12802 if let Some(a) = &t.alias {
12803 out.push(a.clone());
12804 }
12805 out.push(t.name.clone());
12806 };
12807 push(&f.primary);
12808 for j in &f.joins {
12809 push(&j.table);
12810 }
12811 }
12812 out
12813}
12814
12815fn validate_aggregate_placement(stmt: &SelectStatement) -> Result<(), EngineError> {
12816 use spg_sql::ast::Expr;
12817 if let Some(w) = &stmt.where_
12818 && aggregate::contains_aggregate(w)
12819 {
12820 return Err(EngineError::Unsupported(
12821 "aggregate functions are not allowed in WHERE".into(),
12822 ));
12823 }
12824 let mut nested = false;
12825 let mut check = |e: &Expr| {
12826 let mut probe = e.clone();
12827 crate::expr_analysis::rewrite_nodes_mut(&mut probe, &mut |n| {
12828 let args = match n {
12829 Expr::FunctionCall { name, args } if aggregate::is_aggregate_name(name) => args,
12830 _ => return false,
12831 };
12832 if args.iter().any(aggregate::contains_aggregate) {
12833 nested = true;
12834 }
12835 false
12836 });
12837 };
12838 for it in &stmt.items {
12839 if let spg_sql::ast::SelectItem::Expr { expr, .. } = it {
12840 check(expr);
12841 }
12842 }
12843 if let Some(h) = &stmt.having {
12844 check(h);
12845 }
12846 for o in &stmt.order_by {
12847 check(&o.expr);
12848 }
12849 if nested {
12850 return Err(EngineError::Unsupported(
12851 "aggregate function calls cannot be nested".into(),
12852 ));
12853 }
12854 Ok(())
12855}
12856
12857/// v7.39 (read01 round 78) — an SRF may sit ANYWHERE inside a target-list
12858/// expression, not only as the whole item: `upper(unnest(a))`, `unnest(a) + 10`,
12859/// `'x:' || unnest(a)`, `(regexp_matches(s, p, 'g'))::text`. PG evaluates the SRF
12860/// to a set and then applies the enclosing expression once per element. SPG only
12861/// ever recognised an SRF that WAS the item, so everything above died on
12862/// "unknown function unnest" — the set-returning call, wrapped in anything at
12863/// all, fell through to the scalar function dispatcher which has no such name.
12864///
12865/// Each SRF node is lifted out into a synthetic column (`__srf_k`), the tree is
12866/// rewritten to read that column, and the rewritten expression is evaluated once
12867/// per output row against the input row extended with the lifted values. The
12868/// lift is by VALUE, not by literal: a text[] or a jsonb keeps its type exactly.
12869/// v7.39 (read01 round 80) — `ORDER BY <n>` names the Nth OUTPUT column. Three
12870/// executors (the single-table scan, the synthetic-table pipeline, and the
12871/// unnest FROM path) each evaluated the key as an ordinary expression, where the
12872/// literal `n` is just the constant n — the same sort key for every row. The
12873/// sort therefore ran and changed nothing, which is why nobody noticed: rows came
12874/// back in input order, not in a wrong order. Statement prep resolves the common
12875/// case, but only when the SELECT item is an expression — a `*` is not one, and
12876/// `SELECT unnest(a) x` becomes `SELECT * FROM unnest(a) x`, so the everyday
12877/// spelling landed on exactly the shape prep could not resolve.
12878///
12879/// A set-returning item is left alone: copying it into ORDER BY would make the
12880/// key "the whole set", evaluated once per INPUT row.
12881fn resolve_positional_order_by(
12882 order_by: &[spg_sql::ast::OrderBy],
12883 projection: &[ProjectedItem],
12884) -> alloc::vec::Vec<spg_sql::ast::OrderBy> {
12885 order_by
12886 .iter()
12887 .filter_map(|o| {
12888 let mut o = o.clone();
12889 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &o.expr
12890 && *n >= 1
12891 && let Ok(idx) = usize::try_from(*n - 1)
12892 && let Some(item) = projection.get(idx)
12893 && !expr_contains_builtin_srf(&item.expr)
12894 {
12895 // 7.38.1 S6.1 (gendiff fourth leg) — an ordinal whose
12896 // item is itself an integer LITERAL must not be
12897 // substituted textually: the literal would read as an
12898 // ordinal again downstream, and `SELECT 10 … ORDER BY
12899 // 1` died with "position 10 is not in select list"
12900 // where PG happily returns the rows. Ordering by a
12901 // constant orders nothing, so the key drops.
12902 if matches!(item.expr, Expr::Literal(spg_sql::ast::Literal::Integer(_))) {
12903 return None;
12904 }
12905 o.expr = item.expr.clone();
12906 }
12907 Some(o)
12908 })
12909 .collect()
12910}
12911
12912/// v7.39 (read01 round 80) — does a BUILTIN set-returning call appear anywhere in
12913/// this expression? Statement preparation (`resolve_order_by_position`) runs
12914/// before any catalog is in hand, and it only needs to know "is this item's value
12915/// a set", which the builtin SRFs answer syntactically.
12916pub(crate) fn expr_contains_builtin_srf(e: &spg_sql::ast::Expr) -> bool {
12917 let mut found = false;
12918 let mut probe = e.clone();
12919 crate::expr_analysis::rewrite_nodes_mut(&mut probe, &mut |n| {
12920 if is_top_level_unnest(n) {
12921 found = true;
12922 return true;
12923 }
12924 false
12925 });
12926 found
12927}
12928
12929/// v7.39 (round 599) — everything about a target-list SRF that does not
12930/// depend on the row.
12931///
12932/// `expand_srf_row` derived all of this again for EVERY input row: it cloned
12933/// each SRF-bearing projection expression, walked and rewrote the tree,
12934/// formatted a `__srf_N` name per node, and copied the whole column schema.
12935/// A counting allocator put the path at 24 allocations per input row for a
12936/// single-element `unnest`, against 0 for the same scan without one — 211 MB
12937/// where the plain scan took 4.3 — and the shape held whatever the array
12938/// contained, which is what invariant work looks like.
12939struct SrfPlan {
12940 /// The lifted SRF calls, in slot order.
12941 nodes: alloc::vec::Vec<spg_sql::ast::Expr>,
12942 /// Per projection position, the expression with its SRF calls replaced
12943 /// by `__srf_N` column references. `None` means the item has none.
12944 rewritten: alloc::vec::Vec<Option<spg_sql::ast::Expr>>,
12945 /// The input schema followed by one column per slot. Only the slots'
12946 /// TYPES vary per row, and they are patched in place.
12947 ext_cols: alloc::vec::Vec<ColumnSchema>,
12948 /// v7.39 (round 743) — the rewritten projection COMPILED against the
12949 /// extended schema, once per plan. The per-output-row evaluation ran
12950 /// the interpreter (~560 ns/row on the unnest panel cell); the Step
12951 /// VM reads the `__srf_N` slots as plain columns. `None` = that item
12952 /// is not fully compilable and keeps the interpreter.
12953 compiled: alloc::vec::Vec<Option<eval::CompiledExpr>>,
12954 base_cols: usize,
12955}
12956
12957fn build_srf_plan(
12958 engine: &Engine,
12959 projection: &[ProjectedItem],
12960 srf_idxs: &[usize],
12961 ctx: &EvalContext<'_>,
12962) -> Result<SrfPlan, EngineError> {
12963 // Lift every SRF node out of every item that contains one.
12964 let mut nodes: Vec<spg_sql::ast::Expr> = Vec::new();
12965 let mut rewritten: Vec<Option<spg_sql::ast::Expr>> = alloc::vec![None; projection.len()];
12966 let mut reject: Option<EngineError> = None;
12967 for &i in srf_idxs {
12968 let mut e = projection[i].expr.clone();
12969 crate::expr_analysis::rewrite_nodes_mut(&mut e, &mut |n| {
12970 if reject.is_some() {
12971 return true;
12972 }
12973 // PG refuses a set-returning function inside a conditional: the set
12974 // would have to be produced before anyone knows whether the branch
12975 // is even taken.
12976 let conditional = match n {
12977 spg_sql::ast::Expr::Case { .. } => Some("CASE"),
12978 spg_sql::ast::Expr::FunctionCall { name, .. }
12979 if name.eq_ignore_ascii_case("coalesce") =>
12980 {
12981 Some("COALESCE")
12982 }
12983 _ => None,
12984 };
12985 if let Some(kind) = conditional
12986 && engine.expr_contains_srf(n)
12987 {
12988 reject = Some(EngineError::Unsupported(alloc::format!(
12989 "set-returning functions are not allowed in {kind}"
12990 )));
12991 return true;
12992 }
12993 if !engine.is_srf_node(n) {
12994 return false;
12995 }
12996 let slot = nodes.len();
12997 nodes.push(n.clone());
12998 *n = spg_sql::ast::Expr::Column(spg_sql::ast::ColumnName {
12999 qualifier: None,
13000 name: alloc::format!("__srf_{slot}"),
13001 });
13002 true
13003 });
13004 rewritten[i] = Some(e);
13005 }
13006 if let Some(err) = reject {
13007 return Err(err);
13008 }
13009 let base_cols = ctx.columns.len();
13010 let mut ext_cols: Vec<ColumnSchema> = ctx.columns.to_vec();
13011 for slot in 0..nodes.len() {
13012 ext_cols.push(ColumnSchema::new(
13013 alloc::format!("__srf_{slot}"),
13014 DataType::Text,
13015 true,
13016 ));
13017 }
13018 // v7.39 (round 743) — compile the rewritten items against the
13019 // EXTENDED schema. The slot columns' declared type is a per-row
13020 // patched detail the compiled column read does not consult.
13021 let compiled: Vec<Option<eval::CompiledExpr>> = {
13022 let mut ext_ctx = ctx.clone();
13023 ext_ctx.columns = &ext_cols;
13024 projection
13025 .iter()
13026 .enumerate()
13027 .map(|(i, p)| {
13028 let e = rewritten[i].as_ref().unwrap_or(&p.expr);
13029 if eval::fully_compilable(e) {
13030 Some(eval::compile_expr(e, &ext_ctx))
13031 } else {
13032 None
13033 }
13034 })
13035 .collect()
13036 };
13037 Ok(SrfPlan {
13038 nodes,
13039 rewritten,
13040 ext_cols,
13041 compiled,
13042 base_cols,
13043 })
13044}
13045
13046/// One input row expanded through a plan built once for the whole scan.
13047/// v7.39 (round 621) — expand a projection whose target list contains
13048/// set-returning items, remembering which INPUT row each output row came from.
13049///
13050/// The three materialised-source tails — `FROM unnest(…)`, `FROM
13051/// generate_series(…)`, and the one that serves VALUES / a derived table /
13052/// `ROWS FROM (…)` — are near-copies of each other, and only the first knew
13053/// about target-list SRFs. So `SELECT unnest(ARRAY[1,2]), x FROM (VALUES (3),(4))
13054/// v(x)` answered `function unnest(integer[]) does not exist` on all the
13055/// others, for a query PG answers. Sharing the expansion is the point: a
13056/// fourth copy would have been the fourth place to forget.
13057fn expand_projection_srfs(
13058 engine: &Engine,
13059 projection: &[ProjectedItem],
13060 srf_idxs: &[usize],
13061 filtered: &[Row<'static>],
13062 ctx: &EvalContext<'_>,
13063) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<usize>), EngineError> {
13064 let mut out = alloc::vec::Vec::with_capacity(filtered.len());
13065 let mut src = alloc::vec::Vec::with_capacity(filtered.len());
13066 // v7.39 (round 726) — ONE plan for the whole scan. The per-row
13067 // spelling rebuilt it for every input row: a full clone of the
13068 // rewritten projection trees and the extended schema, 50k times on
13069 // the panel's unnest cell.
13070 let mut plan = build_srf_plan(engine, projection, srf_idxs, ctx)?;
13071 // v7.39 (round 733) — shard the expansion. Each shard clones the
13072 // plan (its ext_cols slot types are per-row mutable) and builds a
13073 // MINIMAL context — EvalContext is not Sync — which is sound only
13074 // when every expression involved is pure: the whole projection and
13075 // every SRF argument must be fully_compilable, or the row loop
13076 // stays serial with the full session context.
13077 // The projection is judged in its REWRITTEN form — the SRF call
13078 // itself is never compilable, but after the lift it is a plain
13079 // `__srf_N` column reference.
13080 let all_pure = projection
13081 .iter()
13082 .enumerate()
13083 .all(|(i, p)| eval::fully_compilable(plan.rewritten[i].as_ref().unwrap_or(&p.expr)))
13084 && plan.nodes.iter().all(|n| match n {
13085 Expr::FunctionCall { args, .. } => args.iter().all(eval::fully_compilable),
13086 other => eval::fully_compilable(other),
13087 });
13088 if all_pure
13089 && filtered.len() >= crate::PARALLEL_MIN_ROWS / 5
13090 && let Some(r) = engine.parallel_runner.0.as_deref()
13091 {
13092 let n_shards = (filtered.len() / (crate::PARALLEL_MIN_ROWS / 5)).clamp(2, 8);
13093 let chunk = filtered.len().div_ceil(n_shards);
13094 type ShardOut = Result<(Vec<Row<'static>>, Vec<usize>), EngineError>;
13095 let schema_cols = ctx.columns;
13096 let alias = ctx.table_alias;
13097 let mysql = ctx.mysql_dialect;
13098 let style = ctx.render_style;
13099 let plan_ref = &plan;
13100 let results = r.run_shards(n_shards, &|si| {
13101 let lo = si * chunk;
13102 let hi = ((si + 1) * chunk).min(filtered.len());
13103 let mut sctx = eval::EvalContext::new(schema_cols, alias);
13104 sctx.mysql_dialect = mysql;
13105 sctx.render_style = style;
13106 // v7.39 (round 743) — SrfPlan is no longer Clone (it carries
13107 // compiled programs); each shard rebuilds it, which also
13108 // recompiles against the shard's own context. Build errors
13109 // were already surfaced by the outer build above.
13110 let mut local_plan = match build_srf_plan(engine, projection, srf_idxs, &sctx) {
13111 Ok(p) => p,
13112 Err(e) => return alloc::boxed::Box::new(ShardOut::Err(e)) as _,
13113 };
13114 let mut run = || -> ShardOut {
13115 let mut o: Vec<Row<'static>> = Vec::with_capacity(hi - lo);
13116 let mut sidx: Vec<usize> = Vec::with_capacity(hi - lo);
13117 for (i, row) in filtered[lo..hi].iter().enumerate() {
13118 let expanded =
13119 expand_srf_row_with(engine, &mut local_plan, projection, row, &sctx)?;
13120 sidx.extend(core::iter::repeat_n(lo + i, expanded.len()));
13121 o.extend(expanded);
13122 }
13123 Ok((o, sidx))
13124 };
13125 alloc::boxed::Box::new(run())
13126 });
13127 for boxed in results {
13128 let shard = boxed
13129 .downcast::<ShardOut>()
13130 .expect("runner echoes the closure's box");
13131 let (o, sidx) = (*shard)?;
13132 out.extend(o);
13133 src.extend(sidx);
13134 }
13135 return Ok((out, src));
13136 }
13137 for (i, row) in filtered.iter().enumerate() {
13138 let expanded = expand_srf_row_with(engine, &mut plan, projection, row, ctx)?;
13139 src.extend(core::iter::repeat_n(i, expanded.len()));
13140 out.extend(expanded);
13141 }
13142 Ok((out, src))
13143}
13144
13145/// v7.39 (round 621) — one ORDER BY key, read from wherever it lives.
13146///
13147/// A key that names a select-list item reads it out of the EXPANDED row,
13148/// because PG sorts after the expansion. A key that names a source column the
13149/// query does not project is evaluated against the input row that output row
13150/// came from. `out_col` is `srf_order_output_cols`'s verdict for this key.
13151fn srf_order_key(
13152 ob: &spg_sql::ast::OrderBy,
13153 out_col: Option<usize>,
13154 out: &Row<'static>,
13155 src: &Row<'static>,
13156 ctx: &EvalContext<'_>,
13157) -> Result<Value<'static>, EngineError> {
13158 match out_col {
13159 Some(i) => Ok(out.values.get(i).cloned().unwrap_or(Value::Null)),
13160 None => eval::eval_expr(&ob.expr, src, ctx).map_err(EngineError::Eval),
13161 }
13162}
13163
13164fn expand_srf_row_with(
13165 engine: &Engine,
13166 plan: &mut SrfPlan,
13167 projection: &[ProjectedItem],
13168 row: &Row<'static>,
13169 ctx: &EvalContext<'_>,
13170) -> Result<Vec<Row<'static>>, EngineError> {
13171 let mut lists: Vec<Vec<Value<'static>>> = Vec::with_capacity(plan.nodes.len());
13172 for n in &plan.nodes {
13173 lists.push(engine.srf_values(n, row, ctx)?);
13174 }
13175 let n_rows = lists.iter().map(Vec::len).max().unwrap_or(0);
13176 // Only the slots' element types depend on the row; the names and the
13177 // input schema around them do not.
13178 for (slot, list) in lists.iter().enumerate() {
13179 plan.ext_cols[plan.base_cols + slot].ty = list
13180 .iter()
13181 .find_map(|v| v.data_type())
13182 .unwrap_or(DataType::Text);
13183 }
13184 let mut ext_ctx = ctx.clone();
13185 ext_ctx.columns = &plan.ext_cols;
13186 let mut out = Vec::with_capacity(n_rows);
13187 // v7.39 (round 726) — the base columns are the SAME for every
13188 // expanded row; clone them once and rewrite only the SRF slots per
13189 // k. The old form cloned the whole input row per OUTPUT row — for
13190 // `unnest(ARRAY[id, g])` over d that was a 100k-fold clone of a
13191 // TEXT column the projection never reads.
13192 let base_len = row.values.len();
13193 let mut ext_vals = row.values.clone();
13194 ext_vals.resize(base_len + lists.len(), Value::Null);
13195 let mut eval_stack: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
13196 for k in 0..n_rows {
13197 for (slot, list) in lists.iter().enumerate() {
13198 // Past the end of THIS srf's rows → NULL (PG pads).
13199 ext_vals[base_len + slot] = list.get(k).cloned().unwrap_or(Value::Null);
13200 }
13201 let ext_row = Row::new(core::mem::take(&mut ext_vals));
13202 let mut vals = Vec::with_capacity(projection.len());
13203 for (i, p) in projection.iter().enumerate() {
13204 // v7.39 (round 743) — compiled when possible; the
13205 // interpreter for the rest, with its exact wording.
13206 vals.push(match &plan.compiled[i] {
13207 Some(c) => eval::eval_compiled(c, &ext_row, &ext_ctx, &mut eval_stack)
13208 .map_err(EngineError::Eval)?,
13209 None => {
13210 let expr = plan.rewritten[i].as_ref().unwrap_or(&p.expr);
13211 eval::eval_expr(expr, &ext_row, &ext_ctx).map_err(EngineError::Eval)?
13212 }
13213 });
13214 }
13215 ext_vals = ext_row.values;
13216 out.push(Row::new(vals));
13217 }
13218 Ok(out)
13219}
13220
13221/// The one-shot spelling, for the callers that expand a single row.
13222/// v7.39 (round 600) — which output column each ORDER BY key names, for a
13223/// query whose target list contains a set-returning function.
13224///
13225/// The keys used to be built from the INPUT row, before the SRF expanded, so
13226/// anything that named the SRF's own output was evaluated as a scalar call:
13227/// `SELECT unnest(ARRAY[g,id]) v FROM sr ORDER BY v` answered
13228/// "function unnest(integer[]) does not exist", and so did the spellings that
13229/// repeat the call or reach it through `ORDER BY 1`. Where it did not error
13230/// it silently did nothing — `SELECT DISTINCT unnest(…) … ORDER BY 1` came
13231/// back in input order. PG sorts AFTER the expansion, so a key that names a
13232/// select-list item reads that item's value out of the expanded row.
13233///
13234/// `None` keeps the key on the input row, which is where an ORDER BY naming
13235/// a column the query does not project has to be evaluated.
13236fn srf_order_output_cols(
13237 order_by: &[spg_sql::ast::OrderBy],
13238 projection: &[ProjectedItem],
13239) -> Vec<Option<usize>> {
13240 order_by
13241 .iter()
13242 .map(|ob| {
13243 // A positive ordinal is the Nth output column, directly.
13244 // `resolve_positional_order_by` deliberately leaves an ordinal
13245 // pointing at a set-returning item alone — copying the call into
13246 // ORDER BY would have made the key "the whole set" back when keys
13247 // came from the input row. Reading the expanded row's column is
13248 // what it should have meant, and is what this does.
13249 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &ob.expr
13250 && *n >= 1
13251 && let Ok(idx) = usize::try_from(*n - 1)
13252 && idx < projection.len()
13253 {
13254 return Some(idx);
13255 }
13256 // An unqualified name matching exactly one output name. SQL
13257 // resolves ORDER BY against the select list first, so this wins
13258 // over an input column of the same name — which is the whole
13259 // point of `SELECT g AS id … ORDER BY id`.
13260 if let Expr::Column(c) = &ob.expr
13261 && c.qualifier.is_none()
13262 {
13263 let mut hit = None;
13264 for (i, p) in projection.iter().enumerate() {
13265 if p.output_name.eq_ignore_ascii_case(&c.name) {
13266 if hit.is_some() {
13267 hit = None;
13268 break;
13269 }
13270 hit = Some(i);
13271 }
13272 }
13273 if hit.is_some() {
13274 return hit;
13275 }
13276 }
13277 // Or the same expression as a select-list item — which is what
13278 // `ORDER BY 1` becomes once `resolve_positional_order_by` has
13279 // run, and what a repeated `ORDER BY unnest(…)` is.
13280 projection.iter().position(|p| p.expr == ob.expr)
13281 })
13282 .collect()
13283}
13284
13285fn expand_srf_row(
13286 engine: &Engine,
13287 projection: &[ProjectedItem],
13288 srf_idxs: &[usize],
13289 row: &Row<'static>,
13290 ctx: &EvalContext<'_>,
13291) -> Result<Vec<Row<'static>>, EngineError> {
13292 let mut plan = build_srf_plan(engine, projection, srf_idxs, ctx)?;
13293 expand_srf_row_with(engine, &mut plan, projection, row, ctx)
13294}
13295
13296impl Engine {
13297 /// The rows one target-list SRF yields for an input row. `None` from
13298 /// `srf_target_idxs` means the expression is not set-returning at all.
13299 fn srf_values(
13300 &self,
13301 expr: &spg_sql::ast::Expr,
13302 row: &Row<'static>,
13303 ctx: &EvalContext<'_>,
13304 ) -> Result<Vec<Value<'static>>, EngineError> {
13305 if top_level_srf_kind(expr).is_some() {
13306 return top_level_srf_output(expr, row, ctx);
13307 }
13308 // A user set-returning function. Its body runs through the real
13309 // executor, like every function body since round 63.
13310 let spg_sql::ast::Expr::FunctionCall { name, args } = expr else {
13311 return Err(EngineError::Unsupported(
13312 "expected a SELECT-list SRF call".into(),
13313 ));
13314 };
13315 let mut vals: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
13316 for a in args {
13317 vals.push(eval::eval_expr(a, row, ctx).map_err(EngineError::Eval)?);
13318 }
13319 let (rows, cols) = self.setof_rows_of(name, &vals, None)?;
13320 // v7.39 (read01 round 68) — in a target list a multi-column function is
13321 // a RECORD, one composite value per row: `SELECT rows_of(2)` gives
13322 // `(2,b)`, `(3,c)`. Value::Composite has existed since round 56; this is
13323 // what it is for. A single-column function contributes its bare value.
13324 Ok(rows
13325 .into_iter()
13326 .map(|r| {
13327 if r.values.len() == 1 {
13328 r.values.into_iter().next().unwrap_or(Value::Null)
13329 } else {
13330 Value::Composite(
13331 cols.iter()
13332 .map(|c| c.name.clone())
13333 .zip(r.values)
13334 .collect::<alloc::vec::Vec<_>>(),
13335 )
13336 }
13337 })
13338 .collect())
13339 }
13340
13341 /// Is THIS node a set-returning call: one of the builtin kinds, or a user
13342 /// function declared `RETURNS SETOF` / `RETURNS TABLE`.
13343 fn is_srf_node(&self, e: &spg_sql::ast::Expr) -> bool {
13344 if is_top_level_unnest(e) {
13345 return true;
13346 }
13347 let spg_sql::ast::Expr::FunctionCall { name, .. } = e else {
13348 return false;
13349 };
13350 self.active_catalog().functions_named(name).iter().any(|f| {
13351 let r = f.returns.trim().to_ascii_uppercase();
13352 r.starts_with("SETOF") || r.starts_with("TABLE(")
13353 })
13354 }
13355
13356 /// Does an SRF appear ANYWHERE in this expression (not only as its root)?
13357 fn expr_contains_srf(&self, e: &spg_sql::ast::Expr) -> bool {
13358 let mut found = false;
13359 let mut probe = e.clone();
13360 crate::expr_analysis::rewrite_nodes_mut(&mut probe, &mut |n| {
13361 if self.is_srf_node(n) {
13362 found = true;
13363 return true;
13364 }
13365 false
13366 });
13367 found
13368 }
13369
13370 /// Which projection items CONTAIN a set-returning call. Before round 78 this
13371 /// asked whether the item WAS one, so `upper(unnest(a))` looked like an
13372 /// ordinary scalar call all the way down to the function dispatcher, which
13373 /// then reported `unnest` as an unknown function.
13374 fn srf_target_idxs(&self, projection: &[ProjectedItem]) -> alloc::vec::Vec<usize> {
13375 projection
13376 .iter()
13377 .enumerate()
13378 .filter(|(_, p)| self.expr_contains_srf(&p.expr))
13379 .map(|(i, _)| i)
13380 .collect()
13381 }
13382}
13383
13384impl Engine {
13385 /// v7.39 (read01 round 74) — see the call site. `None` when the statement has
13386 /// no `(f(args)).*` item.
13387 fn lower_record_expansion(
13388 &self,
13389 stmt: &SelectStatement,
13390 ) -> Result<Option<SelectStatement>, EngineError> {
13391 use spg_sql::ast::{Expr, SelectItem};
13392 let is_marker = |it: &SelectItem| {
13393 matches!(it, SelectItem::Expr { expr: Expr::FunctionCall { name, .. }, .. }
13394 if name == "__record_expand")
13395 };
13396 if !stmt.items.iter().any(is_marker) {
13397 return Ok(None);
13398 }
13399 let mut out = stmt.clone();
13400 let mut items: alloc::vec::Vec<SelectItem> = alloc::vec::Vec::new();
13401 let mut lateral_refs: alloc::vec::Vec<TableRef> = alloc::vec::Vec::new();
13402 for (n, item) in stmt.items.iter().enumerate() {
13403 if !is_marker(item) {
13404 items.push(item.clone());
13405 continue;
13406 }
13407 let SelectItem::Expr {
13408 expr: Expr::FunctionCall { args, .. },
13409 ..
13410 } = item
13411 else {
13412 unreachable!("checked by is_marker");
13413 };
13414 let Some(Expr::FunctionCall {
13415 name: fname,
13416 args: fargs,
13417 }) = args.first()
13418 else {
13419 return Err(EngineError::Unsupported(
13420 "(<expr>).* expands a function's record — it needs a function call".into(),
13421 ));
13422 };
13423 let cols = self.setof_declared_columns(fname)?;
13424 let alias = alloc::format!("__rec{n}");
13425 let mut tref = bare_table_ref_named(&alias);
13426 tref.table_fn_call = Some(alloc::boxed::Box::new((
13427 fname.to_ascii_lowercase(),
13428 fargs.clone(),
13429 )));
13430 tref.alias = Some(alias.clone());
13431 lateral_refs.push(tref);
13432 for c in cols {
13433 items.push(SelectItem::Expr {
13434 expr: Expr::Column(spg_sql::ast::ColumnName {
13435 qualifier: Some(alias.clone()),
13436 name: c,
13437 }),
13438 alias: None,
13439 });
13440 }
13441 }
13442 out.items = items;
13443 // The function joins the FROM. With no FROM it BECOMES the FROM; with one
13444 // it is a cross join, which is what `SELECT …, (f(t.c)).* FROM t` means
13445 // (the arguments may reference the outer row — the round-69 correlation).
13446 for tref in lateral_refs {
13447 match &mut out.from {
13448 None => {
13449 out.from = Some(spg_sql::ast::FromClause {
13450 primary: tref,
13451 joins: alloc::vec::Vec::new(),
13452 });
13453 }
13454 Some(from) => from.joins.push(spg_sql::ast::FromJoin {
13455 kind: spg_sql::ast::JoinKind::Cross,
13456 table: tref,
13457 on: None,
13458 using_cols: None,
13459 natural: false,
13460 }),
13461 }
13462 }
13463 Ok(Some(out))
13464 }
13465
13466 /// The column NAMES a set-returning function declares: `RETURNS TABLE(id int,
13467 /// v text)` names them; a `SETOF <scalar>` is one column named after the
13468 /// function.
13469 fn setof_declared_columns(
13470 &self,
13471 name: &str,
13472 ) -> Result<alloc::vec::Vec<alloc::string::String>, EngineError> {
13473 let cat = self.active_catalog();
13474 let overloads = cat.functions_named(name);
13475 let def = overloads.first().ok_or_else(|| {
13476 EngineError::Unsupported(alloc::format!("function {name} does not exist"))
13477 })?;
13478 let declared = def.returns.trim();
13479 let upper = declared.to_ascii_uppercase();
13480 if upper.starts_with("TABLE(") {
13481 let raw = &declared["TABLE(".len()..declared.len() - 1];
13482 return Ok(raw
13483 .split(',')
13484 .map(|d| d.split_whitespace().next().unwrap_or("col").to_string())
13485 .collect());
13486 }
13487 Ok(alloc::vec![name.to_string()])
13488 }
13489}
13490
13491/// A bare `TableRef` with a name — the FROM item a lowered record expansion adds.
13492/// v7.39 (round 205, JSON_TABLE) — the static output schema of a
13493/// COLUMNS list (data-independent), NESTED children inlined in
13494/// declaration order (PG's flattened output shape).
13495/// v7.39 (round 205) — pub(crate) shim so join.rs infers a wrapped
13496/// correlated JSON_TABLE's static schema without evaluating its doc.
13497pub(crate) fn json_table_schema_pub(
13498 cols: &[spg_sql::ast::JsonTableColumn],
13499) -> alloc::vec::Vec<ColumnSchema> {
13500 json_table_schema(cols)
13501}
13502
13503fn json_table_schema(cols: &[spg_sql::ast::JsonTableColumn]) -> alloc::vec::Vec<ColumnSchema> {
13504 use spg_sql::ast::JsonTableColumn as C;
13505 let mut out = alloc::vec::Vec::new();
13506 for c in cols {
13507 match c {
13508 C::Ordinality { name } => {
13509 out.push(ColumnSchema::new(name.clone(), DataType::BigInt, false));
13510 }
13511 C::Regular {
13512 name, ty, exists, ..
13513 } => {
13514 let dt = if *exists {
13515 DataType::Bool
13516 } else {
13517 crate::conversions::column_type_to_data_type(*ty)
13518 };
13519 out.push(ColumnSchema::new(name.clone(), dt, true));
13520 }
13521 C::Nested { columns, .. } => out.extend(json_table_schema(columns)),
13522 }
13523 }
13524 out
13525}
13526
13527/// v7.39 (round 205) — coerce a DEFAULT / literal value to a
13528/// JSON_TABLE column's declared type (the DEFAULT expr may be a
13529/// string literal like `'none'` that must land as the column type).
13530fn coerce_json_table_default(
13531 v: Value<'static>,
13532 ty: spg_sql::ast::ColumnTypeName,
13533 name: &str,
13534) -> Result<Value<'static>, EngineError> {
13535 if v.is_null() {
13536 return Ok(Value::Null);
13537 }
13538 let dt = crate::conversions::column_type_to_data_type(ty);
13539 crate::conversions::coerce_value(v, dt, name, 0)
13540}
13541
13542/// v7.39 (round 205) — a runtime Value → JsonValue for PASSING vars.
13543fn value_to_json_value(v: &Value<'_>) -> crate::json::JsonValue {
13544 use crate::json::JsonValue as J;
13545 match v {
13546 Value::Null => J::Null,
13547 Value::Bool(b) => J::Bool(*b),
13548 Value::SmallInt(n) => J::Number(f64::from(*n)),
13549 Value::Int(n) => J::Number(f64::from(*n)),
13550 Value::BigInt(n) => J::Number(*n as f64),
13551 Value::Float(x) => J::Number(*x),
13552 Value::Json(s) => crate::json::parse_doc(s).unwrap_or(J::Null),
13553 other => J::String(crate::eval::value_to_text(other)),
13554 }
13555}
13556
13557fn bare_table_ref_named(name: &str) -> TableRef {
13558 TableRef {
13559 name: name.to_string(),
13560 alias: None,
13561 only: false,
13562 as_of_segment: None,
13563 unnest_expr: None,
13564 unnest_column_aliases: alloc::vec::Vec::new(),
13565 with_ordinality: false,
13566 generate_series_args: None,
13567 lateral_subquery: None,
13568 jsonb_each_text_arg: None,
13569 table_fn_call: None,
13570 rows_from: None,
13571 json_table: None,
13572 scalar_fn_item: false,
13573 }
13574}
13575
13576impl Engine {
13577 /// v7.39 (read01 round 74) — run a `ROWS FROM (…)` list. Each entry yields its
13578 /// own rows; they zip in lockstep and a short one pads with NULL. `__array`
13579 /// entries are the array-able SRFs, already lowered by the parser into their
13580 /// scalar array form.
13581 fn rows_from_rows(
13582 &self,
13583 primary: &TableRef,
13584 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
13585 let entries = primary
13586 .rows_from
13587 .as_ref()
13588 .expect("caller guards rows_from.is_some()");
13589 let empty: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
13590 let ctx = self.ev_ctx(&empty, None);
13591 let dummy = Row::new(alloc::vec::Vec::new());
13592 let mut lists: alloc::vec::Vec<alloc::vec::Vec<Value<'static>>> = alloc::vec::Vec::new();
13593 let mut cols: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
13594 for (name, args) in entries {
13595 let (vals, colname) = if name == "__array" {
13596 // The parser lowered this one to `<array expr>`; its rows are the
13597 // array's elements.
13598 let arr = eval::eval_expr(&args[0], &dummy, &ctx).map_err(EngineError::Eval)?;
13599 (
13600 array_value_to_elements(&arr)?,
13601 alloc::string::String::from("unnest"),
13602 )
13603 } else {
13604 let call = spg_sql::ast::Expr::FunctionCall {
13605 name: name.clone(),
13606 args: args.clone(),
13607 };
13608 (self.srf_values(&call, &dummy, &ctx)?, name.clone())
13609 };
13610 let ty = vals
13611 .first()
13612 .and_then(spg_storage::Value::data_type)
13613 .unwrap_or(DataType::Text);
13614 cols.push(ColumnSchema::new(colname, ty, true));
13615 lists.push(vals);
13616 }
13617 let n = lists.iter().map(alloc::vec::Vec::len).max().unwrap_or(0);
13618 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::with_capacity(n);
13619 for k in 0..n {
13620 let mut vals: alloc::vec::Vec<Value<'static>> =
13621 alloc::vec::Vec::with_capacity(lists.len() + 1);
13622 for l in &lists {
13623 vals.push(l.get(k).cloned().unwrap_or(Value::Null));
13624 }
13625 rows.push(Row::new(vals));
13626 }
13627 if primary.with_ordinality {
13628 cols.push(ColumnSchema::new(
13629 "ordinality".to_string(),
13630 DataType::BigInt,
13631 false,
13632 ));
13633 rows = rows
13634 .into_iter()
13635 .enumerate()
13636 .map(|(i, r)| {
13637 let mut v = r.values;
13638 v.push(Value::BigInt(i as i64 + 1));
13639 Row::new(v)
13640 })
13641 .collect();
13642 }
13643 Ok((rows, cols))
13644 }
13645}
13646
13647/// v7.39 (round 232) — PG names the offending set operation in its
13648/// arity / type-mismatch messages ("each UNION query must have the same
13649/// number of columns"). `UNION ALL` is still spelled UNION there.
13650fn set_op_name(kind: UnionKind) -> &'static str {
13651 match kind {
13652 UnionKind::All | UnionKind::Distinct => "UNION",
13653 UnionKind::Intersect | UnionKind::IntersectAll => "INTERSECT",
13654 UnionKind::Except | UnionKind::ExceptAll => "EXCEPT",
13655 }
13656}
13657
13658/// v7.39 (round 233) — which output columns of a branch are PG's `unknown`
13659/// type: a bare string or NULL literal that no context has typed yet. SPG
13660/// has no `Unknown` DataType (both describe as TEXT), so the witness has to
13661/// be the syntax. A wildcard or a non-literal expression is never unknown.
13662/// 7.38.1 S5.1 — is this branch item a reg* cast? Its result column
13663/// LABELS as text (the wire render) but the value is an oid-carrying
13664/// dual, so a UNION with a numeric column must not be refused on the
13665/// label (pg_dump: `SELECT classid … UNION ALL SELECT
13666/// 'pg_opfamily'::regclass …`).
13667fn branch_regcast_mask(stmt: &SelectStatement) -> Vec<bool> {
13668 fn is_regcast(e: &Expr) -> bool {
13669 matches!(
13670 e,
13671 Expr::Cast {
13672 target: spg_sql::ast::CastTarget::RegType | spg_sql::ast::CastTarget::RegClass,
13673 ..
13674 }
13675 )
13676 }
13677 stmt.items
13678 .iter()
13679 .map(|item| match item {
13680 SelectItem::Expr { expr, .. } => is_regcast(expr),
13681 _ => false,
13682 })
13683 .collect()
13684}
13685
13686fn branch_unknown_mask(stmt: &SelectStatement) -> Vec<bool> {
13687 stmt.items
13688 .iter()
13689 .map(|item| match item {
13690 SelectItem::Expr { expr, .. } => matches!(
13691 expr,
13692 Expr::Literal(spg_sql::ast::Literal::String(_))
13693 | Expr::Literal(spg_sql::ast::Literal::Null)
13694 ),
13695 _ => false,
13696 })
13697 .collect()
13698}
13699
13700/// v7.39 (round 233) — retype one branch column's cells, reporting the
13701/// conversion failure the way PG does rather than leaving the column
13702/// half-converted. Used when the other branch typed an untyped literal.
13703fn coerce_branch_column(
13704 rows: &mut [Row<'static>],
13705 col_idx: usize,
13706 target: DataType,
13707 col_name: &str,
13708) -> Result<(), EngineError> {
13709 for row in rows.iter_mut() {
13710 let Some(slot) = row.values.get_mut(col_idx) else {
13711 continue;
13712 };
13713 if matches!(slot, Value::Null) {
13714 continue;
13715 }
13716 *slot = crate::conversions::coerce_value(slot.clone(), target, col_name, col_idx)?;
13717 }
13718 Ok(())
13719}
13720
13721/// v7.39 (round 727) — PG-style pull-up of a SIMPLE derived table:
13722/// `SELECT … FROM (SELECT <bare columns> FROM t [WHERE …]) q …`
13723/// rewrites to `SELECT …' FROM t [WHERE inner AND outer'] …` with every
13724/// reference to q's output columns substituted by the underlying column.
13725///
13726/// Admission is deliberately narrow — anything that changes cardinality,
13727/// order, or scope stays on the materialising path:
13728/// * outer: no CTEs / unions / DISTINCT [ON] / windows, single derived
13729/// FROM with no ordinality or positional column aliases, and no
13730/// subquery anywhere its expressions (an inner scope could reference
13731/// q too — descending is a later knife);
13732/// * inner: one stored table, bare-column projection only, no
13733/// CTE/union/DISTINCT/GROUP/HAVING/ORDER/LIMIT/OFFSET/windows/locking;
13734/// * every outer column reference must resolve inside q's output list —
13735/// a name that does not is an ERROR today, and flattening would
13736/// silently legalise it against the base table.
13737fn try_flatten_derived(stmt: &SelectStatement, primary: &TableRef) -> Option<SelectStatement> {
13738 use spg_sql::ast::SelectItem;
13739 let inner = primary.lateral_subquery.as_deref()?;
13740 // Outer shape.
13741 if !stmt.ctes.is_empty()
13742 || !stmt.unions.is_empty()
13743 || stmt.distinct
13744 || !stmt.distinct_on.is_empty()
13745 || !stmt.window_check_exprs.is_empty()
13746 || stmt.locking.is_some()
13747 || primary.with_ordinality
13748 || !primary.unnest_column_aliases.is_empty()
13749 {
13750 return None;
13751 }
13752 // Inner shape.
13753 if !inner.ctes.is_empty()
13754 || !inner.unions.is_empty()
13755 || inner.distinct
13756 || !inner.distinct_on.is_empty()
13757 || inner.group_by.is_some()
13758 || inner.group_by_all
13759 || inner.having.is_some()
13760 || !inner.order_by.is_empty()
13761 || inner.limit.is_some()
13762 || inner.offset.is_some()
13763 || !inner.window_check_exprs.is_empty()
13764 || inner.locking.is_some()
13765 {
13766 return None;
13767 }
13768 let ifrom = inner.from.as_ref()?;
13769 let it = &ifrom.primary;
13770 if !ifrom.joins.is_empty()
13771 || it.name.is_empty()
13772 || it.lateral_subquery.is_some()
13773 || it.unnest_expr.is_some()
13774 || it.generate_series_args.is_some()
13775 || it.as_of_segment.is_some()
13776 || it.jsonb_each_text_arg.is_some()
13777 || it.table_fn_call.is_some()
13778 || it.rows_from.is_some()
13779 || it.json_table.is_some()
13780 || it.with_ordinality
13781 || !it.unnest_column_aliases.is_empty()
13782 {
13783 return None;
13784 }
13785 if inner.where_.as_ref().is_some_and(crate::expr_has_subquery) {
13786 return None;
13787 }
13788 // The output map: q's visible name -> the underlying column.
13789 let inner_alias = it.alias.clone().unwrap_or_else(|| it.name.clone());
13790 let mut map: alloc::collections::BTreeMap<String, spg_sql::ast::ColumnName> =
13791 alloc::collections::BTreeMap::new();
13792 for item in &inner.items {
13793 let SelectItem::Expr { expr, alias } = item else {
13794 return None;
13795 };
13796 let Expr::Column(c) = expr else {
13797 return None;
13798 };
13799 if let Some(q) = c.qualifier.as_deref()
13800 && !q.eq_ignore_ascii_case(&inner_alias)
13801 {
13802 return None;
13803 }
13804 let out_name = alias.clone().unwrap_or_else(|| c.name.clone());
13805 // A duplicated output name would make substitution ambiguous.
13806 if map
13807 .insert(out_name.to_ascii_lowercase(), c.clone())
13808 .is_some()
13809 {
13810 return None;
13811 }
13812 }
13813 if map.is_empty() {
13814 return None;
13815 }
13816 let derived_alias = primary
13817 .alias
13818 .clone()
13819 .unwrap_or_else(|| primary.name.clone())
13820 .to_ascii_lowercase();
13821 // Substitute in a clone; bail (None) on the first reference the map
13822 // cannot answer.
13823 let mut out = stmt.clone();
13824 let ok = core::cell::Cell::new(true);
13825 let mut subst = |e: &mut Expr| -> bool {
13826 match e {
13827 Expr::Column(c) => {
13828 match c.qualifier.as_deref() {
13829 Some(q) if q.eq_ignore_ascii_case(&derived_alias) => {}
13830 None => {}
13831 Some(_) => {
13832 ok.set(false);
13833 return true;
13834 }
13835 }
13836 match map.get(&c.name.to_ascii_lowercase()) {
13837 Some(target) => *c = target.clone(),
13838 None => ok.set(false),
13839 }
13840 true
13841 }
13842 // Any subquery could reference q from its own scope;
13843 // descending is a later knife — bail for now.
13844 Expr::ScalarSubquery(_)
13845 | Expr::Exists { .. }
13846 | Expr::InSubquery { .. }
13847 | Expr::RowInSubquery { .. }
13848 | Expr::RowCmpSubquery { .. } => {
13849 ok.set(false);
13850 true
13851 }
13852 _ => false,
13853 }
13854 };
13855 for item in &mut out.items {
13856 match item {
13857 SelectItem::Expr { expr, .. } => {
13858 crate::expr_analysis::rewrite_nodes_mut(expr, &mut subst);
13859 }
13860 // `SELECT * FROM (…) q` means q's columns, in q's order.
13861 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => return None,
13862 }
13863 }
13864 if let Some(w) = &mut out.where_ {
13865 crate::expr_analysis::rewrite_nodes_mut(w, &mut subst);
13866 }
13867 if let Some(gs) = &mut out.group_by {
13868 for g in gs {
13869 crate::expr_analysis::rewrite_nodes_mut(g, &mut subst);
13870 }
13871 }
13872 if let Some(h) = &mut out.having {
13873 crate::expr_analysis::rewrite_nodes_mut(h, &mut subst);
13874 }
13875 for o in &mut out.order_by {
13876 crate::expr_analysis::rewrite_nodes_mut(&mut o.expr, &mut subst);
13877 }
13878 for d in &mut out.distinct_on {
13879 crate::expr_analysis::rewrite_nodes_mut(d, &mut subst);
13880 }
13881 if !ok.get() {
13882 return None;
13883 }
13884 // FROM becomes the stored table; the filters conjoin.
13885 out.from = Some(spg_sql::ast::FromClause {
13886 primary: it.clone(),
13887 joins: Vec::new(),
13888 });
13889 out.where_ = match (inner.where_.clone(), out.where_.take()) {
13890 (Some(a), Some(b)) => Some(Expr::Binary {
13891 lhs: alloc::boxed::Box::new(a),
13892 op: spg_sql::ast::BinOp::And,
13893 rhs: alloc::boxed::Box::new(b),
13894 }),
13895 (Some(a), None) => Some(a),
13896 (None, b) => b,
13897 };
13898 Some(out)
13899}
13900
13901/// v7.39 (round 742) — rewrite `SELECT count(*) FROM (SELECT <plain>
13902/// FROM t [WHERE p] ORDER BY … OFFSET k [no LIMIT]) q` into
13903/// `SELECT greatest(count(*) - k, 0) FROM t [WHERE p]`. Sound because
13904/// ORDER BY is count-invariant and OFFSET k drops exactly min(k, n)
13905/// rows. Admission mirrors the flatten's conservatism; a LIMIT, a
13906/// DISTINCT, an SRF, or an unprovable inner shape stays put.
13907fn try_count_over_offset(stmt: &SelectStatement, primary: &TableRef) -> Option<SelectStatement> {
13908 use spg_sql::ast::{Expr as E, LimitExpr, SelectItem};
13909 let inner = primary.lateral_subquery.as_deref()?;
13910 // Outer: exactly `SELECT count(*)`, nothing else.
13911 if !stmt.ctes.is_empty()
13912 || !stmt.unions.is_empty()
13913 || stmt.distinct
13914 || !stmt.distinct_on.is_empty()
13915 || stmt.where_.is_some()
13916 || stmt.group_by.is_some()
13917 || stmt.having.is_some()
13918 || !stmt.order_by.is_empty()
13919 || stmt.limit.is_some()
13920 || stmt.offset.is_some()
13921 || stmt.items.len() != 1
13922 {
13923 return None;
13924 }
13925 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
13926 return None;
13927 };
13928 let E::FunctionCall { name, args } = expr else {
13929 return None;
13930 };
13931 if !name.eq_ignore_ascii_case("count_star") || !args.is_empty() {
13932 return None;
13933 }
13934 // Inner: flatten-shaped plus ORDER BY and a literal OFFSET, no LIMIT.
13935 let Some(LimitExpr::Literal(k)) = &inner.offset else {
13936 return None;
13937 };
13938 let k = i64::from(*k);
13939 if inner.limit.is_some() || inner.order_by.is_empty() {
13940 return None;
13941 }
13942 let mut counted = inner.clone();
13943 counted.order_by = Vec::new();
13944 counted.offset = None;
13945 // The stripped inner must now be a provable simple shape (its
13946 // items become irrelevant — count(*) reads none of them — but an
13947 // SRF item would change the row count, so the flatten predicate's
13948 // scrutiny still applies).
13949 let base = matview_flatten_probe(&counted)?;
13950 let mut out = stmt.clone();
13951 out.items = alloc::vec![SelectItem::Expr {
13952 expr: E::FunctionCall {
13953 name: String::from("greatest"),
13954 args: alloc::vec![
13955 E::Binary {
13956 lhs: alloc::boxed::Box::new(E::FunctionCall {
13957 name: String::from("count_star"),
13958 args: alloc::vec![],
13959 }),
13960 op: spg_sql::ast::BinOp::Sub,
13961 rhs: alloc::boxed::Box::new(E::Literal(spg_sql::ast::Literal::Integer(k))),
13962 },
13963 E::Literal(spg_sql::ast::Literal::Integer(0)),
13964 ],
13965 },
13966 alias: Some(String::from("count")),
13967 }];
13968 out.from = Some(spg_sql::ast::FromClause {
13969 primary: base,
13970 joins: Vec::new(),
13971 });
13972 out.where_ = counted.where_.clone();
13973 Some(out)
13974}
13975
13976/// The inner-shape probe `try_count_over_offset` shares with the
13977/// flatten: single stored table, no modifiers, no subqueries, no SRF
13978/// items. Returns the base TableRef.
13979fn matview_flatten_probe(inner: &SelectStatement) -> Option<TableRef> {
13980 use spg_sql::ast::SelectItem;
13981 if !inner.ctes.is_empty()
13982 || !inner.unions.is_empty()
13983 || inner.distinct
13984 || !inner.distinct_on.is_empty()
13985 || inner.group_by.is_some()
13986 || inner.group_by_all
13987 || inner.having.is_some()
13988 || !inner.order_by.is_empty()
13989 || inner.limit.is_some()
13990 || inner.offset.is_some()
13991 || !inner.window_check_exprs.is_empty()
13992 || inner.locking.is_some()
13993 {
13994 return None;
13995 }
13996 let ifrom = inner.from.as_ref()?;
13997 let it = &ifrom.primary;
13998 if !ifrom.joins.is_empty()
13999 || it.name.is_empty()
14000 || it.lateral_subquery.is_some()
14001 || it.unnest_expr.is_some()
14002 || it.generate_series_args.is_some()
14003 || it.as_of_segment.is_some()
14004 || it.jsonb_each_text_arg.is_some()
14005 || it.table_fn_call.is_some()
14006 || it.rows_from.is_some()
14007 || it.json_table.is_some()
14008 || it.with_ordinality
14009 {
14010 return None;
14011 }
14012 for item in &inner.items {
14013 match item {
14014 SelectItem::Expr { expr, .. } => {
14015 if crate::expr_has_subquery(expr) || expr_contains_builtin_srf(expr) {
14016 return None;
14017 }
14018 }
14019 SelectItem::Wildcard => {}
14020 SelectItem::QualifiedWildcard(_) => return None,
14021 }
14022 }
14023 if inner.where_.as_ref().is_some_and(crate::expr_has_subquery) {
14024 return None;
14025 }
14026 Some(it.clone())
14027}
14028
14029/// v7.39 (round 743) — rewrite `SELECT count(*) FROM (SELECT
14030/// unnest(ARRAY[e1..ek]) [AS v] FROM t [WHERE p]) q` into
14031/// `SELECT count(*) * k FROM t [WHERE p]`. Sound because a
14032/// constant-LENGTH array literal unnests to exactly k rows per input
14033/// row (NULL elements are rows too). One SRF item only, elements
14034/// subquery-free, and the stripped inner must pass the same probe the
14035/// count-over-offset rewrite uses.
14036fn try_count_over_const_unnest(
14037 stmt: &SelectStatement,
14038 primary: &TableRef,
14039) -> Option<SelectStatement> {
14040 use spg_sql::ast::{Expr as E, SelectItem};
14041 let inner = primary.lateral_subquery.as_deref()?;
14042 if !stmt.ctes.is_empty()
14043 || !stmt.unions.is_empty()
14044 || stmt.distinct
14045 || !stmt.distinct_on.is_empty()
14046 || stmt.where_.is_some()
14047 || stmt.group_by.is_some()
14048 || stmt.having.is_some()
14049 || !stmt.order_by.is_empty()
14050 || stmt.limit.is_some()
14051 || stmt.offset.is_some()
14052 || stmt.items.len() != 1
14053 {
14054 return None;
14055 }
14056 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
14057 return None;
14058 };
14059 let E::FunctionCall { name, args } = expr else {
14060 return None;
14061 };
14062 if !name.eq_ignore_ascii_case("count_star") || !args.is_empty() {
14063 return None;
14064 }
14065 // Inner: exactly one item, and it is unnest(ARRAY[...]).
14066 if inner.items.len() != 1
14067 || !inner.order_by.is_empty()
14068 || inner.limit.is_some()
14069 || inner.offset.is_some()
14070 {
14071 return None;
14072 }
14073 let SelectItem::Expr { expr: item, .. } = &inner.items[0] else {
14074 return None;
14075 };
14076 let E::FunctionCall {
14077 name: fname,
14078 args: fargs,
14079 } = item
14080 else {
14081 return None;
14082 };
14083 if !fname.eq_ignore_ascii_case("unnest") || fargs.len() != 1 {
14084 return None;
14085 }
14086 let E::Array(elems) = &fargs[0] else {
14087 return None;
14088 };
14089 if elems.is_empty() || elems.iter().any(crate::expr_has_subquery) {
14090 return None;
14091 }
14092 let k = elems.len() as i64;
14093 // The stripped inner (the SRF item replaced by a plain constant)
14094 // must be the provable simple shape.
14095 let mut counted = inner.clone();
14096 counted.items = alloc::vec![SelectItem::Expr {
14097 expr: E::Literal(spg_sql::ast::Literal::Integer(1)),
14098 alias: None,
14099 }];
14100 let base = matview_flatten_probe(&counted)?;
14101 let mut out = stmt.clone();
14102 out.items = alloc::vec![SelectItem::Expr {
14103 expr: E::Binary {
14104 lhs: alloc::boxed::Box::new(E::FunctionCall {
14105 name: String::from("count_star"),
14106 args: alloc::vec![],
14107 }),
14108 op: spg_sql::ast::BinOp::Mul,
14109 rhs: alloc::boxed::Box::new(E::Literal(spg_sql::ast::Literal::Integer(k))),
14110 },
14111 alias: Some(String::from("count")),
14112 }];
14113 out.from = Some(spg_sql::ast::FromClause {
14114 primary: base,
14115 joins: Vec::new(),
14116 });
14117 out.where_ = counted.where_.clone();
14118 Some(out)
14119}